| 1 | -- Passages: pages and projects' docs files split by heading, for the |
| 2 | -- semantic index (Vectorize `g1t-docs`) and agents' recall |
| 3 | -- (docs/WORKSPACE.md, "Agents and docs"; services/docs src/chunks.ts, |
| 4 | -- src/indexer.ts). |
| 5 | |
| 6 | -- One passage. `id` is `<page or file id>:<seq>`, the same as its |
| 7 | -- vector's. `hash` is of what is embedded (title, heading, text); |
| 8 | -- `vector_hash` is the hash the index holds a vector for, NULL when it |
| 9 | -- holds none yet: the two differ until the passage is embedded, which a |
| 10 | -- later save or the backfill retries. For a project's docs file, |
| 11 | -- `space_id` is the repo space's id, `repo_file_id` is `rf_<hash of space |
| 12 | -- and path>`, and `path` the file's. |
| 13 | CREATE TABLE doc_chunks ( |
| 14 | id TEXT PRIMARY KEY, |
| 15 | workspace_id TEXT NOT NULL, |
| 16 | space_id TEXT NOT NULL, |
| 17 | page_id TEXT, |
| 18 | repo_file_id TEXT, |
| 19 | repo_id TEXT, |
| 20 | path TEXT, |
| 21 | seq INTEGER NOT NULL, |
| 22 | heading TEXT, |
| 23 | text TEXT NOT NULL, |
| 24 | hash TEXT NOT NULL, |
| 25 | vector_hash TEXT, |
| 26 | updated_at TEXT NOT NULL, |
| 27 | CHECK ((page_id IS NULL) <> (repo_file_id IS NULL)) |
| 28 | ); |
| 29 | CREATE INDEX doc_chunks_page ON doc_chunks (page_id, seq) WHERE page_id IS NOT NULL; |
| 30 | CREATE INDEX doc_chunks_file ON doc_chunks (repo_file_id, seq) WHERE repo_file_id IS NOT NULL; |
| 31 | CREATE INDEX doc_chunks_space ON doc_chunks (space_id); |
| 32 | CREATE INDEX doc_chunks_workspace ON doc_chunks (workspace_id); |
| 33 | CREATE INDEX doc_chunks_unembedded ON doc_chunks (workspace_id) WHERE vector_hash IS NULL OR vector_hash <> hash; |
| 34 | |
| 35 | -- Full text over passages: recall's word fallback, and the heading a |
| 36 | -- search hit sits under. |
| 37 | CREATE VIRTUAL TABLE doc_chunks_fts USING fts5 (chunk_id UNINDEXED, space_id UNINDEXED, doc_id UNINDEXED, heading, text, tokenize = 'unicode61 remove_diacritics 2'); |
| 38 | |
| 39 | -- Passages embedded per workspace and hour: the cap (src/indexer.ts |
| 40 | -- EMBED_PER_HOUR) and what it cost (`tokens`, estimated as characters / 4). |
| 41 | CREATE TABLE doc_embed_usage ( |
| 42 | workspace_id TEXT NOT NULL, |
| 43 | hour TEXT NOT NULL, |
| 44 | chunks INTEGER NOT NULL DEFAULT 0, |
| 45 | tokens INTEGER NOT NULL DEFAULT 0, |
| 46 | PRIMARY KEY (workspace_id, hour) |
| 47 | ); |
| 48 | |
| 49 | -- A workspace's backfill: (re)indexing its existing pages and projects' |
| 50 | -- docs in batches on the queue. `cursor` is where the next batch starts. |
| 51 | CREATE TABLE doc_index_runs ( |
| 52 | workspace_id TEXT PRIMARY KEY, |
| 53 | started_at TEXT NOT NULL, |
| 54 | finished_at TEXT, |
| 55 | cursor TEXT, |
| 56 | pages INTEGER NOT NULL DEFAULT 0, |
| 57 | files INTEGER NOT NULL DEFAULT 0 |
| 58 | ); |