Pick any line to see why it is the way it is: the commit, the pull request and issue it came from, and what the agent was thinking.
| Docs index by meaning: passages of every page and project doc, embedded on save and recalled for agents; hybrid search for people | 1 | import assert from "node:assert/strict"; |
| 2 | import { readFileSync, readdirSync } from "node:fs"; | |
| 3 | import { DatabaseSync } from "node:sqlite"; | |
| 4 | import { test } from "node:test"; | |
| 5 | ||
| 6 | import { EMBED_PER_HOUR, indexDoc, type IndexDoc } from "./indexer.ts"; | |
| 7 | import type { Embedder, VectorMetadata, VectorStore } from "./vectors.ts"; | |
| 8 | ||
| 9 | /** D1, as far as the indexer uses it, over node's SQLite with the service's migrations. */ | |
| 10 | function fakeD1(): D1Database { | |
| 11 | const db = new DatabaseSync(":memory:"); | |
| 12 | const dir = new URL("../migrations/", import.meta.url); | |
| 13 | for (const file of readdirSync(dir).sort()) db.exec(readFileSync(new URL(file, dir), "utf8")); | |
| 14 | const statement = (sql: string, params: unknown[] = []) => ({ | |
| 15 | sql, | |
| 16 | params, | |
| 17 | bind: (...values: unknown[]) => statement(sql, values), | |
| 18 | first: async () => (db.prepare(sql).get(...(params as never[])) as unknown) ?? null, | |
| 19 | all: async () => ({ results: db.prepare(sql).all(...(params as never[])) }), | |
| 20 | run: async () => db.prepare(sql).run(...(params as never[])), | |
| 21 | }); | |
| 22 | return { | |
| 23 | prepare: (sql: string) => statement(sql), | |
| 24 | batch: async (list: ReturnType<typeof statement>[]) => list.map((s) => db.prepare(s.sql).run(...(s.params as never[]))), | |
| 25 | } as unknown as D1Database; | |
| 26 | } | |
| 27 | ||
| 28 | function fakes() { | |
| 29 | const vectors = new Map<string, { values: number[]; metadata: VectorMetadata }>(); | |
| 30 | let embedded = 0; | |
| 31 | const embedder: Embedder = { | |
| 32 | async embed(texts) { | |
| 33 | embedded += texts.length; | |
| 34 | return texts.map((t) => [t.length, 1]); | |
| 35 | }, | |
| 36 | }; | |
| 37 | const store: VectorStore = { | |
| 38 | async upsert(list) { | |
| 39 | for (const v of list) vectors.set(v.id, { values: v.values, metadata: v.metadata }); | |
| 40 | }, | |
| 41 | async get(ids) { | |
| 42 | return ids.filter((id) => vectors.has(id)).map((id) => ({ id, values: vectors.get(id)!.values })); | |
| 43 | }, | |
| 44 | async delete(ids) { | |
| 45 | for (const id of ids) vectors.delete(id); | |
| 46 | }, | |
| 47 | async query() { | |
| 48 | return []; | |
| 49 | }, | |
| 50 | }; | |
| 51 | return { vectors, embedder, store, embeddedCount: () => embedded }; | |
| 52 | } | |
| 53 | ||
| 54 | const section = (name: string) => `## ${name}\n\n${Array.from({ length: 60 }, (_, i) => `${name.toLowerCase()}${i}`).join(" ")}.`; | |
| 55 | const page = (markdown: string, over: Partial<IndexDoc> = {}): IndexDoc => ({ kind: "page", doc_id: "pag_1", workspace_id: "w1", space_id: "s1", title: "Ops", markdown, ...over }); | |
| 56 | ||
| 57 | test("a page's passages are stored, embedded once, and only changed ones again", async () => { | |
| 58 | const db = fakeD1(); | |
| 59 | const f = fakes(); | |
| 60 | const first = await indexDoc(db, f.embedder, f.store, page([section("Deploy"), section("Rollback")].join("\n\n"))); | |
| 61 | assert.deepEqual(first, { chunks: 2, embedded: 2, capped: false }); | |
| 62 | assert.deepEqual([...f.vectors.keys()].sort(), ["pag_1:0", "pag_1:1"]); | |
| 63 | assert.deepEqual(f.vectors.get("pag_1:0")!.metadata, { workspace_id: "w1", space_id: "s1", kind: "page", page_id: "pag_1" }); | |
| 64 | // Saved again unchanged: nothing embedded. | |
| 65 | assert.equal((await indexDoc(db, f.embedder, f.store, page([section("Deploy"), section("Rollback")].join("\n\n")))).embedded, 0); | |
| 66 | // One section changed. | |
| 67 | assert.equal((await indexDoc(db, f.embedder, f.store, page([section("Deploy"), section("Restore")].join("\n\n")))).embedded, 1); | |
| 68 | assert.equal(f.embeddedCount(), 3); | |
| 69 | // Word search finds passages by heading and text. | |
| 70 | const hit = await db.prepare("SELECT chunk_id FROM doc_chunks_fts WHERE doc_chunks_fts MATCH ?").bind('"restore"').all<{ chunk_id: string }>(); | |
| 71 | assert.deepEqual( | |
| 72 | hit.results.map((r) => r.chunk_id), | |
| 73 | ["pag_1:1"], | |
| 74 | ); | |
| 75 | }); | |
| 76 | ||
| 77 | test("a passage that only moved keeps its vector; gone passages leave the index", async () => { | |
| 78 | const db = fakeD1(); | |
| 79 | const f = fakes(); | |
| 80 | await indexDoc(db, f.embedder, f.store, page([section("Alpha"), section("Beta"), section("Gamma")].join("\n\n"))); | |
| 81 | const before = f.embeddedCount(); | |
| 82 | // Alpha removed: Beta and Gamma move up a place, nothing new to embed. | |
| 83 | const r = await indexDoc(db, f.embedder, f.store, page([section("Beta"), section("Gamma")].join("\n\n"))); | |
| 84 | assert.equal(r.embedded, 0); | |
| 85 | assert.equal(f.embeddedCount(), before); | |
| 86 | assert.deepEqual([...f.vectors.keys()].sort(), ["pag_1:0", "pag_1:1"]); | |
| 87 | const rows = await db.prepare("SELECT id, heading, hash = vector_hash AS current FROM doc_chunks ORDER BY seq").all<{ id: string; heading: string; current: number }>(); | |
| 88 | assert.deepEqual( | |
| 89 | rows.results.map((x) => [x.id, x.heading, x.current]), | |
| 90 | [ | |
| 91 | ["pag_1:0", "Beta", 1], | |
| 92 | ["pag_1:1", "Gamma", 1], | |
| 93 | ], | |
| 94 | ); | |
| 95 | }); | |
| 96 | ||
| 97 | test("moving a page to another space files its vectors there without embedding", async () => { | |
| 98 | const db = fakeD1(); | |
| 99 | const f = fakes(); | |
| 100 | await indexDoc(db, f.embedder, f.store, page(section("Deploy"))); | |
| 101 | const r = await indexDoc(db, f.embedder, f.store, page(section("Deploy"), { space_id: "s2" })); | |
| 102 | assert.equal(r.embedded, 0); | |
| 103 | assert.equal(f.vectors.get("pag_1:0")!.metadata.space_id, "s2"); | |
| 104 | assert.equal((await db.prepare("SELECT space_id FROM doc_chunks WHERE id = 'pag_1:0'").first<{ space_id: string }>())!.space_id, "s2"); | |
| 105 | }); | |
| 106 | ||
| 107 | test("embedding stops at the hourly cap and the rest waits, kept for words", async () => { | |
| 108 | const db = fakeD1(); | |
| 109 | const f = fakes(); | |
| 110 | const at = new Date("2026-10-09T10:15:00Z"); | |
| 111 | await db.prepare("INSERT INTO doc_embed_usage (workspace_id, hour, chunks, tokens) VALUES ('w1', '2026-10-09T10', ?, 0)").bind(EMBED_PER_HOUR - 1).run(); | |
| 112 | const r = await indexDoc(db, f.embedder, f.store, page([section("One"), section("Two"), section("Three")].join("\n\n")), at); | |
| 113 | assert.deepEqual(r, { chunks: 3, embedded: 1, capped: true }); | |
| 114 | const waiting = await db.prepare("SELECT COUNT(*) AS n FROM doc_chunks WHERE vector_hash IS NULL OR vector_hash <> hash").first<{ n: number }>(); | |
| 115 | assert.equal(waiting!.n, 2); | |
| 116 | // The next hour, a save picks up the rest. | |
| 117 | const later = await indexDoc(db, f.embedder, f.store, page([section("One"), section("Two"), section("Three")].join("\n\n")), new Date("2026-10-09T11:01:00Z")); | |
| 118 | assert.deepEqual(later, { chunks: 3, embedded: 2, capped: false }); | |
| 119 | }); | |
| 120 | ||
| 121 | test("an embedding failure leaves passages for the next save, never throws", async () => { | |
| 122 | const db = fakeD1(); | |
| 123 | const f = fakes(); | |
| 124 | const broken: Embedder = { | |
| 125 | async embed() { | |
| 126 | throw new Error("model down"); | |
| 127 | }, | |
| 128 | }; | |
| 129 | const r = await indexDoc(db, broken, f.store, page(section("Deploy"))); | |
| 130 | assert.equal(r.chunks, 1); | |
| 131 | assert.equal(f.vectors.size, 0); | |
| 132 | assert.equal((await indexDoc(db, f.embedder, f.store, page(section("Deploy")))).embedded, 1); | |
| 133 | }); | |
| 134 | ||
| 135 | test("without an embedder, passages are kept for words only", async () => { | |
| 136 | const db = fakeD1(); | |
| 137 | const r = await indexDoc(db, null, null, page(section("Deploy"))); | |
| 138 | assert.deepEqual(r, { chunks: 1, embedded: 0, capped: false }); | |
| 139 | assert.equal((await db.prepare("SELECT COUNT(*) AS n FROM doc_chunks_fts").first<{ n: number }>())!.n, 1); | |
| 140 | }); | |
| 141 | ||
| 142 | test("a project's docs file is indexed with its repository", async () => { | |
| 143 | const db = fakeD1(); | |
| 144 | const f = fakes(); | |
| 145 | await indexDoc(db, f.embedder, f.store, { kind: "repo_file", doc_id: "rf_x", workspace_id: "w1", space_id: "rds_1", title: "Setup", markdown: `# Setup\n\n${section("Install")}`, repo_id: "r1", path: "docs/setup.md" }); | |
| 146 | assert.deepEqual(f.vectors.get("rf_x:0")!.metadata, { workspace_id: "w1", space_id: "rds_1", kind: "repo_file", repo_file_id: "rf_x", repo_id: "r1" }); | |
| 147 | const row = await db.prepare("SELECT page_id, repo_file_id, path, heading FROM doc_chunks").first<Record<string, unknown>>(); | |
| 148 | assert.deepEqual({ ...row }, { page_id: null, repo_file_id: "rf_x", path: "docs/setup.md", heading: "Install" }); | |
| 149 | }); |