| 1 | import assert from "node:assert/strict"; |
| 2 | import { readFileSync, readdirSync } from "node:fs"; |
| 3 | import { DatabaseSync } from "node:sqlite"; |
| 4 | import { test } from "node:test"; |
| 5 | |
| 6 | import { EMBED_PER_HOUR, indexDoc, type IndexDoc } from "./indexer.ts"; |
| 7 | import type { Embedder, VectorMetadata, VectorStore } from "./vectors.ts"; |
| 8 | |
| 9 | /** D1, as far as the indexer uses it, over node's SQLite with the service's migrations. */ |
| 10 | function fakeD1(): D1Database { |
| 11 | const db = new DatabaseSync(":memory:"); |
| 12 | const dir = new URL("../migrations/", import.meta.url); |
| 13 | for (const file of readdirSync(dir).sort()) db.exec(readFileSync(new URL(file, dir), "utf8")); |
| 14 | const statement = (sql: string, params: unknown[] = []) => ({ |
| 15 | sql, |
| 16 | params, |
| 17 | bind: (...values: unknown[]) => statement(sql, values), |
| 18 | first: async () => (db.prepare(sql).get(...(params as never[])) as unknown) ?? null, |
| 19 | all: async () => ({ results: db.prepare(sql).all(...(params as never[])) }), |
| 20 | run: async () => db.prepare(sql).run(...(params as never[])), |
| 21 | }); |
| 22 | return { |
| 23 | prepare: (sql: string) => statement(sql), |
| 24 | batch: async (list: ReturnType<typeof statement>[]) => list.map((s) => db.prepare(s.sql).run(...(s.params as never[]))), |
| 25 | } as unknown as D1Database; |
| 26 | } |
| 27 | |
| 28 | function fakes() { |
| 29 | const vectors = new Map<string, { values: number[]; metadata: VectorMetadata }>(); |
| 30 | let embedded = 0; |
| 31 | const embedder: Embedder = { |
| 32 | async embed(texts) { |
| 33 | embedded += texts.length; |
| 34 | return texts.map((t) => [t.length, 1]); |
| 35 | }, |
| 36 | }; |
| 37 | const store: VectorStore = { |
| 38 | async upsert(list) { |
| 39 | for (const v of list) vectors.set(v.id, { values: v.values, metadata: v.metadata }); |
| 40 | }, |
| 41 | async get(ids) { |
| 42 | return ids.filter((id) => vectors.has(id)).map((id) => ({ id, values: vectors.get(id)!.values })); |
| 43 | }, |
| 44 | async delete(ids) { |
| 45 | for (const id of ids) vectors.delete(id); |
| 46 | }, |
| 47 | async query() { |
| 48 | return []; |
| 49 | }, |
| 50 | }; |
| 51 | return { vectors, embedder, store, embeddedCount: () => embedded }; |
| 52 | } |
| 53 | |
| 54 | const section = (name: string) => `## ${name}\n\n${Array.from({ length: 60 }, (_, i) => `${name.toLowerCase()}${i}`).join(" ")}.`; |
| 55 | const page = (markdown: string, over: Partial<IndexDoc> = {}): IndexDoc => ({ kind: "page", doc_id: "pag_1", workspace_id: "w1", space_id: "s1", title: "Ops", markdown, ...over }); |
| 56 | |
| 57 | test("a page's passages are stored, embedded once, and only changed ones again", async () => { |
| 58 | const db = fakeD1(); |
| 59 | const f = fakes(); |
| 60 | const first = await indexDoc(db, f.embedder, f.store, page([section("Deploy"), section("Rollback")].join("\n\n"))); |
| 61 | assert.deepEqual(first, { chunks: 2, embedded: 2, capped: false }); |
| 62 | assert.deepEqual([...f.vectors.keys()].sort(), ["pag_1:0", "pag_1:1"]); |
| 63 | assert.deepEqual(f.vectors.get("pag_1:0")!.metadata, { workspace_id: "w1", space_id: "s1", kind: "page", page_id: "pag_1" }); |
| 64 | // Saved again unchanged: nothing embedded. |
| 65 | assert.equal((await indexDoc(db, f.embedder, f.store, page([section("Deploy"), section("Rollback")].join("\n\n")))).embedded, 0); |
| 66 | // One section changed. |
| 67 | assert.equal((await indexDoc(db, f.embedder, f.store, page([section("Deploy"), section("Restore")].join("\n\n")))).embedded, 1); |
| 68 | assert.equal(f.embeddedCount(), 3); |
| 69 | // Word search finds passages by heading and text. |
| 70 | const hit = await db.prepare("SELECT chunk_id FROM doc_chunks_fts WHERE doc_chunks_fts MATCH ?").bind('"restore"').all<{ chunk_id: string }>(); |
| 71 | assert.deepEqual( |
| 72 | hit.results.map((r) => r.chunk_id), |
| 73 | ["pag_1:1"], |
| 74 | ); |
| 75 | }); |
| 76 | |
| 77 | test("a passage that only moved keeps its vector; gone passages leave the index", async () => { |
| 78 | const db = fakeD1(); |
| 79 | const f = fakes(); |
| 80 | await indexDoc(db, f.embedder, f.store, page([section("Alpha"), section("Beta"), section("Gamma")].join("\n\n"))); |
| 81 | const before = f.embeddedCount(); |
| 82 | // Alpha removed: Beta and Gamma move up a place, nothing new to embed. |
| 83 | const r = await indexDoc(db, f.embedder, f.store, page([section("Beta"), section("Gamma")].join("\n\n"))); |
| 84 | assert.equal(r.embedded, 0); |
| 85 | assert.equal(f.embeddedCount(), before); |
| 86 | assert.deepEqual([...f.vectors.keys()].sort(), ["pag_1:0", "pag_1:1"]); |
| 87 | const rows = await db.prepare("SELECT id, heading, hash = vector_hash AS current FROM doc_chunks ORDER BY seq").all<{ id: string; heading: string; current: number }>(); |
| 88 | assert.deepEqual( |
| 89 | rows.results.map((x) => [x.id, x.heading, x.current]), |
| 90 | [ |
| 91 | ["pag_1:0", "Beta", 1], |
| 92 | ["pag_1:1", "Gamma", 1], |
| 93 | ], |
| 94 | ); |
| 95 | }); |
| 96 | |
| 97 | test("moving a page to another space files its vectors there without embedding", async () => { |
| 98 | const db = fakeD1(); |
| 99 | const f = fakes(); |
| 100 | await indexDoc(db, f.embedder, f.store, page(section("Deploy"))); |
| 101 | const r = await indexDoc(db, f.embedder, f.store, page(section("Deploy"), { space_id: "s2" })); |
| 102 | assert.equal(r.embedded, 0); |
| 103 | assert.equal(f.vectors.get("pag_1:0")!.metadata.space_id, "s2"); |
| 104 | assert.equal((await db.prepare("SELECT space_id FROM doc_chunks WHERE id = 'pag_1:0'").first<{ space_id: string }>())!.space_id, "s2"); |
| 105 | }); |
| 106 | |
| 107 | test("embedding stops at the hourly cap and the rest waits, kept for words", async () => { |
| 108 | const db = fakeD1(); |
| 109 | const f = fakes(); |
| 110 | const at = new Date("2026-10-09T10:15:00Z"); |
| 111 | await db.prepare("INSERT INTO doc_embed_usage (workspace_id, hour, chunks, tokens) VALUES ('w1', '2026-10-09T10', ?, 0)").bind(EMBED_PER_HOUR - 1).run(); |
| 112 | const r = await indexDoc(db, f.embedder, f.store, page([section("One"), section("Two"), section("Three")].join("\n\n")), at); |
| 113 | assert.deepEqual(r, { chunks: 3, embedded: 1, capped: true }); |
| 114 | const waiting = await db.prepare("SELECT COUNT(*) AS n FROM doc_chunks WHERE vector_hash IS NULL OR vector_hash <> hash").first<{ n: number }>(); |
| 115 | assert.equal(waiting!.n, 2); |
| 116 | // The next hour, a save picks up the rest. |
| 117 | const later = await indexDoc(db, f.embedder, f.store, page([section("One"), section("Two"), section("Three")].join("\n\n")), new Date("2026-10-09T11:01:00Z")); |
| 118 | assert.deepEqual(later, { chunks: 3, embedded: 2, capped: false }); |
| 119 | }); |
| 120 | |
| 121 | test("an embedding failure leaves passages for the next save, never throws", async () => { |
| 122 | const db = fakeD1(); |
| 123 | const f = fakes(); |
| 124 | const broken: Embedder = { |
| 125 | async embed() { |
| 126 | throw new Error("model down"); |
| 127 | }, |
| 128 | }; |
| 129 | const r = await indexDoc(db, broken, f.store, page(section("Deploy"))); |
| 130 | assert.equal(r.chunks, 1); |
| 131 | assert.equal(f.vectors.size, 0); |
| 132 | assert.equal((await indexDoc(db, f.embedder, f.store, page(section("Deploy")))).embedded, 1); |
| 133 | }); |
| 134 | |
| 135 | test("without an embedder, passages are kept for words only", async () => { |
| 136 | const db = fakeD1(); |
| 137 | const r = await indexDoc(db, null, null, page(section("Deploy"))); |
| 138 | assert.deepEqual(r, { chunks: 1, embedded: 0, capped: false }); |
| 139 | assert.equal((await db.prepare("SELECT COUNT(*) AS n FROM doc_chunks_fts").first<{ n: number }>())!.n, 1); |
| 140 | }); |
| 141 | |
| 142 | test("a project's docs file is indexed with its repository", async () => { |
| 143 | const db = fakeD1(); |
| 144 | const f = fakes(); |
| 145 | await indexDoc(db, f.embedder, f.store, { kind: "repo_file", doc_id: "rf_x", workspace_id: "w1", space_id: "rds_1", title: "Setup", markdown: `# Setup\n\n${section("Install")}`, repo_id: "r1", path: "docs/setup.md" }); |
| 146 | assert.deepEqual(f.vectors.get("rf_x:0")!.metadata, { workspace_id: "w1", space_id: "rds_1", kind: "repo_file", repo_file_id: "rf_x", repo_id: "r1" }); |
| 147 | const row = await db.prepare("SELECT page_id, repo_file_id, path, heading FROM doc_chunks").first<Record<string, unknown>>(); |
| 148 | assert.deepEqual({ ...row }, { page_id: null, repo_file_id: "rf_x", path: "docs/setup.md", heading: "Install" }); |
| 149 | }); |