Skip to content
149 linesCodeBlameRaw
1import assert from "node:assert/strict";
2import { readFileSync, readdirSync } from "node:fs";
3import { DatabaseSync } from "node:sqlite";
4import { test } from "node:test";
5
6import { EMBED_PER_HOUR, indexDoc, type IndexDoc } from "./indexer.ts";
7import type { Embedder, VectorMetadata, VectorStore } from "./vectors.ts";
8
9/** D1, as far as the indexer uses it, over node's SQLite with the service's migrations. */
10function fakeD1(): D1Database {
11 const db = new DatabaseSync(":memory:");
12 const dir = new URL("../migrations/", import.meta.url);
13 for (const file of readdirSync(dir).sort()) db.exec(readFileSync(new URL(file, dir), "utf8"));
14 const statement = (sql: string, params: unknown[] = []) => ({
15 sql,
16 params,
17 bind: (...values: unknown[]) => statement(sql, values),
18 first: async () => (db.prepare(sql).get(...(params as never[])) as unknown) ?? null,
19 all: async () => ({ results: db.prepare(sql).all(...(params as never[])) }),
20 run: async () => db.prepare(sql).run(...(params as never[])),
21 });
22 return {
23 prepare: (sql: string) => statement(sql),
24 batch: async (list: ReturnType<typeof statement>[]) => list.map((s) => db.prepare(s.sql).run(...(s.params as never[]))),
25 } as unknown as D1Database;
26}
27
28function fakes() {
29 const vectors = new Map<string, { values: number[]; metadata: VectorMetadata }>();
30 let embedded = 0;
31 const embedder: Embedder = {
32 async embed(texts) {
33 embedded += texts.length;
34 return texts.map((t) => [t.length, 1]);
35 },
36 };
37 const store: VectorStore = {
38 async upsert(list) {
39 for (const v of list) vectors.set(v.id, { values: v.values, metadata: v.metadata });
40 },
41 async get(ids) {
42 return ids.filter((id) => vectors.has(id)).map((id) => ({ id, values: vectors.get(id)!.values }));
43 },
44 async delete(ids) {
45 for (const id of ids) vectors.delete(id);
46 },
47 async query() {
48 return [];
49 },
50 };
51 return { vectors, embedder, store, embeddedCount: () => embedded };
52}
53
54const section = (name: string) => `## ${name}\n\n${Array.from({ length: 60 }, (_, i) => `${name.toLowerCase()}${i}`).join(" ")}.`;
55const page = (markdown: string, over: Partial<IndexDoc> = {}): IndexDoc => ({ kind: "page", doc_id: "pag_1", workspace_id: "w1", space_id: "s1", title: "Ops", markdown, ...over });
56
57test("a page's passages are stored, embedded once, and only changed ones again", async () => {
58 const db = fakeD1();
59 const f = fakes();
60 const first = await indexDoc(db, f.embedder, f.store, page([section("Deploy"), section("Rollback")].join("\n\n")));
61 assert.deepEqual(first, { chunks: 2, embedded: 2, capped: false });
62 assert.deepEqual([...f.vectors.keys()].sort(), ["pag_1:0", "pag_1:1"]);
63 assert.deepEqual(f.vectors.get("pag_1:0")!.metadata, { workspace_id: "w1", space_id: "s1", kind: "page", page_id: "pag_1" });
64 // Saved again unchanged: nothing embedded.
65 assert.equal((await indexDoc(db, f.embedder, f.store, page([section("Deploy"), section("Rollback")].join("\n\n")))).embedded, 0);
66 // One section changed.
67 assert.equal((await indexDoc(db, f.embedder, f.store, page([section("Deploy"), section("Restore")].join("\n\n")))).embedded, 1);
68 assert.equal(f.embeddedCount(), 3);
69 // Word search finds passages by heading and text.
70 const hit = await db.prepare("SELECT chunk_id FROM doc_chunks_fts WHERE doc_chunks_fts MATCH ?").bind('"restore"').all<{ chunk_id: string }>();
71 assert.deepEqual(
72 hit.results.map((r) => r.chunk_id),
73 ["pag_1:1"],
74 );
75});
76
77test("a passage that only moved keeps its vector; gone passages leave the index", async () => {
78 const db = fakeD1();
79 const f = fakes();
80 await indexDoc(db, f.embedder, f.store, page([section("Alpha"), section("Beta"), section("Gamma")].join("\n\n")));
81 const before = f.embeddedCount();
82 // Alpha removed: Beta and Gamma move up a place, nothing new to embed.
83 const r = await indexDoc(db, f.embedder, f.store, page([section("Beta"), section("Gamma")].join("\n\n")));
84 assert.equal(r.embedded, 0);
85 assert.equal(f.embeddedCount(), before);
86 assert.deepEqual([...f.vectors.keys()].sort(), ["pag_1:0", "pag_1:1"]);
87 const rows = await db.prepare("SELECT id, heading, hash = vector_hash AS current FROM doc_chunks ORDER BY seq").all<{ id: string; heading: string; current: number }>();
88 assert.deepEqual(
89 rows.results.map((x) => [x.id, x.heading, x.current]),
90 [
91 ["pag_1:0", "Beta", 1],
92 ["pag_1:1", "Gamma", 1],
93 ],
94 );
95});
96
97test("moving a page to another space files its vectors there without embedding", async () => {
98 const db = fakeD1();
99 const f = fakes();
100 await indexDoc(db, f.embedder, f.store, page(section("Deploy")));
101 const r = await indexDoc(db, f.embedder, f.store, page(section("Deploy"), { space_id: "s2" }));
102 assert.equal(r.embedded, 0);
103 assert.equal(f.vectors.get("pag_1:0")!.metadata.space_id, "s2");
104 assert.equal((await db.prepare("SELECT space_id FROM doc_chunks WHERE id = 'pag_1:0'").first<{ space_id: string }>())!.space_id, "s2");
105});
106
107test("embedding stops at the hourly cap and the rest waits, kept for words", async () => {
108 const db = fakeD1();
109 const f = fakes();
110 const at = new Date("2026-10-09T10:15:00Z");
111 await db.prepare("INSERT INTO doc_embed_usage (workspace_id, hour, chunks, tokens) VALUES ('w1', '2026-10-09T10', ?, 0)").bind(EMBED_PER_HOUR - 1).run();
112 const r = await indexDoc(db, f.embedder, f.store, page([section("One"), section("Two"), section("Three")].join("\n\n")), at);
113 assert.deepEqual(r, { chunks: 3, embedded: 1, capped: true });
114 const waiting = await db.prepare("SELECT COUNT(*) AS n FROM doc_chunks WHERE vector_hash IS NULL OR vector_hash <> hash").first<{ n: number }>();
115 assert.equal(waiting!.n, 2);
116 // The next hour, a save picks up the rest.
117 const later = await indexDoc(db, f.embedder, f.store, page([section("One"), section("Two"), section("Three")].join("\n\n")), new Date("2026-10-09T11:01:00Z"));
118 assert.deepEqual(later, { chunks: 3, embedded: 2, capped: false });
119});
120
121test("an embedding failure leaves passages for the next save, never throws", async () => {
122 const db = fakeD1();
123 const f = fakes();
124 const broken: Embedder = {
125 async embed() {
126 throw new Error("model down");
127 },
128 };
129 const r = await indexDoc(db, broken, f.store, page(section("Deploy")));
130 assert.equal(r.chunks, 1);
131 assert.equal(f.vectors.size, 0);
132 assert.equal((await indexDoc(db, f.embedder, f.store, page(section("Deploy")))).embedded, 1);
133});
134
135test("without an embedder, passages are kept for words only", async () => {
136 const db = fakeD1();
137 const r = await indexDoc(db, null, null, page(section("Deploy")));
138 assert.deepEqual(r, { chunks: 1, embedded: 0, capped: false });
139 assert.equal((await db.prepare("SELECT COUNT(*) AS n FROM doc_chunks_fts").first<{ n: number }>())!.n, 1);
140});
141
142test("a project's docs file is indexed with its repository", async () => {
143 const db = fakeD1();
144 const f = fakes();
145 await indexDoc(db, f.embedder, f.store, { kind: "repo_file", doc_id: "rf_x", workspace_id: "w1", space_id: "rds_1", title: "Setup", markdown: `# Setup\n\n${section("Install")}`, repo_id: "r1", path: "docs/setup.md" });
146 assert.deepEqual(f.vectors.get("rf_x:0")!.metadata, { workspace_id: "w1", space_id: "rds_1", kind: "repo_file", repo_file_id: "rf_x", repo_id: "r1" });
147 const row = await db.prepare("SELECT page_id, repo_file_id, path, heading FROM doc_chunks").first<Record<string, unknown>>();
148 assert.deepEqual({ ...row }, { page_id: null, repo_file_id: "rf_x", path: "docs/setup.md", heading: "Install" });
149});