Pick any line to see why it is the way it is: the commit, the pull request and issue it came from, and what the agent was thinking.
| Docs index by meaning: passages of every page and project doc, embedded on save and recalled for agents; hybrid search for people | 1 | import assert from "node:assert/strict"; |
| 2 | import { test } from "node:test"; | |
| 3 | ||
| 4 | import { EMBED_CHARS, MAX_CHARS, MAX_CHUNKS, MIN_CHARS, chunkId, chunkMarkdown, embedText, repoFileId, textHash } from "./chunks.ts"; | |
| 5 | ||
| 6 | const para = (words: number, word = "rollback") => Array.from({ length: words }, (_, i) => `${word}${i % 7}`).join(" ") + "."; | |
| 7 | ||
| 8 | test("passages follow headings, with the heading path", () => { | |
| 9 | const md = [ | |
| 10 | "Intro words that set the scene for the runbook. " + para(40, "intro"), | |
| 11 | "", | |
| 12 | "# Runbook", | |
| 13 | "", | |
| 14 | "## Deploy", | |
| 15 | "", | |
| 16 | para(60, "deploy"), | |
| 17 | "", | |
| 18 | "## Rollback", | |
| 19 | "", | |
| 20 | para(60, "rollback"), | |
| 21 | "", | |
| 22 | "### Database", | |
| 23 | "", | |
| 24 | para(60, "database"), | |
| 25 | ].join("\n"); | |
| 26 | const chunks = chunkMarkdown(md, "Operations"); | |
| 27 | assert.deepEqual( | |
| 28 | chunks.map((c) => c.heading), | |
| 29 | [null, "Runbook › Deploy", "Runbook › Rollback", "Runbook › Rollback › Database"], | |
| 30 | ); | |
| 31 | assert.deepEqual( | |
| 32 | chunks.map((c) => c.seq), | |
| 33 | [0, 1, 2, 3], | |
| 34 | ); | |
| 35 | assert.match(chunks[2]!.text, /^rollback0/); | |
| 36 | // The heading isn't repeated in the text. | |
| 37 | assert.ok(!chunks[2]!.text.includes("## Rollback")); | |
| 38 | }); | |
| 39 | ||
| 40 | test("long sections split on paragraphs, never past the most", () => { | |
| 41 | const md = ["## Big", "", ...Array.from({ length: 12 }, () => para(40)).flatMap((p) => [p, ""])].join("\n"); | |
| 42 | const chunks = chunkMarkdown(md, "T"); | |
| 43 | assert.ok(chunks.length > 1); | |
| 44 | for (const c of chunks) { | |
| 45 | assert.ok(c.text.length <= MAX_CHARS, `${c.text.length}`); | |
| 46 | assert.equal(c.heading, "Big"); | |
| 47 | } | |
| 48 | // Nothing lost. | |
| 49 | const words = (t: string) => t.split(/\s+/).filter(Boolean).length; | |
| 50 | assert.equal(words(chunks.map((c) => c.text).join("\n\n")), words(md) - 2); | |
| 51 | }); | |
| 52 | ||
| 53 | test("one huge paragraph is cut at sentences", () => { | |
| 54 | const huge = Array.from({ length: 80 }, (_, i) => `Sentence number ${i} explains a part of the system.`).join(" "); | |
| 55 | const chunks = chunkMarkdown(huge, "T"); | |
| 56 | assert.ok(chunks.length >= 3); | |
| 57 | for (const c of chunks) assert.ok(c.text.length <= MAX_CHARS); | |
| 58 | assert.ok(chunks[0]!.text.endsWith(".")); | |
| 59 | }); | |
| 60 | ||
| 61 | test("tiny sections join the next, naming its heading", () => { | |
| 62 | const md = ["## A", "", "Short.", "", "## B", "", "Also short.", "", "## C", "", para(80)].join("\n"); | |
| 63 | const chunks = chunkMarkdown(md, "T"); | |
| 64 | assert.equal(chunks[0]!.heading, "A"); | |
| 65 | assert.match(chunks[0]!.text, /Short\.\n\n\*\*B\*\*\n\nAlso short\./); | |
| 66 | for (const c of chunks.slice(0, -1)) assert.ok(c.text.length >= MIN_CHARS || chunks.length === 1); | |
| 67 | }); | |
| 68 | ||
| 69 | test("a tiny last section joins the one before", () => { | |
| 70 | const md = ["## A", "", para(60), "", "## B", "", "The end."].join("\n"); | |
| 71 | const chunks = chunkMarkdown(md, "T"); | |
| 72 | assert.equal(chunks.length, 1); | |
| 73 | assert.match(chunks[0]!.text, /\*\*B\*\*\n\nThe end\.$/); | |
| 74 | }); | |
| 75 | ||
| 76 | test("headings inside code fences are code, and fences stay whole", () => { | |
| 77 | const code = ["```sh", "# not a heading", ...Array.from({ length: 10 }, (_, i) => `echo step ${i}`), "```"].join("\n"); | |
| 78 | const md = ["## Script", "", para(30), "", code, "", para(30)].join("\n"); | |
| 79 | const chunks = chunkMarkdown(md, "T"); | |
| 80 | assert.ok(chunks.every((c) => c.heading === "Script")); | |
| 81 | const withCode = chunks.find((c) => c.text.includes("```sh"))!; | |
| 82 | assert.ok(withCode.text.includes("# not a heading")); | |
| 83 | assert.equal((withCode.text.match(/```/g) ?? []).length, 2); | |
| 84 | }); | |
| 85 | ||
| 86 | test("a file's front matter and leading title are not passages", () => { | |
| 87 | const md = ["---", "title: Setup", "---", "# Setup", "", para(60, "install")].join("\n"); | |
| 88 | const chunks = chunkMarkdown(md, "Setup"); | |
| 89 | assert.equal(chunks.length, 1); | |
| 90 | assert.equal(chunks[0]!.heading, null); | |
| 91 | assert.ok(!chunks[0]!.text.includes("title:")); | |
| 92 | }); | |
| 93 | ||
| 94 | test("empty documents have no passages; huge ones stop at the most", () => { | |
| 95 | assert.deepEqual(chunkMarkdown("", "T"), []); | |
| 96 | assert.deepEqual(chunkMarkdown("\n\n \n", "T"), []); | |
| 97 | const many = Array.from({ length: MAX_CHUNKS + 30 }, (_, i) => `## S${i}\n\n${para(70)}`).join("\n\n"); | |
| 98 | assert.equal(chunkMarkdown(many, "T").length, MAX_CHUNKS); | |
| 99 | }); | |
| 100 | ||
| 101 | test("what is embedded leads with the title and heading", () => { | |
| 102 | assert.equal(embedText("Ops", { heading: "Runbook › Rollback", text: "Do this." }), "Ops › Runbook › Rollback\n\nDo this."); | |
| 103 | assert.equal(embedText("", { heading: null, text: "Just text." }), "Just text."); | |
| 104 | assert.equal(embedText("T", { heading: null, text: "x".repeat(5000) }).length, EMBED_CHARS); | |
| 105 | }); | |
| 106 | ||
| 107 | test("hashes and ids are stable and short", () => { | |
| 108 | assert.equal(textHash("a"), textHash("a")); | |
| 109 | assert.notEqual(textHash("a"), textHash("b")); | |
| 110 | assert.match(textHash("anything"), /^[0-9a-f]{16}$/); | |
| 111 | const id = repoFileId("rds_1", "docs/a/very/long/path/that/goes/on/and/on/README.md"); | |
| 112 | assert.equal(id, repoFileId("rds_1", "docs/a/very/long/path/that/goes/on/and/on/README.md")); | |
| 113 | assert.notEqual(id, repoFileId("rds_2", "docs/a/very/long/path/that/goes/on/and/on/README.md")); | |
| 114 | // A vector id is at most 64 bytes. | |
| 115 | assert.ok(chunkId(id, 149).length <= 64); | |
| 116 | assert.equal(chunkId("pag_1", 3), "pag_1:3"); | |
| 117 | }); |