| 1 | import assert from "node:assert/strict"; |
| 2 | import { test } from "node:test"; |
| 3 | |
| 4 | import { EMBED_CHARS, MAX_CHARS, MAX_CHUNKS, MIN_CHARS, chunkId, chunkMarkdown, embedText, repoFileId, textHash } from "./chunks.ts"; |
| 5 | |
| 6 | const para = (words: number, word = "rollback") => Array.from({ length: words }, (_, i) => `${word}${i % 7}`).join(" ") + "."; |
| 7 | |
| 8 | test("passages follow headings, with the heading path", () => { |
| 9 | const md = [ |
| 10 | "Intro words that set the scene for the runbook. " + para(40, "intro"), |
| 11 | "", |
| 12 | "# Runbook", |
| 13 | "", |
| 14 | "## Deploy", |
| 15 | "", |
| 16 | para(60, "deploy"), |
| 17 | "", |
| 18 | "## Rollback", |
| 19 | "", |
| 20 | para(60, "rollback"), |
| 21 | "", |
| 22 | "### Database", |
| 23 | "", |
| 24 | para(60, "database"), |
| 25 | ].join("\n"); |
| 26 | const chunks = chunkMarkdown(md, "Operations"); |
| 27 | assert.deepEqual( |
| 28 | chunks.map((c) => c.heading), |
| 29 | [null, "Runbook › Deploy", "Runbook › Rollback", "Runbook › Rollback › Database"], |
| 30 | ); |
| 31 | assert.deepEqual( |
| 32 | chunks.map((c) => c.seq), |
| 33 | [0, 1, 2, 3], |
| 34 | ); |
| 35 | assert.match(chunks[2]!.text, /^rollback0/); |
| 36 | // The heading isn't repeated in the text. |
| 37 | assert.ok(!chunks[2]!.text.includes("## Rollback")); |
| 38 | }); |
| 39 | |
| 40 | test("long sections split on paragraphs, never past the most", () => { |
| 41 | const md = ["## Big", "", ...Array.from({ length: 12 }, () => para(40)).flatMap((p) => [p, ""])].join("\n"); |
| 42 | const chunks = chunkMarkdown(md, "T"); |
| 43 | assert.ok(chunks.length > 1); |
| 44 | for (const c of chunks) { |
| 45 | assert.ok(c.text.length <= MAX_CHARS, `${c.text.length}`); |
| 46 | assert.equal(c.heading, "Big"); |
| 47 | } |
| 48 | // Nothing lost. |
| 49 | const words = (t: string) => t.split(/\s+/).filter(Boolean).length; |
| 50 | assert.equal(words(chunks.map((c) => c.text).join("\n\n")), words(md) - 2); |
| 51 | }); |
| 52 | |
| 53 | test("one huge paragraph is cut at sentences", () => { |
| 54 | const huge = Array.from({ length: 80 }, (_, i) => `Sentence number ${i} explains a part of the system.`).join(" "); |
| 55 | const chunks = chunkMarkdown(huge, "T"); |
| 56 | assert.ok(chunks.length >= 3); |
| 57 | for (const c of chunks) assert.ok(c.text.length <= MAX_CHARS); |
| 58 | assert.ok(chunks[0]!.text.endsWith(".")); |
| 59 | }); |
| 60 | |
| 61 | test("tiny sections join the next, naming its heading", () => { |
| 62 | const md = ["## A", "", "Short.", "", "## B", "", "Also short.", "", "## C", "", para(80)].join("\n"); |
| 63 | const chunks = chunkMarkdown(md, "T"); |
| 64 | assert.equal(chunks[0]!.heading, "A"); |
| 65 | assert.match(chunks[0]!.text, /Short\.\n\n\*\*B\*\*\n\nAlso short\./); |
| 66 | for (const c of chunks.slice(0, -1)) assert.ok(c.text.length >= MIN_CHARS || chunks.length === 1); |
| 67 | }); |
| 68 | |
| 69 | test("a tiny last section joins the one before", () => { |
| 70 | const md = ["## A", "", para(60), "", "## B", "", "The end."].join("\n"); |
| 71 | const chunks = chunkMarkdown(md, "T"); |
| 72 | assert.equal(chunks.length, 1); |
| 73 | assert.match(chunks[0]!.text, /\*\*B\*\*\n\nThe end\.$/); |
| 74 | }); |
| 75 | |
| 76 | test("headings inside code fences are code, and fences stay whole", () => { |
| 77 | const code = ["```sh", "# not a heading", ...Array.from({ length: 10 }, (_, i) => `echo step ${i}`), "```"].join("\n"); |
| 78 | const md = ["## Script", "", para(30), "", code, "", para(30)].join("\n"); |
| 79 | const chunks = chunkMarkdown(md, "T"); |
| 80 | assert.ok(chunks.every((c) => c.heading === "Script")); |
| 81 | const withCode = chunks.find((c) => c.text.includes("```sh"))!; |
| 82 | assert.ok(withCode.text.includes("# not a heading")); |
| 83 | assert.equal((withCode.text.match(/```/g) ?? []).length, 2); |
| 84 | }); |
| 85 | |
| 86 | test("a file's front matter and leading title are not passages", () => { |
| 87 | const md = ["---", "title: Setup", "---", "# Setup", "", para(60, "install")].join("\n"); |
| 88 | const chunks = chunkMarkdown(md, "Setup"); |
| 89 | assert.equal(chunks.length, 1); |
| 90 | assert.equal(chunks[0]!.heading, null); |
| 91 | assert.ok(!chunks[0]!.text.includes("title:")); |
| 92 | }); |
| 93 | |
| 94 | test("empty documents have no passages; huge ones stop at the most", () => { |
| 95 | assert.deepEqual(chunkMarkdown("", "T"), []); |
| 96 | assert.deepEqual(chunkMarkdown("\n\n \n", "T"), []); |
| 97 | const many = Array.from({ length: MAX_CHUNKS + 30 }, (_, i) => `## S${i}\n\n${para(70)}`).join("\n\n"); |
| 98 | assert.equal(chunkMarkdown(many, "T").length, MAX_CHUNKS); |
| 99 | }); |
| 100 | |
| 101 | test("what is embedded leads with the title and heading", () => { |
| 102 | assert.equal(embedText("Ops", { heading: "Runbook › Rollback", text: "Do this." }), "Ops › Runbook › Rollback\n\nDo this."); |
| 103 | assert.equal(embedText("", { heading: null, text: "Just text." }), "Just text."); |
| 104 | assert.equal(embedText("T", { heading: null, text: "x".repeat(5000) }).length, EMBED_CHARS); |
| 105 | }); |
| 106 | |
| 107 | test("hashes and ids are stable and short", () => { |
| 108 | assert.equal(textHash("a"), textHash("a")); |
| 109 | assert.notEqual(textHash("a"), textHash("b")); |
| 110 | assert.match(textHash("anything"), /^[0-9a-f]{16}$/); |
| 111 | const id = repoFileId("rds_1", "docs/a/very/long/path/that/goes/on/and/on/README.md"); |
| 112 | assert.equal(id, repoFileId("rds_1", "docs/a/very/long/path/that/goes/on/and/on/README.md")); |
| 113 | assert.notEqual(id, repoFileId("rds_2", "docs/a/very/long/path/that/goes/on/and/on/README.md")); |
| 114 | // A vector id is at most 64 bytes. |
| 115 | assert.ok(chunkId(id, 149).length <= 64); |
| 116 | assert.equal(chunkId("pag_1", 3), "pag_1:3"); |
| 117 | }); |