Skip to content
117 linesCodeBlameRaw
1import assert from "node:assert/strict";
2import { test } from "node:test";
3
4import { EMBED_CHARS, MAX_CHARS, MAX_CHUNKS, MIN_CHARS, chunkId, chunkMarkdown, embedText, repoFileId, textHash } from "./chunks.ts";
5
6const para = (words: number, word = "rollback") => Array.from({ length: words }, (_, i) => `${word}${i % 7}`).join(" ") + ".";
7
8test("passages follow headings, with the heading path", () => {
9 const md = [
10 "Intro words that set the scene for the runbook. " + para(40, "intro"),
11 "",
12 "# Runbook",
13 "",
14 "## Deploy",
15 "",
16 para(60, "deploy"),
17 "",
18 "## Rollback",
19 "",
20 para(60, "rollback"),
21 "",
22 "### Database",
23 "",
24 para(60, "database"),
25 ].join("\n");
26 const chunks = chunkMarkdown(md, "Operations");
27 assert.deepEqual(
28 chunks.map((c) => c.heading),
29 [null, "Runbook › Deploy", "Runbook › Rollback", "Runbook › Rollback › Database"],
30 );
31 assert.deepEqual(
32 chunks.map((c) => c.seq),
33 [0, 1, 2, 3],
34 );
35 assert.match(chunks[2]!.text, /^rollback0/);
36 // The heading isn't repeated in the text.
37 assert.ok(!chunks[2]!.text.includes("## Rollback"));
38});
39
40test("long sections split on paragraphs, never past the most", () => {
41 const md = ["## Big", "", ...Array.from({ length: 12 }, () => para(40)).flatMap((p) => [p, ""])].join("\n");
42 const chunks = chunkMarkdown(md, "T");
43 assert.ok(chunks.length > 1);
44 for (const c of chunks) {
45 assert.ok(c.text.length <= MAX_CHARS, `${c.text.length}`);
46 assert.equal(c.heading, "Big");
47 }
48 // Nothing lost.
49 const words = (t: string) => t.split(/\s+/).filter(Boolean).length;
50 assert.equal(words(chunks.map((c) => c.text).join("\n\n")), words(md) - 2);
51});
52
53test("one huge paragraph is cut at sentences", () => {
54 const huge = Array.from({ length: 80 }, (_, i) => `Sentence number ${i} explains a part of the system.`).join(" ");
55 const chunks = chunkMarkdown(huge, "T");
56 assert.ok(chunks.length >= 3);
57 for (const c of chunks) assert.ok(c.text.length <= MAX_CHARS);
58 assert.ok(chunks[0]!.text.endsWith("."));
59});
60
61test("tiny sections join the next, naming its heading", () => {
62 const md = ["## A", "", "Short.", "", "## B", "", "Also short.", "", "## C", "", para(80)].join("\n");
63 const chunks = chunkMarkdown(md, "T");
64 assert.equal(chunks[0]!.heading, "A");
65 assert.match(chunks[0]!.text, /Short\.\n\n\*\*B\*\*\n\nAlso short\./);
66 for (const c of chunks.slice(0, -1)) assert.ok(c.text.length >= MIN_CHARS || chunks.length === 1);
67});
68
69test("a tiny last section joins the one before", () => {
70 const md = ["## A", "", para(60), "", "## B", "", "The end."].join("\n");
71 const chunks = chunkMarkdown(md, "T");
72 assert.equal(chunks.length, 1);
73 assert.match(chunks[0]!.text, /\*\*B\*\*\n\nThe end\.$/);
74});
75
76test("headings inside code fences are code, and fences stay whole", () => {
77 const code = ["```sh", "# not a heading", ...Array.from({ length: 10 }, (_, i) => `echo step ${i}`), "```"].join("\n");
78 const md = ["## Script", "", para(30), "", code, "", para(30)].join("\n");
79 const chunks = chunkMarkdown(md, "T");
80 assert.ok(chunks.every((c) => c.heading === "Script"));
81 const withCode = chunks.find((c) => c.text.includes("```sh"))!;
82 assert.ok(withCode.text.includes("# not a heading"));
83 assert.equal((withCode.text.match(/```/g) ?? []).length, 2);
84});
85
86test("a file's front matter and leading title are not passages", () => {
87 const md = ["---", "title: Setup", "---", "# Setup", "", para(60, "install")].join("\n");
88 const chunks = chunkMarkdown(md, "Setup");
89 assert.equal(chunks.length, 1);
90 assert.equal(chunks[0]!.heading, null);
91 assert.ok(!chunks[0]!.text.includes("title:"));
92});
93
94test("empty documents have no passages; huge ones stop at the most", () => {
95 assert.deepEqual(chunkMarkdown("", "T"), []);
96 assert.deepEqual(chunkMarkdown("\n\n \n", "T"), []);
97 const many = Array.from({ length: MAX_CHUNKS + 30 }, (_, i) => `## S${i}\n\n${para(70)}`).join("\n\n");
98 assert.equal(chunkMarkdown(many, "T").length, MAX_CHUNKS);
99});
100
101test("what is embedded leads with the title and heading", () => {
102 assert.equal(embedText("Ops", { heading: "Runbook › Rollback", text: "Do this." }), "Ops › Runbook › Rollback\n\nDo this.");
103 assert.equal(embedText("", { heading: null, text: "Just text." }), "Just text.");
104 assert.equal(embedText("T", { heading: null, text: "x".repeat(5000) }).length, EMBED_CHARS);
105});
106
107test("hashes and ids are stable and short", () => {
108 assert.equal(textHash("a"), textHash("a"));
109 assert.notEqual(textHash("a"), textHash("b"));
110 assert.match(textHash("anything"), /^[0-9a-f]{16}$/);
111 const id = repoFileId("rds_1", "docs/a/very/long/path/that/goes/on/and/on/README.md");
112 assert.equal(id, repoFileId("rds_1", "docs/a/very/long/path/that/goes/on/and/on/README.md"));
113 assert.notEqual(id, repoFileId("rds_2", "docs/a/very/long/path/that/goes/on/and/on/README.md"));
114 // A vector id is at most 64 bytes.
115 assert.ok(chunkId(id, 149).length <= 64);
116 assert.equal(chunkId("pag_1", 3), "pag_1:3");
117});