Skip to content
196 linesCodeBlameRaw
1import assert from "node:assert/strict";
2import { test } from "node:test";
3
4import { parseBlocks, parseSpans } from "./file-markdown.ts";
5import { base64, csv, fileName, makeFile, previewTable } from "./files.ts";
6import { cellValue, columnName, crc32, markdownDocx, sheetNames, workbook } from "./ooxml.ts";
7import { markdownPdf, winAnsi } from "./pdf.ts";
8
9const latin1 = (bytes: Uint8Array) => String.fromCharCode(...bytes);
10
11/** A stored zip's entries, each checked against its CRC. */
12function unzip(bytes: Uint8Array): Map<string, string> {
13 const view = new DataView(bytes.buffer, bytes.byteOffset, bytes.byteLength);
14 let end = bytes.length - 22;
15 while (end >= 0 && view.getUint32(end, true) !== 0x06054b50) end--;
16 assert.ok(end >= 0, "has an end of central directory");
17 const count = view.getUint16(end + 10, true);
18 let at = view.getUint32(end + 16, true);
19 const out = new Map<string, string>();
20 for (let i = 0; i < count; i++) {
21 assert.equal(view.getUint32(at, true), 0x02014b50);
22 const crc = view.getUint32(at + 16, true);
23 const size = view.getUint32(at + 20, true);
24 const nameLength = view.getUint16(at + 28, true);
25 const local = view.getUint32(at + 42, true);
26 const name = new TextDecoder().decode(bytes.subarray(at + 46, at + 46 + nameLength));
27 assert.equal(view.getUint32(local, true), 0x04034b50);
28 const start = local + 30 + view.getUint16(local + 26, true) + view.getUint16(local + 28, true);
29 const data = bytes.subarray(start, start + size);
30 assert.equal(crc32(data), crc, `${name}'s CRC`);
31 out.set(name, new TextDecoder().decode(data));
32 at += 46 + nameLength;
33 }
34 return out;
35}
36
37const REPORT = `# Quarterly report
38
39Revenue grew **12%** to *$1.2M*. See [the dashboard](https://example.com/q3) and \`acme/web\`.
40
41## Highlights
42
43- Shipped the merge queue
44- Cut p95 latency
45 1. In the API
462. Ordered too
47
48> [!NOTE]
49> Numbers are unaudited.
50
51| Region | Revenue | Growth |
52| --- | ---: | --- |
53| North | 400,000 | 10% |
54| South | 800,000 | 13% |
55
56\`\`\`ts
57const total = north + south;
58\`\`\`
59
60---
61
62Done.`;
63
64test("Markdown reads into the blocks a file needs", () => {
65 const blocks = parseBlocks(REPORT);
66 assert.deepEqual(
67 blocks.map((b) => b.kind),
68 ["heading", "paragraph", "heading", "item", "item", "item", "item", "quote", "table", "code", "rule", "paragraph"],
69 );
70 const table = blocks.find((b) => b.kind === "table");
71 assert.ok(table && table.kind === "table");
72 assert.equal(table.rows.length, 2);
73 const quote = blocks.find((b) => b.kind === "quote");
74 assert.ok(quote && quote.kind === "quote");
75 assert.match(quote.spans.map((s) => s.text).join(""), /^Note: Numbers are unaudited/);
76 assert.deepEqual(parseSpans("a **b** _c_ `d` [e](https://x.y) snake_case_word"), [
77 { text: "a " },
78 { text: "b", bold: true },
79 { text: " " },
80 { text: "c", italic: true },
81 { text: " " },
82 { text: "d", code: true },
83 { text: " " },
84 { text: "e", href: "https://x.y" },
85 { text: " snake_case_word" },
86 ]);
87});
88
89test("a PDF is well formed: every object where the cross-reference table says, pages counted, links annotated", () => {
90 const { bytes, pages, truncated } = markdownPdf("Quarterly report", REPORT);
91 const text = latin1(bytes);
92 assert.ok(text.startsWith("%PDF-1.4\n"));
93 assert.ok(text.endsWith("%%EOF\n"));
94 assert.equal(pages, 1);
95 assert.equal(truncated, false);
96 const xref = Number(/startxref\n(\d+)\n/.exec(text)![1]);
97 assert.ok(text.slice(xref).startsWith("xref\n"));
98 const entries = [...text.slice(xref).matchAll(/^(\d{10}) 00000 n $/gm)].map((m) => Number(m[1]));
99 entries.forEach((offset, i) => assert.ok(text.slice(offset).startsWith(`${i + 1} 0 obj\n`), `object ${i + 1} is where the table says`));
100 assert.match(text, /\/Count 1 >>/);
101 assert.match(text, /\(Quarterly report\) Tj/);
102 assert.match(text, /\(Region\) Tj/);
103 assert.match(text, /\/URI \(https:\/\/example.com\/q3\)/);
104 for (const m of text.matchAll(/<< \/Length (\d+) >>\nstream\n/g)) {
105 const start = m.index! + m[0].length;
106 assert.ok(text.slice(start + Number(m[1])).startsWith("\nendstream"), "each stream is as long as it says");
107 }
108 // The title isn't repeated when the Markdown starts with it.
109 assert.equal(text.match(/\(Quarterly report\) Tj/g)!.length, 1);
110});
111
112test("a long PDF breaks across pages and numbers them", () => {
113 const long = Array.from({ length: 200 }, (_, i) => `Paragraph ${i + 1}: ${"words ".repeat(40)}`).join("\n\n");
114 const { bytes, pages } = markdownPdf("Long", long);
115 assert.ok(pages > 10);
116 const text = latin1(bytes);
117 assert.match(text, new RegExp(`/Count ${pages} >>`));
118 assert.match(text, new RegExp(`\\(${pages} of ${pages}\\) Tj`));
119});
120
121test("PDF text is Windows-1252: curly quotes and dashes kept, accents folded, emoji dropped", () => {
122 assert.deepEqual(winAnsi("“a” – b"), [0x93, 0x61, 0x94, 0x20, 0x96, 0x20, 0x62]);
123 assert.deepEqual(winAnsi("é"), [0xe9]);
124 assert.deepEqual(winAnsi("ő"), [0x6f]);
125 assert.deepEqual(winAnsi("ok 🚀"), [0x6f, 0x6b, 0x20]);
126 assert.deepEqual(winAnsi("中"), [0x3f]);
127});
128
129test("a Word document is a valid package with the text, styles and links", () => {
130 const files = unzip(markdownDocx("Quarterly report", REPORT, new Date("2026-10-10T12:00:00Z")));
131 assert.deepEqual([...files.keys()].sort(), ["[Content_Types].xml", "_rels/.rels", "docProps/core.xml", "word/_rels/document.xml.rels", "word/document.xml", "word/styles.xml"]);
132 const document = files.get("word/document.xml")!;
133 assert.match(document, /<w:pStyle w:val="Heading1"\/><\/w:pPr><w:r><w:t xml:space="preserve">Quarterly report<\/w:t>/);
134 assert.match(document, /<w:b\/><\/w:rPr><w:t xml:space="preserve">12%<\/w:t>/);
135 assert.match(document, /<w:hyperlink r:id="rIdLink1">/);
136 assert.match(document, /<w:tbl>/);
137 assert.match(files.get("word/_rels/document.xml.rels")!, /Target="https:\/\/example.com\/q3" TargetMode="External"/);
138 assert.match(files.get("docProps/core.xml")!, /<dc:title>Quarterly report<\/dc:title>/);
139 // Text is escaped.
140 const escaped = unzip(markdownDocx("x", "a < b & c"));
141 assert.match(escaped.get("word/document.xml")!, /a &lt; b &amp; c/);
142});
143
144test("a workbook keeps numbers as numbers, codes as text, and a bold frozen header", () => {
145 const files = unzip(
146 workbook("Sales", [
147 { name: "Q3: by region", rows: [["Region", "Revenue", "Zip"], ["North", 400000, "02139"], ["South", "800000.5", null]] },
148 { name: "Q3: by region", rows: [["a"]] },
149 ]),
150 );
151 assert.match(files.get("xl/workbook.xml")!, /<sheet name="Q3 by region" sheetId="1"/);
152 assert.match(files.get("xl/workbook.xml")!, /<sheet name="Q3 by region 2" sheetId="2"/);
153 const sheet = files.get("xl/worksheets/sheet1.xml")!;
154 assert.match(sheet, /<c r="A1" s="1" t="inlineStr"><is><t xml:space="preserve">Region<\/t>/);
155 assert.match(sheet, /<c r="B2"><v>400000<\/v><\/c>/);
156 assert.match(sheet, /<c r="C2" t="inlineStr"><is><t xml:space="preserve">02139<\/t>/);
157 assert.match(sheet, /<c r="B3"><v>800000.5<\/v><\/c>/);
158 assert.match(sheet, /state="frozen"/);
159 assert.equal(columnName(0), "A");
160 assert.equal(columnName(25), "Z");
161 assert.equal(columnName(26), "AA");
162 assert.equal(columnName(701), "ZZ");
163 assert.deepEqual(cellValue("007"), { kind: "text", value: "007" });
164 assert.deepEqual(sheetNames(["", "a/b", "x".repeat(40)]), ["Sheet1", "a b", "x".repeat(31)]);
165});
166
167test("CSV quotes what it must, and starts with a byte-order mark", () => {
168 assert.equal(csv([["a", "b,c"], ['say "hi"', 3], [null, "line\nbreak"]]), 'a,"b,c"\r\n"say ""hi""",3\r\n,"line\nbreak"\r\n');
169});
170
171test("make_file's checks: formats it can't write, missing content, a CSV of two sheets", () => {
172 const slides = makeFile({ format: "pptx", title: "Deck" });
173 assert.equal(slides.ok, false);
174 assert.match(!slides.ok ? slides.message : "", /Slide decks and images aren't available yet/);
175 assert.equal(makeFile({ format: "pdf", title: "x", content: " " }).ok, false);
176 assert.equal(makeFile({ format: "pdf", title: "", content: "x" }).ok, false);
177 const two = makeFile({ format: "csv", title: "x", sheets: [{ rows: [["a"]] }, { rows: [["b"]] }] });
178 assert.match(!two.ok ? two.message : "", /one sheet/);
179 const pdf = makeFile({ format: ".PDF", title: "Q3/Q4 report.pdf", content: "Hello" });
180 assert.ok(pdf.ok);
181 assert.equal(pdf.ok && pdf.file.name, "Q3 Q4 report.pdf");
182 assert.equal(pdf.ok && pdf.file.content_type, "application/pdf");
183 assert.equal(pdf.ok && pdf.file.summary, "1 page");
184 const sheet = makeFile({ format: "xlsx", title: "Sales", sheets: [{ name: "S", rows: [["a", "b"], [1, 2], [3, 4]] }] });
185 assert.equal(sheet.ok && sheet.file.summary, "2 rows");
186 assert.equal(fileName("", "md"), "file.md");
187});
188
189test("a spreadsheet's preview in its doc, and base64 for the trip to the docs service", () => {
190 const rows = [["Name", "Total"], ...Array.from({ length: 25 }, (_, i) => [`r${i}`, i])];
191 const preview = previewTable({ name: "S", rows });
192 assert.match(preview, /^\| Name \| Total \|\n\| --- \| --- \|\n\| r0 \| 0 \|/);
193 assert.match(preview, /_First 20 of 25 rows._$/);
194 const bytes = new Uint8Array(100_000).map((_, i) => i % 256);
195 assert.equal(Buffer.from(base64(bytes), "base64").equals(Buffer.from(bytes)), true);
196});