| 1 | import assert from "node:assert/strict"; |
| 2 | import { test } from "node:test"; |
| 3 | |
| 4 | import { parseBlocks, parseSpans } from "./file-markdown.ts"; |
| 5 | import { base64, csv, fileName, makeFile, previewTable } from "./files.ts"; |
| 6 | import { cellValue, columnName, crc32, markdownDocx, sheetNames, workbook } from "./ooxml.ts"; |
| 7 | import { markdownPdf, winAnsi } from "./pdf.ts"; |
| 8 | |
| 9 | const latin1 = (bytes: Uint8Array) => String.fromCharCode(...bytes); |
| 10 | |
| 11 | /** A stored zip's entries, each checked against its CRC. */ |
| 12 | function unzip(bytes: Uint8Array): Map<string, string> { |
| 13 | const view = new DataView(bytes.buffer, bytes.byteOffset, bytes.byteLength); |
| 14 | let end = bytes.length - 22; |
| 15 | while (end >= 0 && view.getUint32(end, true) !== 0x06054b50) end--; |
| 16 | assert.ok(end >= 0, "has an end of central directory"); |
| 17 | const count = view.getUint16(end + 10, true); |
| 18 | let at = view.getUint32(end + 16, true); |
| 19 | const out = new Map<string, string>(); |
| 20 | for (let i = 0; i < count; i++) { |
| 21 | assert.equal(view.getUint32(at, true), 0x02014b50); |
| 22 | const crc = view.getUint32(at + 16, true); |
| 23 | const size = view.getUint32(at + 20, true); |
| 24 | const nameLength = view.getUint16(at + 28, true); |
| 25 | const local = view.getUint32(at + 42, true); |
| 26 | const name = new TextDecoder().decode(bytes.subarray(at + 46, at + 46 + nameLength)); |
| 27 | assert.equal(view.getUint32(local, true), 0x04034b50); |
| 28 | const start = local + 30 + view.getUint16(local + 26, true) + view.getUint16(local + 28, true); |
| 29 | const data = bytes.subarray(start, start + size); |
| 30 | assert.equal(crc32(data), crc, `${name}'s CRC`); |
| 31 | out.set(name, new TextDecoder().decode(data)); |
| 32 | at += 46 + nameLength; |
| 33 | } |
| 34 | return out; |
| 35 | } |
| 36 | |
| 37 | const REPORT = `# Quarterly report |
| 38 | |
| 39 | Revenue grew **12%** to *$1.2M*. See [the dashboard](https://example.com/q3) and \`acme/web\`. |
| 40 | |
| 41 | ## Highlights |
| 42 | |
| 43 | - Shipped the merge queue |
| 44 | - Cut p95 latency |
| 45 | 1. In the API |
| 46 | 2. Ordered too |
| 47 | |
| 48 | > [!NOTE] |
| 49 | > Numbers are unaudited. |
| 50 | |
| 51 | | Region | Revenue | Growth | |
| 52 | | --- | ---: | --- | |
| 53 | | North | 400,000 | 10% | |
| 54 | | South | 800,000 | 13% | |
| 55 | |
| 56 | \`\`\`ts |
| 57 | const total = north + south; |
| 58 | \`\`\` |
| 59 | |
| 60 | --- |
| 61 | |
| 62 | Done.`; |
| 63 | |
| 64 | test("Markdown reads into the blocks a file needs", () => { |
| 65 | const blocks = parseBlocks(REPORT); |
| 66 | assert.deepEqual( |
| 67 | blocks.map((b) => b.kind), |
| 68 | ["heading", "paragraph", "heading", "item", "item", "item", "item", "quote", "table", "code", "rule", "paragraph"], |
| 69 | ); |
| 70 | const table = blocks.find((b) => b.kind === "table"); |
| 71 | assert.ok(table && table.kind === "table"); |
| 72 | assert.equal(table.rows.length, 2); |
| 73 | const quote = blocks.find((b) => b.kind === "quote"); |
| 74 | assert.ok(quote && quote.kind === "quote"); |
| 75 | assert.match(quote.spans.map((s) => s.text).join(""), /^Note: Numbers are unaudited/); |
| 76 | assert.deepEqual(parseSpans("a **b** _c_ `d` [e](https://x.y) snake_case_word"), [ |
| 77 | { text: "a " }, |
| 78 | { text: "b", bold: true }, |
| 79 | { text: " " }, |
| 80 | { text: "c", italic: true }, |
| 81 | { text: " " }, |
| 82 | { text: "d", code: true }, |
| 83 | { text: " " }, |
| 84 | { text: "e", href: "https://x.y" }, |
| 85 | { text: " snake_case_word" }, |
| 86 | ]); |
| 87 | }); |
| 88 | |
| 89 | test("a PDF is well formed: every object where the cross-reference table says, pages counted, links annotated", () => { |
| 90 | const { bytes, pages, truncated } = markdownPdf("Quarterly report", REPORT); |
| 91 | const text = latin1(bytes); |
| 92 | assert.ok(text.startsWith("%PDF-1.4\n")); |
| 93 | assert.ok(text.endsWith("%%EOF\n")); |
| 94 | assert.equal(pages, 1); |
| 95 | assert.equal(truncated, false); |
| 96 | const xref = Number(/startxref\n(\d+)\n/.exec(text)![1]); |
| 97 | assert.ok(text.slice(xref).startsWith("xref\n")); |
| 98 | const entries = [...text.slice(xref).matchAll(/^(\d{10}) 00000 n $/gm)].map((m) => Number(m[1])); |
| 99 | entries.forEach((offset, i) => assert.ok(text.slice(offset).startsWith(`${i + 1} 0 obj\n`), `object ${i + 1} is where the table says`)); |
| 100 | assert.match(text, /\/Count 1 >>/); |
| 101 | assert.match(text, /\(Quarterly report\) Tj/); |
| 102 | assert.match(text, /\(Region\) Tj/); |
| 103 | assert.match(text, /\/URI \(https:\/\/example.com\/q3\)/); |
| 104 | for (const m of text.matchAll(/<< \/Length (\d+) >>\nstream\n/g)) { |
| 105 | const start = m.index! + m[0].length; |
| 106 | assert.ok(text.slice(start + Number(m[1])).startsWith("\nendstream"), "each stream is as long as it says"); |
| 107 | } |
| 108 | // The title isn't repeated when the Markdown starts with it. |
| 109 | assert.equal(text.match(/\(Quarterly report\) Tj/g)!.length, 1); |
| 110 | }); |
| 111 | |
| 112 | test("a long PDF breaks across pages and numbers them", () => { |
| 113 | const long = Array.from({ length: 200 }, (_, i) => `Paragraph ${i + 1}: ${"words ".repeat(40)}`).join("\n\n"); |
| 114 | const { bytes, pages } = markdownPdf("Long", long); |
| 115 | assert.ok(pages > 10); |
| 116 | const text = latin1(bytes); |
| 117 | assert.match(text, new RegExp(`/Count ${pages} >>`)); |
| 118 | assert.match(text, new RegExp(`\\(${pages} of ${pages}\\) Tj`)); |
| 119 | }); |
| 120 | |
| 121 | test("PDF text is Windows-1252: curly quotes and dashes kept, accents folded, emoji dropped", () => { |
| 122 | assert.deepEqual(winAnsi("“a” – b"), [0x93, 0x61, 0x94, 0x20, 0x96, 0x20, 0x62]); |
| 123 | assert.deepEqual(winAnsi("é"), [0xe9]); |
| 124 | assert.deepEqual(winAnsi("ő"), [0x6f]); |
| 125 | assert.deepEqual(winAnsi("ok 🚀"), [0x6f, 0x6b, 0x20]); |
| 126 | assert.deepEqual(winAnsi("中"), [0x3f]); |
| 127 | }); |
| 128 | |
| 129 | test("a Word document is a valid package with the text, styles and links", () => { |
| 130 | const files = unzip(markdownDocx("Quarterly report", REPORT, new Date("2026-10-10T12:00:00Z"))); |
| 131 | assert.deepEqual([...files.keys()].sort(), ["[Content_Types].xml", "_rels/.rels", "docProps/core.xml", "word/_rels/document.xml.rels", "word/document.xml", "word/styles.xml"]); |
| 132 | const document = files.get("word/document.xml")!; |
| 133 | assert.match(document, /<w:pStyle w:val="Heading1"\/><\/w:pPr><w:r><w:t xml:space="preserve">Quarterly report<\/w:t>/); |
| 134 | assert.match(document, /<w:b\/><\/w:rPr><w:t xml:space="preserve">12%<\/w:t>/); |
| 135 | assert.match(document, /<w:hyperlink r:id="rIdLink1">/); |
| 136 | assert.match(document, /<w:tbl>/); |
| 137 | assert.match(files.get("word/_rels/document.xml.rels")!, /Target="https:\/\/example.com\/q3" TargetMode="External"/); |
| 138 | assert.match(files.get("docProps/core.xml")!, /<dc:title>Quarterly report<\/dc:title>/); |
| 139 | // Text is escaped. |
| 140 | const escaped = unzip(markdownDocx("x", "a < b & c")); |
| 141 | assert.match(escaped.get("word/document.xml")!, /a < b & c/); |
| 142 | }); |
| 143 | |
| 144 | test("a workbook keeps numbers as numbers, codes as text, and a bold frozen header", () => { |
| 145 | const files = unzip( |
| 146 | workbook("Sales", [ |
| 147 | { name: "Q3: by region", rows: [["Region", "Revenue", "Zip"], ["North", 400000, "02139"], ["South", "800000.5", null]] }, |
| 148 | { name: "Q3: by region", rows: [["a"]] }, |
| 149 | ]), |
| 150 | ); |
| 151 | assert.match(files.get("xl/workbook.xml")!, /<sheet name="Q3 by region" sheetId="1"/); |
| 152 | assert.match(files.get("xl/workbook.xml")!, /<sheet name="Q3 by region 2" sheetId="2"/); |
| 153 | const sheet = files.get("xl/worksheets/sheet1.xml")!; |
| 154 | assert.match(sheet, /<c r="A1" s="1" t="inlineStr"><is><t xml:space="preserve">Region<\/t>/); |
| 155 | assert.match(sheet, /<c r="B2"><v>400000<\/v><\/c>/); |
| 156 | assert.match(sheet, /<c r="C2" t="inlineStr"><is><t xml:space="preserve">02139<\/t>/); |
| 157 | assert.match(sheet, /<c r="B3"><v>800000.5<\/v><\/c>/); |
| 158 | assert.match(sheet, /state="frozen"/); |
| 159 | assert.equal(columnName(0), "A"); |
| 160 | assert.equal(columnName(25), "Z"); |
| 161 | assert.equal(columnName(26), "AA"); |
| 162 | assert.equal(columnName(701), "ZZ"); |
| 163 | assert.deepEqual(cellValue("007"), { kind: "text", value: "007" }); |
| 164 | assert.deepEqual(sheetNames(["", "a/b", "x".repeat(40)]), ["Sheet1", "a b", "x".repeat(31)]); |
| 165 | }); |
| 166 | |
| 167 | test("CSV quotes what it must, and starts with a byte-order mark", () => { |
| 168 | assert.equal(csv([["a", "b,c"], ['say "hi"', 3], [null, "line\nbreak"]]), 'a,"b,c"\r\n"say ""hi""",3\r\n,"line\nbreak"\r\n'); |
| 169 | }); |
| 170 | |
| 171 | test("make_file's checks: formats it can't write, missing content, a CSV of two sheets", () => { |
| 172 | const slides = makeFile({ format: "pptx", title: "Deck" }); |
| 173 | assert.equal(slides.ok, false); |
| 174 | assert.match(!slides.ok ? slides.message : "", /Slide decks and images aren't available yet/); |
| 175 | assert.equal(makeFile({ format: "pdf", title: "x", content: " " }).ok, false); |
| 176 | assert.equal(makeFile({ format: "pdf", title: "", content: "x" }).ok, false); |
| 177 | const two = makeFile({ format: "csv", title: "x", sheets: [{ rows: [["a"]] }, { rows: [["b"]] }] }); |
| 178 | assert.match(!two.ok ? two.message : "", /one sheet/); |
| 179 | const pdf = makeFile({ format: ".PDF", title: "Q3/Q4 report.pdf", content: "Hello" }); |
| 180 | assert.ok(pdf.ok); |
| 181 | assert.equal(pdf.ok && pdf.file.name, "Q3 Q4 report.pdf"); |
| 182 | assert.equal(pdf.ok && pdf.file.content_type, "application/pdf"); |
| 183 | assert.equal(pdf.ok && pdf.file.summary, "1 page"); |
| 184 | const sheet = makeFile({ format: "xlsx", title: "Sales", sheets: [{ name: "S", rows: [["a", "b"], [1, 2], [3, 4]] }] }); |
| 185 | assert.equal(sheet.ok && sheet.file.summary, "2 rows"); |
| 186 | assert.equal(fileName("", "md"), "file.md"); |
| 187 | }); |
| 188 | |
| 189 | test("a spreadsheet's preview in its doc, and base64 for the trip to the docs service", () => { |
| 190 | const rows = [["Name", "Total"], ...Array.from({ length: 25 }, (_, i) => [`r${i}`, i])]; |
| 191 | const preview = previewTable({ name: "S", rows }); |
| 192 | assert.match(preview, /^\| Name \| Total \|\n\| --- \| --- \|\n\| r0 \| 0 \|/); |
| 193 | assert.match(preview, /_First 20 of 25 rows._$/); |
| 194 | const bytes = new Uint8Array(100_000).map((_, i) => i % 256); |
| 195 | assert.equal(Buffer.from(base64(bytes), "base64").equals(Buffer.from(bytes)), true); |
| 196 | }); |