| 1 | /** |
| 2 | * A chat message's Markdown, parsed into a small tree that React renders |
| 3 | * as text (components/chat/text.tsx). Nothing in a message is ever HTML: |
| 4 | * `<b>` is shown as typed, a link goes only to the web, mail or a page on |
| 5 | * this site, and an image is shown as a link to it. Pure, so it is tested |
| 6 | * on its own (chat-markdown.test.ts). |
| 7 | * |
| 8 | * The dialect is what people and agents write in chat: bold, italic, |
| 9 | * strikethrough (`~~x~~` or `~x~`), inline and fenced code, links and bare |
| 10 | * URLs, lists (nested by indenting), quotes, headings, rules and tables |
| 11 | * (a header row, a `---` row with optional `:` alignment, then rows), plus what |
| 12 | * g1t adds: `@mentions`, `#channels` and `#123` references. A line break |
| 13 | * is kept where it was typed. |
| 14 | */ |
| 15 | |
| 16 | export type Span = |
| 17 | | { t: "text"; v: string } |
| 18 | | { t: "code"; v: string } |
| 19 | | { t: "strong"; c: Span[] } |
| 20 | | { t: "em"; c: Span[] } |
| 21 | | { t: "del"; c: Span[] } |
| 22 | | { t: "link"; href: string; c: Span[] } |
| 23 | | { t: "mention"; name: string } |
| 24 | | { t: "channel"; name: string } |
| 25 | | { t: "ref"; repo: string | null; number: number }; |
| 26 | |
| 27 | export type Block = |
| 28 | | { t: "p"; lines: Span[][] } |
| 29 | | { t: "heading"; level: number; c: Span[] } |
| 30 | | { t: "code"; lang: string | null; v: string } |
| 31 | | { t: "list"; ordered: boolean; start: number; items: Block[][] } |
| 32 | | { t: "quote"; c: Block[] } |
| 33 | | { t: "hr" } |
| 34 | | { t: "table"; align: Align[]; head: Span[][]; rows: Span[][][] }; |
| 35 | |
| 36 | /** How a table's column lines up, from its `---` row: `:--`, `:-:` or `--:`. */ |
| 37 | export type Align = "left" | "center" | "right" | null; |
| 38 | |
| 39 | /** Where a link may go: the web, mail, or a page on this site. */ |
| 40 | export function safeHref(href: string): string | null { |
| 41 | const trimmed = href.trim(); |
| 42 | // Control characters and spaces never belong in a link someone can follow. |
| 43 | // eslint-disable-next-line no-control-regex |
| 44 | if (/[\u0000-\u0020\u007f]/.test(trimmed)) return null; |
| 45 | if (/^https?:\/\/[^\s\\]+$/i.test(trimmed)) return trimmed; |
| 46 | if (/^mailto:[^\s/\\]+$/i.test(trimmed)) return trimmed; |
| 47 | // A path on this site; `//host` and `/\host` are other sites to a browser. |
| 48 | if (/^\/(?![/\\])[^\s\\]*$/.test(trimmed)) return trimmed; |
| 49 | return null; |
| 50 | } |
| 51 | |
| 52 | /** The ASCII punctuation a backslash can escape. */ |
| 53 | const ESCAPABLE = "\\\\`*_{}\\[\\]()#+\\-.!~>|<=:@"; |
| 54 | |
| 55 | const INLINE = new RegExp( |
| 56 | [ |
| 57 | `\\\\([${ESCAPABLE}])`, // 1 an escaped character |
| 58 | "``\\s?([^\\n]+?)\\s?``(?!`)", // 2 code with two backticks |
| 59 | "`([^`\\n]+)`", // 3 code |
| 60 | "\\*\\*([^*\\s](?:[^\\n]*?[^*\\s])?)\\*\\*", // 4 strong |
| 61 | "(?<![\\w_])__([^_\\s](?:[^\\n]*?[^_\\s])?)__(?![\\w_])", // 5 strong with _ |
| 62 | "~~([^~\\n]+)~~", // 6 del |
| 63 | "(?<![\\w~])~([^~\\s](?:[^~\\n]*[^~\\s])?)~(?![\\w~])", // 7 del with one ~ |
| 64 | "(?<![\\w*])\\*([^*\\s](?:[^*\\n]*[^*\\s])?)\\*(?![\\w*])", // 8 em with * |
| 65 | "(?<![\\w_])_([^_\\s](?:[^_\\n]*[^_\\s])?)_(?![\\w_])", // 9 em with _ |
| 66 | "!\\[([^\\]\\n]{0,1000})\\]\\(<?([^)\\s>]{1,2048})>?(?:\\s+\"[^\"\\n]{0,1000}\")?\\)", // 10 alt, 11 src: an image, shown as a link |
| 67 | "\\[([^\\]\\n]{1,1000})\\]\\(<?([^)\\s>]{1,2048})>?(?:\\s+\"[^\"\\n]{0,1000}\")?\\)", // 12 text, 13 href |
| 68 | "<(https?:\\/\\/[^\\s<>]+|mailto:[^\\s<>]+)>", // 14 <autolink> |
| 69 | "(https?:\\/\\/(?:[^\\s<>()]|\\([^\\s<>()]*\\))*(?:[^\\s<>().,;:!?'\"*_~]|\\([^\\s<>()]*\\)))", // 15 bare link |
| 70 | "(?<![\\w/@])@([A-Za-z0-9][\\w.-]*[A-Za-z0-9_]|[A-Za-z0-9])(?:\\/([A-Za-z0-9][\\w.-]*))?", // 16 mention, 17 team |
| 71 | "(?<![\\w/])([A-Za-z0-9][\\w.-]*(?:\\/[A-Za-z0-9][\\w.-]*)?)#(\\d+)\\b", // 18 repo, 19 number |
| 72 | "(?<![\\w&#/])#(\\d+)\\b", // 20 number alone |
| 73 | "(?<![\\w&#/])#([a-z0-9][a-z0-9_-]*)", // 21 channel |
| 74 | ].join("|"), |
| 75 | "g", |
| 76 | ); |
| 77 | |
| 78 | /** One line of text as spans. */ |
| 79 | export function inline(text: string): Span[] { |
| 80 | const spans: Span[] = []; |
| 81 | let at = 0; |
| 82 | const push = (span: Span) => { |
| 83 | const last = spans[spans.length - 1]; |
| 84 | if (span.t === "text" && last?.t === "text") last.v += span.v; |
| 85 | else spans.push(span); |
| 86 | }; |
| 87 | const link = (href: string, inner: Span[], raw: string) => { |
| 88 | const safe = safeHref(href); |
| 89 | if (safe) push({ t: "link", href: safe, c: inner }); |
| 90 | else push({ t: "text", v: raw }); |
| 91 | }; |
| 92 | // Its own copy: the spans inside bold and the like are parsed on the way. |
| 93 | const pattern = new RegExp(INLINE.source, "g"); |
| 94 | for (let m = pattern.exec(text); m; m = pattern.exec(text)) { |
| 95 | if (m.index > at) push({ t: "text", v: text.slice(at, m.index) }); |
| 96 | at = m.index + m[0].length; |
| 97 | if (m[1] != null) push({ t: "text", v: m[1] }); |
| 98 | else if (m[2] != null) push({ t: "code", v: m[2] }); |
| 99 | else if (m[3] != null) push({ t: "code", v: m[3] }); |
| 100 | else if (m[4] != null) push({ t: "strong", c: inline(m[4]) }); |
| 101 | else if (m[5] != null) push({ t: "strong", c: inline(m[5]) }); |
| 102 | else if (m[6] != null) push({ t: "del", c: inline(m[6]) }); |
| 103 | else if (m[7] != null) push({ t: "del", c: inline(m[7]) }); |
| 104 | else if (m[8] != null) push({ t: "em", c: inline(m[8]) }); |
| 105 | else if (m[9] != null) push({ t: "em", c: inline(m[9]) }); |
| 106 | else if (m[11] != null) link(m[11], [{ t: "text", v: m[10] || m[11] }], m[0]); |
| 107 | else if (m[13] != null) link(m[13], plainSpans(inline(m[12]!)), m[0]); |
| 108 | else if (m[14] != null) link(m[14], [{ t: "text", v: m[14] }], m[0]); |
| 109 | else if (m[15] != null) push({ t: "link", href: m[15], c: [{ t: "text", v: m[15] }] }); |
| 110 | else if (m[16] != null) push({ t: "mention", name: m[17] ? `${m[16]}/${m[17]}` : m[16] }); |
| 111 | else if (m[18] != null) push({ t: "ref", repo: m[18], number: Number(m[19]) }); |
| 112 | else if (m[20] != null) push({ t: "ref", repo: null, number: Number(m[20]) }); |
| 113 | else if (m[21] != null) push({ t: "channel", name: m[21] }); |
| 114 | } |
| 115 | if (at < text.length) push({ t: "text", v: text.slice(at) }); |
| 116 | return spans; |
| 117 | } |
| 118 | |
| 119 | /** A link's text keeps its formatting, but never a link, mention or reference inside a link. */ |
| 120 | function plainSpans(list: Span[]): Span[] { |
| 121 | const out: Span[] = []; |
| 122 | for (const span of list.map(plainSpan)) { |
| 123 | const last = out[out.length - 1]; |
| 124 | if (span.t === "text" && last?.t === "text") last.v += span.v; |
| 125 | else out.push(span); |
| 126 | } |
| 127 | return out; |
| 128 | } |
| 129 | |
| 130 | function plainSpan(span: Span): Span { |
| 131 | switch (span.t) { |
| 132 | case "link": |
| 133 | return { t: "text", v: spansText(span.c) }; |
| 134 | case "mention": |
| 135 | return { t: "text", v: `@${span.name}` }; |
| 136 | case "channel": |
| 137 | return { t: "text", v: `#${span.name}` }; |
| 138 | case "ref": |
| 139 | return { t: "text", v: `${span.repo ?? ""}#${span.number}` }; |
| 140 | case "strong": |
| 141 | case "em": |
| 142 | case "del": |
| 143 | return { ...span, c: plainSpans(span.c) }; |
| 144 | default: |
| 145 | return { ...span }; |
| 146 | } |
| 147 | } |
| 148 | |
| 149 | // --------------------------------------------------------------------------- |
| 150 | // Blocks. |
| 151 | |
| 152 | const FENCE = /^( {0,3})(`{3,}|~{3,})\s*([\w+#.-]*)[^`]*$/; |
| 153 | const HEADING = /^ {0,3}(#{1,6})(?:[ \t]+(.*?))?(?:[ \t]+#+)?[ \t]*$/; |
| 154 | const RULE = /^ {0,3}([-*_])(?:[ \t]*\1){2,}[ \t]*$/; |
| 155 | const QUOTE = /^ {0,3}>[ \t]?/; |
| 156 | /** A heading has words after its `#`s: `#general` is a channel. */ |
| 157 | const HEADED = /^ {0,3}#{1,6}[ \t]+\S/; |
| 158 | const ITEM = /^([ \t]*)([-*+•]|\d{1,9}[.)])(?:([ \t]+)(.*)|[ \t]*$)/; |
| 159 | |
| 160 | /** A table's `---` row: cells of dashes, each with an optional `:` either side. */ |
| 161 | const DELIMITER = /^ {0,3}\|?[ \t]*:?-+:?[ \t]*(?:\|[ \t]*:?-+:?[ \t]*)*\|?[ \t]*$/; |
| 162 | |
| 163 | /** |
| 164 | * A table row's cells: split on `|`, but not on `\|` or a `|` inside |
| 165 | * backticks, with the pipes at either end optional. |
| 166 | */ |
| 167 | function cells(line: string): string[] { |
| 168 | let text = line.trim(); |
| 169 | if (text.startsWith("|")) text = text.slice(1); |
| 170 | if (text.endsWith("|") && !text.endsWith("\\|")) text = text.slice(0, -1); |
| 171 | const out: string[] = []; |
| 172 | let cell = ""; |
| 173 | let ticks = 0; |
| 174 | for (let i = 0; i < text.length; i++) { |
| 175 | const char = text[i]!; |
| 176 | if (char === "\\" && text[i + 1] === "|") { |
| 177 | cell += "|"; |
| 178 | i++; |
| 179 | } else if (char === "`") { |
| 180 | let run = 1; |
| 181 | while (text[i + run] === "`") run++; |
| 182 | ticks = ticks === 0 ? run : ticks === run ? 0 : ticks; |
| 183 | cell += "`".repeat(run); |
| 184 | i += run - 1; |
| 185 | } else if (char === "|" && ticks === 0) { |
| 186 | out.push(cell.trim()); |
| 187 | cell = ""; |
| 188 | } else cell += char; |
| 189 | } |
| 190 | out.push(cell.trim()); |
| 191 | return out; |
| 192 | } |
| 193 | |
| 194 | /** Whether `line` and the one after it start a table: a row with a pipe, then a `---` row as wide. */ |
| 195 | function tableAt(lines: string[], i: number): boolean { |
| 196 | const head = lines[i]!; |
| 197 | const rule = lines[i + 1]; |
| 198 | if (rule == null || !head.includes("|") || !rule.includes("-") || !DELIMITER.test(rule)) return false; |
| 199 | // One column needs its pipes, so a `---` under a line of text stays a paragraph and a rule. |
| 200 | if (!rule.includes("|") && !head.trim().startsWith("|")) return false; |
| 201 | return cells(head).length === cells(rule).length; |
| 202 | } |
| 203 | |
| 204 | function alignOf(cell: string): Align { |
| 205 | const left = cell.startsWith(":"); |
| 206 | const right = cell.endsWith(":"); |
| 207 | return left && right ? "center" : right ? "right" : left ? "left" : null; |
| 208 | } |
| 209 | |
| 210 | function indentOf(line: string): number { |
| 211 | let width = 0; |
| 212 | for (const char of line) { |
| 213 | if (char === " ") width++; |
| 214 | else if (char === "\t") width += 4 - (width % 4); |
| 215 | else break; |
| 216 | } |
| 217 | return width; |
| 218 | } |
| 219 | |
| 220 | /** A line with up to `width` columns of its indent taken off. */ |
| 221 | function dedent(line: string, width: number): string { |
| 222 | let taken = 0; |
| 223 | let i = 0; |
| 224 | while (i < line.length && taken < width) { |
| 225 | if (line[i] === " ") taken++; |
| 226 | else if (line[i] === "\t") taken += 4 - (taken % 4); |
| 227 | else break; |
| 228 | i++; |
| 229 | } |
| 230 | return line.slice(i); |
| 231 | } |
| 232 | |
| 233 | /** Whether a line starts a block of its own, ending a paragraph. */ |
| 234 | function startsBlock(line: string): boolean { |
| 235 | if (FENCE.test(line) || HEADED.test(line) || RULE.test(line) || QUOTE.test(line)) return true; |
| 236 | // A numbered list breaks into a paragraph only when it counts from 1. |
| 237 | const item = ITEM.exec(line); |
| 238 | return item != null && (!/\d/.test(item[2]!) || Number.parseInt(item[2]!, 10) === 1); |
| 239 | } |
| 240 | |
| 241 | /** A paragraph's line without the marks of a hard break at its end. */ |
| 242 | function lineText(line: string): string { |
| 243 | return line.replace(/(?: {2,}|\\)$/, "").trim(); |
| 244 | } |
| 245 | |
| 246 | function parse(lines: string[], depth: number): Block[] { |
| 247 | const out: Block[] = []; |
| 248 | let paragraph: Span[][] = []; |
| 249 | const flush = () => { |
| 250 | if (paragraph.length) out.push({ t: "p", lines: paragraph }); |
| 251 | paragraph = []; |
| 252 | }; |
| 253 | let i = 0; |
| 254 | while (i < lines.length) { |
| 255 | const line = lines[i]!; |
| 256 | if (line.trim() === "") { |
| 257 | // A blank line ends a paragraph. |
| 258 | flush(); |
| 259 | i++; |
| 260 | continue; |
| 261 | } |
| 262 | if (tableAt(lines, i)) { |
| 263 | flush(); |
| 264 | const align = cells(lines[i + 1]!).map(alignOf); |
| 265 | const width = align.length; |
| 266 | // Every row has the header's columns: cut what is beyond, fill what is missing. |
| 267 | const row = (text: string) => { |
| 268 | const list = cells(text).slice(0, width); |
| 269 | while (list.length < width) list.push(""); |
| 270 | return list.map((cell) => inline(cell)); |
| 271 | }; |
| 272 | const head = row(line); |
| 273 | const rows: Span[][][] = []; |
| 274 | i += 2; |
| 275 | while (i < lines.length && lines[i]!.trim() !== "" && lines[i]!.includes("|") && !startsBlock(lines[i]!)) rows.push(row(lines[i++]!)); |
| 276 | out.push({ t: "table", align, head, rows }); |
| 277 | continue; |
| 278 | } |
| 279 | const fence = FENCE.exec(line); |
| 280 | if (fence) { |
| 281 | flush(); |
| 282 | const [, indent, marks] = fence; |
| 283 | const close = new RegExp(`^ {0,3}${marks![0] === "`" ? "`" : "~"}{${marks!.length},}\\s*$`); |
| 284 | const body: string[] = []; |
| 285 | i++; |
| 286 | while (i < lines.length && !close.test(lines[i]!)) body.push(dedent(lines[i++]!, indent!.length)); |
| 287 | i++; // the closing fence, or the end |
| 288 | out.push({ t: "code", lang: fence[3] || null, v: body.join("\n") }); |
| 289 | continue; |
| 290 | } |
| 291 | const heading = HEADED.test(line) ? HEADING.exec(line) : null; |
| 292 | if (heading) { |
| 293 | flush(); |
| 294 | out.push({ t: "heading", level: heading[1]!.length, c: inline((heading[2] ?? "").trim()) }); |
| 295 | i++; |
| 296 | continue; |
| 297 | } |
| 298 | if (RULE.test(line)) { |
| 299 | flush(); |
| 300 | out.push({ t: "hr" }); |
| 301 | i++; |
| 302 | continue; |
| 303 | } |
| 304 | if (QUOTE.test(line)) { |
| 305 | flush(); |
| 306 | const quoted: string[] = []; |
| 307 | while (i < lines.length && QUOTE.test(lines[i]!)) quoted.push(lines[i++]!.replace(QUOTE, "")); |
| 308 | out.push({ t: "quote", c: depth < 8 ? parse(quoted, depth + 1) : [{ t: "p", lines: quoted.map((q) => inline(q)) }] }); |
| 309 | continue; |
| 310 | } |
| 311 | const item = ITEM.exec(line); |
| 312 | if (item && depth < 8 && (paragraph.length === 0 || startsBlock(line))) { |
| 313 | flush(); |
| 314 | const ordered = /\d/.test(item[2]!); |
| 315 | const base = indentOf(item[1]!); |
| 316 | const items: Block[][] = []; |
| 317 | const start = ordered ? Number.parseInt(item[2]!, 10) : 1; |
| 318 | while (i < lines.length) { |
| 319 | const head = ITEM.exec(lines[i]!); |
| 320 | if (!head || /\d/.test(head[2]!) !== ordered || indentOf(head[1]!) > base + 1) break; |
| 321 | // Where the item's text starts: what is indented that far belongs to it. |
| 322 | const gap = head[3] ? Math.min(indentOf(head[3]), 4) : 1; |
| 323 | const content = base + head[2]!.length + gap; |
| 324 | const body = [head[4] ?? ""]; |
| 325 | i++; |
| 326 | while (i < lines.length) { |
| 327 | const next = lines[i]!; |
| 328 | if (next.trim() === "") { |
| 329 | // A blank line inside an item, if what follows is still indented under it. |
| 330 | const after = lines.slice(i + 1).find((l) => l.trim() !== ""); |
| 331 | if (after != null && indentOf(after) > base && !(ITEM.test(after) && indentOf(after) <= base + 1)) { |
| 332 | body.push(""); |
| 333 | i++; |
| 334 | continue; |
| 335 | } |
| 336 | break; |
| 337 | } |
| 338 | if (indentOf(next) <= base) break; |
| 339 | if (ITEM.test(next) && indentOf(next) <= base + 1) break; |
| 340 | body.push(dedent(next, Math.min(indentOf(next), content))); |
| 341 | i++; |
| 342 | } |
| 343 | items.push(parse(body, depth + 1)); |
| 344 | // Blank lines between items keep one list. |
| 345 | let j = i; |
| 346 | while (j < lines.length && lines[j]!.trim() === "") j++; |
| 347 | const following = j < lines.length ? ITEM.exec(lines[j]!) : null; |
| 348 | if (j > i && following && /\d/.test(following[2]!) === ordered && indentOf(following[1]!) <= base + 1) i = j; |
| 349 | } |
| 350 | out.push({ t: "list", ordered, start, items }); |
| 351 | continue; |
| 352 | } |
| 353 | if (paragraph.length && (startsBlock(line) || tableAt(lines, i))) flush(); |
| 354 | paragraph.push(inline(lineText(line))); |
| 355 | i++; |
| 356 | } |
| 357 | flush(); |
| 358 | return out; |
| 359 | } |
| 360 | |
| 361 | /** A message's whole text as blocks: paragraphs, headings, code, lists, quotes, rules and tables. */ |
| 362 | export function blocks(text: string): Block[] { |
| 363 | return parse(text.replace(/\r\n?/g, "\n").split("\n"), 0); |
| 364 | } |
| 365 | |
| 366 | // --------------------------------------------------------------------------- |
| 367 | // Plain text. |
| 368 | |
| 369 | /** Spans as the words they show. */ |
| 370 | export function spansText(list: Span[]): string { |
| 371 | return list |
| 372 | .map((span) => { |
| 373 | switch (span.t) { |
| 374 | case "text": |
| 375 | case "code": |
| 376 | return span.v; |
| 377 | case "strong": |
| 378 | case "em": |
| 379 | case "del": |
| 380 | case "link": |
| 381 | return spansText(span.c); |
| 382 | case "mention": |
| 383 | return `@${span.name}`; |
| 384 | case "channel": |
| 385 | return `#${span.name}`; |
| 386 | case "ref": |
| 387 | return `${span.repo ?? ""}#${span.number}`; |
| 388 | } |
| 389 | }) |
| 390 | .join(""); |
| 391 | } |
| 392 | |
| 393 | function blocksText(list: Block[]): string[] { |
| 394 | const out: string[] = []; |
| 395 | for (const block of list) { |
| 396 | switch (block.t) { |
| 397 | case "p": |
| 398 | out.push(block.lines.map(spansText).join(" ")); |
| 399 | break; |
| 400 | case "heading": |
| 401 | out.push(spansText(block.c)); |
| 402 | break; |
| 403 | case "code": |
| 404 | out.push(block.v); |
| 405 | break; |
| 406 | case "list": |
| 407 | block.items.forEach((item, index) => { |
| 408 | const text = blocksText(item).join(" "); |
| 409 | out.push(block.ordered ? `${block.start + index}. ${text}` : text); |
| 410 | }); |
| 411 | break; |
| 412 | case "quote": |
| 413 | out.push(...blocksText(block.c)); |
| 414 | break; |
| 415 | case "hr": |
| 416 | break; |
| 417 | case "table": |
| 418 | for (const row of [block.head, ...block.rows]) out.push(row.map(spansText).join(" · ")); |
| 419 | break; |
| 420 | } |
| 421 | } |
| 422 | return out.filter((line) => line.trim() !== ""); |
| 423 | } |
| 424 | |
| 425 | /** |
| 426 | * A message as plain words on one line, for a preview: the Markdown's |
| 427 | * marks gone, a link as its text, a mention as `@name`. Cut to `max` |
| 428 | * characters with an ellipsis. |
| 429 | */ |
| 430 | export function plainText(markdown: string, max = Infinity): string { |
| 431 | const text = blocksText(blocks(markdown)).join(" ").replace(/\s+/g, " ").trim(); |
| 432 | return text.length > max ? `${text.slice(0, max - 1).trimEnd()}…` : text; |
| 433 | } |