| 1 | /** |
| 2 | * A chat message's Markdown, parsed into a small tree that React renders |
| 3 | * as text (components/chat/text.tsx). Nothing in a message is ever HTML: |
| 4 | * `<b>` is shown as typed, a link goes only to the web, mail or a page on |
| 5 | * this site, and an image is shown as a link to it. Pure, so it is tested |
| 6 | * on its own (chat-markdown.test.ts). |
| 7 | * |
| 8 | * The dialect is what people and agents write in chat: bold, italic, |
| 9 | * strikethrough (`~~x~~` or `~x~`), inline and fenced code, links and bare |
| 10 | * URLs, lists (nested by indenting), quotes, headings and rules, plus what |
| 11 | * g1t adds: `@mentions`, `#channels` and `#123` references. A line break |
| 12 | * is kept where it was typed. |
| 13 | */ |
| 14 | |
| 15 | export type Span = |
| 16 | | { t: "text"; v: string } |
| 17 | | { t: "code"; v: string } |
| 18 | | { t: "strong"; c: Span[] } |
| 19 | | { t: "em"; c: Span[] } |
| 20 | | { t: "del"; c: Span[] } |
| 21 | | { t: "link"; href: string; c: Span[] } |
| 22 | | { t: "mention"; name: string } |
| 23 | | { t: "channel"; name: string } |
| 24 | | { t: "ref"; repo: string | null; number: number }; |
| 25 | |
| 26 | export type Block = |
| 27 | | { t: "p"; lines: Span[][] } |
| 28 | | { t: "heading"; level: number; c: Span[] } |
| 29 | | { t: "code"; lang: string | null; v: string } |
| 30 | | { t: "list"; ordered: boolean; start: number; items: Block[][] } |
| 31 | | { t: "quote"; c: Block[] } |
| 32 | | { t: "hr" }; |
| 33 | |
| 34 | /** Where a link may go: the web, mail, or a page on this site. */ |
| 35 | export function safeHref(href: string): string | null { |
| 36 | const trimmed = href.trim(); |
| 37 | // Control characters and spaces never belong in a link someone can follow. |
| 38 | // eslint-disable-next-line no-control-regex |
| 39 | if (/[\u0000-\u0020\u007f]/.test(trimmed)) return null; |
| 40 | if (/^https?:\/\/[^\s\\]+$/i.test(trimmed)) return trimmed; |
| 41 | if (/^mailto:[^\s/\\]+$/i.test(trimmed)) return trimmed; |
| 42 | // A path on this site; `//host` and `/\host` are other sites to a browser. |
| 43 | if (/^\/(?![/\\])[^\s\\]*$/.test(trimmed)) return trimmed; |
| 44 | return null; |
| 45 | } |
| 46 | |
| 47 | /** The ASCII punctuation a backslash can escape. */ |
| 48 | const ESCAPABLE = "\\\\`*_{}\\[\\]()#+\\-.!~>|<=:@"; |
| 49 | |
| 50 | const INLINE = new RegExp( |
| 51 | [ |
| 52 | `\\\\([${ESCAPABLE}])`, // 1 an escaped character |
| 53 | "``\\s?([^\\n]+?)\\s?``(?!`)", // 2 code with two backticks |
| 54 | "`([^`\\n]+)`", // 3 code |
| 55 | "\\*\\*([^*\\s](?:[^\\n]*?[^*\\s])?)\\*\\*", // 4 strong |
| 56 | "(?<![\\w_])__([^_\\s](?:[^\\n]*?[^_\\s])?)__(?![\\w_])", // 5 strong with _ |
| 57 | "~~([^~\\n]+)~~", // 6 del |
| 58 | "(?<![\\w~])~([^~\\s](?:[^~\\n]*[^~\\s])?)~(?![\\w~])", // 7 del with one ~ |
| 59 | "(?<![\\w*])\\*([^*\\s](?:[^*\\n]*[^*\\s])?)\\*(?![\\w*])", // 8 em with * |
| 60 | "(?<![\\w_])_([^_\\s](?:[^_\\n]*[^_\\s])?)_(?![\\w_])", // 9 em with _ |
| 61 | "!\\[([^\\]\\n]{0,1000})\\]\\(<?([^)\\s>]{1,2048})>?(?:\\s+\"[^\"\\n]{0,1000}\")?\\)", // 10 alt, 11 src: an image, shown as a link |
| 62 | "\\[([^\\]\\n]{1,1000})\\]\\(<?([^)\\s>]{1,2048})>?(?:\\s+\"[^\"\\n]{0,1000}\")?\\)", // 12 text, 13 href |
| 63 | "<(https?:\\/\\/[^\\s<>]+|mailto:[^\\s<>]+)>", // 14 <autolink> |
| 64 | "(https?:\\/\\/(?:[^\\s<>()]|\\([^\\s<>()]*\\))*(?:[^\\s<>().,;:!?'\"*_~]|\\([^\\s<>()]*\\)))", // 15 bare link |
| 65 | "(?<![\\w/@])@([A-Za-z0-9][\\w.-]*[A-Za-z0-9_]|[A-Za-z0-9])(?:\\/([A-Za-z0-9][\\w.-]*))?", // 16 mention, 17 team |
| 66 | "(?<![\\w/])([A-Za-z0-9][\\w.-]*(?:\\/[A-Za-z0-9][\\w.-]*)?)#(\\d+)\\b", // 18 repo, 19 number |
| 67 | "(?<![\\w&#/])#(\\d+)\\b", // 20 number alone |
| 68 | "(?<![\\w&#/])#([a-z0-9][a-z0-9_-]*)", // 21 channel |
| 69 | ].join("|"), |
| 70 | "g", |
| 71 | ); |
| 72 | |
| 73 | /** One line of text as spans. */ |
| 74 | export function inline(text: string): Span[] { |
| 75 | const spans: Span[] = []; |
| 76 | let at = 0; |
| 77 | const push = (span: Span) => { |
| 78 | const last = spans[spans.length - 1]; |
| 79 | if (span.t === "text" && last?.t === "text") last.v += span.v; |
| 80 | else spans.push(span); |
| 81 | }; |
| 82 | const link = (href: string, inner: Span[], raw: string) => { |
| 83 | const safe = safeHref(href); |
| 84 | if (safe) push({ t: "link", href: safe, c: inner }); |
| 85 | else push({ t: "text", v: raw }); |
| 86 | }; |
| 87 | // Its own copy: the spans inside bold and the like are parsed on the way. |
| 88 | const pattern = new RegExp(INLINE.source, "g"); |
| 89 | for (let m = pattern.exec(text); m; m = pattern.exec(text)) { |
| 90 | if (m.index > at) push({ t: "text", v: text.slice(at, m.index) }); |
| 91 | at = m.index + m[0].length; |
| 92 | if (m[1] != null) push({ t: "text", v: m[1] }); |
| 93 | else if (m[2] != null) push({ t: "code", v: m[2] }); |
| 94 | else if (m[3] != null) push({ t: "code", v: m[3] }); |
| 95 | else if (m[4] != null) push({ t: "strong", c: inline(m[4]) }); |
| 96 | else if (m[5] != null) push({ t: "strong", c: inline(m[5]) }); |
| 97 | else if (m[6] != null) push({ t: "del", c: inline(m[6]) }); |
| 98 | else if (m[7] != null) push({ t: "del", c: inline(m[7]) }); |
| 99 | else if (m[8] != null) push({ t: "em", c: inline(m[8]) }); |
| 100 | else if (m[9] != null) push({ t: "em", c: inline(m[9]) }); |
| 101 | else if (m[11] != null) link(m[11], [{ t: "text", v: m[10] || m[11] }], m[0]); |
| 102 | else if (m[13] != null) link(m[13], plainSpans(inline(m[12]!)), m[0]); |
| 103 | else if (m[14] != null) link(m[14], [{ t: "text", v: m[14] }], m[0]); |
| 104 | else if (m[15] != null) push({ t: "link", href: m[15], c: [{ t: "text", v: m[15] }] }); |
| 105 | else if (m[16] != null) push({ t: "mention", name: m[17] ? `${m[16]}/${m[17]}` : m[16] }); |
| 106 | else if (m[18] != null) push({ t: "ref", repo: m[18], number: Number(m[19]) }); |
| 107 | else if (m[20] != null) push({ t: "ref", repo: null, number: Number(m[20]) }); |
| 108 | else if (m[21] != null) push({ t: "channel", name: m[21] }); |
| 109 | } |
| 110 | if (at < text.length) push({ t: "text", v: text.slice(at) }); |
| 111 | return spans; |
| 112 | } |
| 113 | |
| 114 | /** A link's text keeps its formatting, but never a link, mention or reference inside a link. */ |
| 115 | function plainSpans(list: Span[]): Span[] { |
| 116 | const out: Span[] = []; |
| 117 | for (const span of list.map(plainSpan)) { |
| 118 | const last = out[out.length - 1]; |
| 119 | if (span.t === "text" && last?.t === "text") last.v += span.v; |
| 120 | else out.push(span); |
| 121 | } |
| 122 | return out; |
| 123 | } |
| 124 | |
| 125 | function plainSpan(span: Span): Span { |
| 126 | switch (span.t) { |
| 127 | case "link": |
| 128 | return { t: "text", v: spansText(span.c) }; |
| 129 | case "mention": |
| 130 | return { t: "text", v: `@${span.name}` }; |
| 131 | case "channel": |
| 132 | return { t: "text", v: `#${span.name}` }; |
| 133 | case "ref": |
| 134 | return { t: "text", v: `${span.repo ?? ""}#${span.number}` }; |
| 135 | case "strong": |
| 136 | case "em": |
| 137 | case "del": |
| 138 | return { ...span, c: plainSpans(span.c) }; |
| 139 | default: |
| 140 | return { ...span }; |
| 141 | } |
| 142 | } |
| 143 | |
| 144 | // --------------------------------------------------------------------------- |
| 145 | // Blocks. |
| 146 | |
| 147 | const FENCE = /^( {0,3})(`{3,}|~{3,})\s*([\w+#.-]*)[^`]*$/; |
| 148 | const HEADING = /^ {0,3}(#{1,6})(?:[ \t]+(.*?))?(?:[ \t]+#+)?[ \t]*$/; |
| 149 | const RULE = /^ {0,3}([-*_])(?:[ \t]*\1){2,}[ \t]*$/; |
| 150 | const QUOTE = /^ {0,3}>[ \t]?/; |
| 151 | /** A heading has words after its `#`s: `#general` is a channel. */ |
| 152 | const HEADED = /^ {0,3}#{1,6}[ \t]+\S/; |
| 153 | const ITEM = /^([ \t]*)([-*+•]|\d{1,9}[.)])(?:([ \t]+)(.*)|[ \t]*$)/; |
| 154 | |
| 155 | function indentOf(line: string): number { |
| 156 | let width = 0; |
| 157 | for (const char of line) { |
| 158 | if (char === " ") width++; |
| 159 | else if (char === "\t") width += 4 - (width % 4); |
| 160 | else break; |
| 161 | } |
| 162 | return width; |
| 163 | } |
| 164 | |
| 165 | /** A line with up to `width` columns of its indent taken off. */ |
| 166 | function dedent(line: string, width: number): string { |
| 167 | let taken = 0; |
| 168 | let i = 0; |
| 169 | while (i < line.length && taken < width) { |
| 170 | if (line[i] === " ") taken++; |
| 171 | else if (line[i] === "\t") taken += 4 - (taken % 4); |
| 172 | else break; |
| 173 | i++; |
| 174 | } |
| 175 | return line.slice(i); |
| 176 | } |
| 177 | |
| 178 | /** Whether a line starts a block of its own, ending a paragraph. */ |
| 179 | function startsBlock(line: string): boolean { |
| 180 | if (FENCE.test(line) || HEADED.test(line) || RULE.test(line) || QUOTE.test(line)) return true; |
| 181 | // A numbered list breaks into a paragraph only when it counts from 1. |
| 182 | const item = ITEM.exec(line); |
| 183 | return item != null && (!/\d/.test(item[2]!) || Number.parseInt(item[2]!, 10) === 1); |
| 184 | } |
| 185 | |
| 186 | /** A paragraph's line without the marks of a hard break at its end. */ |
| 187 | function lineText(line: string): string { |
| 188 | return line.replace(/(?: {2,}|\\)$/, "").trim(); |
| 189 | } |
| 190 | |
| 191 | function parse(lines: string[], depth: number): Block[] { |
| 192 | const out: Block[] = []; |
| 193 | let paragraph: Span[][] = []; |
| 194 | const flush = () => { |
| 195 | if (paragraph.length) out.push({ t: "p", lines: paragraph }); |
| 196 | paragraph = []; |
| 197 | }; |
| 198 | let i = 0; |
| 199 | while (i < lines.length) { |
| 200 | const line = lines[i]!; |
| 201 | if (line.trim() === "") { |
| 202 | // A blank line ends a paragraph. |
| 203 | flush(); |
| 204 | i++; |
| 205 | continue; |
| 206 | } |
| 207 | const fence = FENCE.exec(line); |
| 208 | if (fence) { |
| 209 | flush(); |
| 210 | const [, indent, marks] = fence; |
| 211 | const close = new RegExp(`^ {0,3}${marks![0] === "`" ? "`" : "~"}{${marks!.length},}\\s*$`); |
| 212 | const body: string[] = []; |
| 213 | i++; |
| 214 | while (i < lines.length && !close.test(lines[i]!)) body.push(dedent(lines[i++]!, indent!.length)); |
| 215 | i++; // the closing fence, or the end |
| 216 | out.push({ t: "code", lang: fence[3] || null, v: body.join("\n") }); |
| 217 | continue; |
| 218 | } |
| 219 | const heading = HEADED.test(line) ? HEADING.exec(line) : null; |
| 220 | if (heading) { |
| 221 | flush(); |
| 222 | out.push({ t: "heading", level: heading[1]!.length, c: inline((heading[2] ?? "").trim()) }); |
| 223 | i++; |
| 224 | continue; |
| 225 | } |
| 226 | if (RULE.test(line)) { |
| 227 | flush(); |
| 228 | out.push({ t: "hr" }); |
| 229 | i++; |
| 230 | continue; |
| 231 | } |
| 232 | if (QUOTE.test(line)) { |
| 233 | flush(); |
| 234 | const quoted: string[] = []; |
| 235 | while (i < lines.length && QUOTE.test(lines[i]!)) quoted.push(lines[i++]!.replace(QUOTE, "")); |
| 236 | out.push({ t: "quote", c: depth < 8 ? parse(quoted, depth + 1) : [{ t: "p", lines: quoted.map((q) => inline(q)) }] }); |
| 237 | continue; |
| 238 | } |
| 239 | const item = ITEM.exec(line); |
| 240 | if (item && depth < 8 && (paragraph.length === 0 || startsBlock(line))) { |
| 241 | flush(); |
| 242 | const ordered = /\d/.test(item[2]!); |
| 243 | const base = indentOf(item[1]!); |
| 244 | const items: Block[][] = []; |
| 245 | const start = ordered ? Number.parseInt(item[2]!, 10) : 1; |
| 246 | while (i < lines.length) { |
| 247 | const head = ITEM.exec(lines[i]!); |
| 248 | if (!head || /\d/.test(head[2]!) !== ordered || indentOf(head[1]!) > base + 1) break; |
| 249 | // Where the item's text starts: what is indented that far belongs to it. |
| 250 | const gap = head[3] ? Math.min(indentOf(head[3]), 4) : 1; |
| 251 | const content = base + head[2]!.length + gap; |
| 252 | const body = [head[4] ?? ""]; |
| 253 | i++; |
| 254 | while (i < lines.length) { |
| 255 | const next = lines[i]!; |
| 256 | if (next.trim() === "") { |
| 257 | // A blank line inside an item, if what follows is still indented under it. |
| 258 | const after = lines.slice(i + 1).find((l) => l.trim() !== ""); |
| 259 | if (after != null && indentOf(after) > base && !(ITEM.test(after) && indentOf(after) <= base + 1)) { |
| 260 | body.push(""); |
| 261 | i++; |
| 262 | continue; |
| 263 | } |
| 264 | break; |
| 265 | } |
| 266 | if (indentOf(next) <= base) break; |
| 267 | if (ITEM.test(next) && indentOf(next) <= base + 1) break; |
| 268 | body.push(dedent(next, Math.min(indentOf(next), content))); |
| 269 | i++; |
| 270 | } |
| 271 | items.push(parse(body, depth + 1)); |
| 272 | // Blank lines between items keep one list. |
| 273 | let j = i; |
| 274 | while (j < lines.length && lines[j]!.trim() === "") j++; |
| 275 | const following = j < lines.length ? ITEM.exec(lines[j]!) : null; |
| 276 | if (j > i && following && /\d/.test(following[2]!) === ordered && indentOf(following[1]!) <= base + 1) i = j; |
| 277 | } |
| 278 | out.push({ t: "list", ordered, start, items }); |
| 279 | continue; |
| 280 | } |
| 281 | if (paragraph.length && startsBlock(line)) flush(); |
| 282 | paragraph.push(inline(lineText(line))); |
| 283 | i++; |
| 284 | } |
| 285 | flush(); |
| 286 | return out; |
| 287 | } |
| 288 | |
| 289 | /** A message's whole text as blocks: paragraphs, headings, code, lists, quotes and rules. */ |
| 290 | export function blocks(text: string): Block[] { |
| 291 | return parse(text.replace(/\r\n?/g, "\n").split("\n"), 0); |
| 292 | } |
| 293 | |
| 294 | // --------------------------------------------------------------------------- |
| 295 | // Plain text. |
| 296 | |
| 297 | /** Spans as the words they show. */ |
| 298 | export function spansText(list: Span[]): string { |
| 299 | return list |
| 300 | .map((span) => { |
| 301 | switch (span.t) { |
| 302 | case "text": |
| 303 | case "code": |
| 304 | return span.v; |
| 305 | case "strong": |
| 306 | case "em": |
| 307 | case "del": |
| 308 | case "link": |
| 309 | return spansText(span.c); |
| 310 | case "mention": |
| 311 | return `@${span.name}`; |
| 312 | case "channel": |
| 313 | return `#${span.name}`; |
| 314 | case "ref": |
| 315 | return `${span.repo ?? ""}#${span.number}`; |
| 316 | } |
| 317 | }) |
| 318 | .join(""); |
| 319 | } |
| 320 | |
| 321 | function blocksText(list: Block[]): string[] { |
| 322 | const out: string[] = []; |
| 323 | for (const block of list) { |
| 324 | switch (block.t) { |
| 325 | case "p": |
| 326 | out.push(block.lines.map(spansText).join(" ")); |
| 327 | break; |
| 328 | case "heading": |
| 329 | out.push(spansText(block.c)); |
| 330 | break; |
| 331 | case "code": |
| 332 | out.push(block.v); |
| 333 | break; |
| 334 | case "list": |
| 335 | block.items.forEach((item, index) => { |
| 336 | const text = blocksText(item).join(" "); |
| 337 | out.push(block.ordered ? `${block.start + index}. ${text}` : text); |
| 338 | }); |
| 339 | break; |
| 340 | case "quote": |
| 341 | out.push(...blocksText(block.c)); |
| 342 | break; |
| 343 | case "hr": |
| 344 | break; |
| 345 | } |
| 346 | } |
| 347 | return out.filter((line) => line.trim() !== ""); |
| 348 | } |
| 349 | |
| 350 | /** |
| 351 | * A message as plain words on one line, for a preview: the Markdown's |
| 352 | * marks gone, a link as its text, a mention as `@name`. Cut to `max` |
| 353 | * characters with an ellipsis. |
| 354 | */ |
| 355 | export function plainText(markdown: string, max = Infinity): string { |
| 356 | const text = blocksText(blocks(markdown)).join(" ").replace(/\s+/g, " ").trim(); |
| 357 | return text.length > max ? `${text.slice(0, max - 1).trimEnd()}…` : text; |
| 358 | } |