| 1 | /** |
| 2 | * The chat composer's document and the Markdown a message is stored as, |
| 3 | * each turned into the other. The composer (components/chat/editor.tsx) |
| 4 | * edits a small rich-text document; what it sends, keeps as a draft and |
| 5 | * edits again is Markdown, the same text the API, agents and notifications |
| 6 | * read. Pure, on the editor's JSON, so it is tested without a browser |
| 7 | * (chat-compose.test.ts). |
| 8 | * |
| 9 | * In the document each paragraph is one line of the message: Shift+Enter |
| 10 | * starts a new one, an empty one is a blank line. |
| 11 | */ |
| 12 | import { type Block, type Span, blocks } from "@g1t/contracts/chat-markdown"; |
| 13 | |
| 14 | /** The editor's JSON, as much of it as is read here. */ |
| 15 | export type DocNode = { |
| 16 | type: string; |
| 17 | attrs?: Record<string, unknown>; |
| 18 | content?: DocNode[]; |
| 19 | text?: string; |
| 20 | marks?: { type: string; attrs?: Record<string, unknown> }[]; |
| 21 | }; |
| 22 | |
| 23 | type Mark = { type: string; attrs?: Record<string, unknown> }; |
| 24 | |
| 25 | // --------------------------------------------------------------------------- |
| 26 | // Markdown to the document. |
| 27 | |
| 28 | function textNode(text: string, marks: Mark[]): DocNode | null { |
| 29 | if (!text) return null; |
| 30 | return marks.length ? { type: "text", text, marks } : { type: "text", text }; |
| 31 | } |
| 32 | |
| 33 | function spanNodes(list: Span[], marks: Mark[] = []): DocNode[] { |
| 34 | const out: DocNode[] = []; |
| 35 | const add = (node: DocNode | null) => node && out.push(node); |
| 36 | for (const span of list) { |
| 37 | switch (span.t) { |
| 38 | case "text": |
| 39 | add(textNode(span.v, marks)); |
| 40 | break; |
| 41 | case "code": |
| 42 | add(textNode(span.v, [...marks, { type: "code" }])); |
| 43 | break; |
| 44 | case "strong": |
| 45 | out.push(...spanNodes(span.c, [...marks, { type: "bold" }])); |
| 46 | break; |
| 47 | case "em": |
| 48 | out.push(...spanNodes(span.c, [...marks, { type: "italic" }])); |
| 49 | break; |
| 50 | case "del": |
| 51 | out.push(...spanNodes(span.c, [...marks, { type: "strike" }])); |
| 52 | break; |
| 53 | case "link": |
| 54 | out.push(...spanNodes(span.c, [{ type: "link", attrs: { href: span.href } }, ...marks])); |
| 55 | break; |
| 56 | case "mention": |
| 57 | add(textNode(`@${span.name}`, marks)); |
| 58 | break; |
| 59 | case "channel": |
| 60 | add(textNode(`#${span.name}`, marks)); |
| 61 | break; |
| 62 | case "ref": |
| 63 | add(textNode(`${span.repo ?? ""}#${span.number}`, marks)); |
| 64 | break; |
| 65 | } |
| 66 | } |
| 67 | return out; |
| 68 | } |
| 69 | |
| 70 | function paragraph(spans: Span[]): DocNode { |
| 71 | const content = spanNodes(spans); |
| 72 | return content.length ? { type: "paragraph", content } : { type: "paragraph" }; |
| 73 | } |
| 74 | |
| 75 | function blockNodes(list: Block[]): DocNode[] { |
| 76 | const out: DocNode[] = []; |
| 77 | list.forEach((block, index) => { |
| 78 | // Paragraphs one after another were apart by a blank line: an empty line between them. |
| 79 | if (block.t === "p" && index > 0 && list[index - 1]!.t === "p") out.push({ type: "paragraph" }); |
| 80 | switch (block.t) { |
| 81 | case "p": |
| 82 | for (const line of block.lines) out.push(paragraph(line)); |
| 83 | break; |
| 84 | case "heading": |
| 85 | // Chat has no headings to write; one comes back as a bold line. |
| 86 | out.push(paragraph([{ t: "strong", c: block.c }])); |
| 87 | break; |
| 88 | case "hr": |
| 89 | break; |
| 90 | case "code": |
| 91 | out.push({ type: "codeBlock", attrs: { language: block.lang }, ...(block.v ? { content: [{ type: "text", text: block.v }] } : {}) }); |
| 92 | break; |
| 93 | case "quote": { |
| 94 | const content = blockNodes(block.c); |
| 95 | out.push({ type: "blockquote", content: content.length ? content : [{ type: "paragraph" }] }); |
| 96 | break; |
| 97 | } |
| 98 | case "list": |
| 99 | out.push({ |
| 100 | type: block.ordered ? "orderedList" : "bulletList", |
| 101 | ...(block.ordered ? { attrs: { start: block.start } } : {}), |
| 102 | content: block.items.map((item) => { |
| 103 | const content = blockNodes(item); |
| 104 | // An item starts with a line of text. |
| 105 | if (content[0]?.type !== "paragraph") content.unshift({ type: "paragraph" }); |
| 106 | return { type: "listItem", content }; |
| 107 | }), |
| 108 | }); |
| 109 | break; |
| 110 | } |
| 111 | }); |
| 112 | return out; |
| 113 | } |
| 114 | |
| 115 | /** A message's Markdown as the composer's document. */ |
| 116 | export function markdownToDoc(markdown: string): DocNode { |
| 117 | const content = blockNodes(blocks(markdown)); |
| 118 | return { type: "doc", content: content.length ? content : [{ type: "paragraph" }] }; |
| 119 | } |
| 120 | |
| 121 | // --------------------------------------------------------------------------- |
| 122 | // The document to Markdown. |
| 123 | |
| 124 | /** Marks from the outside in: a link holds bold, bold holds code. */ |
| 125 | const ORDER = ["link", "bold", "italic", "strike", "code"]; |
| 126 | const OPEN: Record<string, string> = { bold: "**", italic: "_", strike: "~~" }; |
| 127 | |
| 128 | function sameMark(a: Mark, b: Mark): boolean { |
| 129 | return a.type === b.type && (a.type !== "link" || a.attrs?.href === b.attrs?.href); |
| 130 | } |
| 131 | |
| 132 | /** Text as Markdown shows it as typed: its marks escaped. */ |
| 133 | export function escapeText(text: string): string { |
| 134 | return text |
| 135 | .replace(/\\(?=[\\`*_{}[\]()#+\-.!~>|<=:@])/g, "\\\\") |
| 136 | .replace(/[`*~[\]]/g, (char) => `\\${char}`) |
| 137 | .replace(/_/g, (char, at: number, all: string) => { |
| 138 | // `snake_case` stays as typed: only an `_` at a word's edge could start or end italics. |
| 139 | const before = all[at - 1] ?? ""; |
| 140 | const after = all[at + 1] ?? ""; |
| 141 | return /\w/.test(before) && /\w/.test(after) ? char : `\\${char}`; |
| 142 | }); |
| 143 | } |
| 144 | |
| 145 | /** A line that would read as a block's start (`# `, `- `, `1. `, `> `, `---`) when it is only text. */ |
| 146 | function escapeLineStart(line: string): string { |
| 147 | return line |
| 148 | .replace(/^(\s*)(#{1,6}(?:\s|$))/, "$1\\$2") |
| 149 | .replace(/^(\s*)([-+](?:\s|$))/, "$1\\$2") |
| 150 | .replace(/^(\s*)(\d{1,9})([.)])(\s|$)/, "$1$2\\$3$4") |
| 151 | .replace(/^(\s*)(>)/, "$1\\$2") |
| 152 | .replace(/^(\s*)([-_=]{3,}\s*)$/, "$1\\$2") |
| 153 | .replace(/^(\s*)•/, "$1\\•"); |
| 154 | } |
| 155 | |
| 156 | function codeSpan(text: string): string { |
| 157 | if (!text.includes("`")) return `\`${text}\``; |
| 158 | return `\`\` ${text} \`\``; |
| 159 | } |
| 160 | |
| 161 | function linkTarget(href: string): string { |
| 162 | return /[\s()<>]/.test(href) ? `<${href.replace(/[<>\s]/g, encodeURIComponent)}>` : href; |
| 163 | } |
| 164 | |
| 165 | /** A line of inline content (text with marks) as Markdown. */ |
| 166 | function inlineMarkdown(nodes: DocNode[]): string { |
| 167 | let out = ""; |
| 168 | const open: { mark: Mark; text: string }[] = []; |
| 169 | // A link's text is gathered so a bare URL can be written bare. |
| 170 | const close = (count: number) => { |
| 171 | for (let n = 0; n < count; n++) { |
| 172 | const top = open.pop()!; |
| 173 | // Spaces at a mark's end go after its closing mark: `**a **` would not be bold. |
| 174 | const body = top.text; |
| 175 | const trailing = /\s*$/.exec(body)![0]; |
| 176 | const inner = body.slice(0, body.length - trailing.length); |
| 177 | let written: string; |
| 178 | if (top.mark.type === "link") { |
| 179 | const href = String(top.mark.attrs?.href ?? ""); |
| 180 | written = inner === escapeText(href) && /^https?:\/\//i.test(href) && !/[\s()<>]/.test(href) ? href : `[${inner}](${linkTarget(href)})`; |
| 181 | } else if (!inner) { |
| 182 | written = ""; |
| 183 | } else { |
| 184 | const delimiter = OPEN[top.mark.type] ?? ""; |
| 185 | written = `${delimiter}${inner}${delimiter}`; |
| 186 | } |
| 187 | append(written + trailing); |
| 188 | } |
| 189 | }; |
| 190 | const append = (text: string) => { |
| 191 | if (open.length) open[open.length - 1]!.text += text; |
| 192 | else out += text; |
| 193 | }; |
| 194 | for (const node of nodes) { |
| 195 | if (node.type === "hardBreak") { |
| 196 | close(open.length); |
| 197 | out += "\n"; |
| 198 | continue; |
| 199 | } |
| 200 | if (node.type !== "text" || !node.text) continue; |
| 201 | const marks = [...(node.marks ?? [])] |
| 202 | .filter((mark) => ORDER.includes(mark.type)) |
| 203 | .sort((a, b) => ORDER.indexOf(a.type) - ORDER.indexOf(b.type)); |
| 204 | // Close what this text no longer has, and everything opened after it. |
| 205 | let keep = 0; |
| 206 | while (keep < open.length && keep < marks.length && sameMark(open[keep]!.mark, marks[keep]!)) keep++; |
| 207 | close(open.length - keep); |
| 208 | const code = marks.some((mark) => mark.type === "code"); |
| 209 | let text = node.text; |
| 210 | // Spaces before a mark's start go before it: `** a**` would not be bold. |
| 211 | const opening = marks.slice(keep).filter((mark) => mark.type !== "code"); |
| 212 | if (opening.length) { |
| 213 | const leading = /^\s*/.exec(text)![0]; |
| 214 | append(leading); |
| 215 | text = text.slice(leading.length); |
| 216 | } |
| 217 | for (const mark of marks.slice(keep)) if (mark.type !== "code") open.push({ mark, text: "" }); |
| 218 | append(code ? codeSpan(text) : escapeText(text)); |
| 219 | } |
| 220 | close(open.length); |
| 221 | return out; |
| 222 | } |
| 223 | |
| 224 | function prefixLines(text: string, first: string, rest: string): string { |
| 225 | return text |
| 226 | .split("\n") |
| 227 | .map((line, index) => (index === 0 ? first : line ? rest : rest.trimEnd()) + line) |
| 228 | .join("\n"); |
| 229 | } |
| 230 | |
| 231 | function blockMarkdown(node: DocNode): string | null { |
| 232 | switch (node.type) { |
| 233 | case "paragraph": { |
| 234 | const line = inlineMarkdown(node.content ?? []); |
| 235 | return line |
| 236 | .split("\n") |
| 237 | .map((part) => escapeLineStart(part)) |
| 238 | .join("\n"); |
| 239 | } |
| 240 | case "codeBlock": { |
| 241 | const text = (node.content ?? []).map((child) => child.text ?? "").join(""); |
| 242 | const longest = Math.max(2, ...(text.match(/`{3,}/g) ?? []).map((run) => run.length)); |
| 243 | const fence = "`".repeat(longest + 1); |
| 244 | const language = typeof node.attrs?.language === "string" ? node.attrs.language : ""; |
| 245 | return `${fence}${language}\n${text}\n${fence}`; |
| 246 | } |
| 247 | case "blockquote": { |
| 248 | const inner = blocksMarkdown(node.content ?? []); |
| 249 | return inner |
| 250 | .split("\n") |
| 251 | .map((line) => (line ? `> ${line}` : ">")) |
| 252 | .join("\n"); |
| 253 | } |
| 254 | case "bulletList": |
| 255 | case "orderedList": { |
| 256 | const ordered = node.type === "orderedList"; |
| 257 | const start = ordered ? Number(node.attrs?.start ?? 1) || 1 : 1; |
| 258 | return (node.content ?? []) |
| 259 | .map((item, index) => { |
| 260 | const marker = ordered ? `${start + index}. ` : "- "; |
| 261 | const inner = blocksMarkdown(item.content ?? [], true); |
| 262 | return prefixLines(inner, marker, " ".repeat(marker.length)); |
| 263 | }) |
| 264 | .join("\n"); |
| 265 | } |
| 266 | default: |
| 267 | return null; |
| 268 | } |
| 269 | } |
| 270 | |
| 271 | function blocksMarkdown(nodes: DocNode[], tight = false): string { |
| 272 | let out = ""; |
| 273 | let previous: string | null = null; |
| 274 | for (const node of nodes) { |
| 275 | const text = blockMarkdown(node); |
| 276 | if (text == null) continue; |
| 277 | if (previous != null) { |
| 278 | // A line follows a line, and in a list item everything does; anything else stands apart by a blank line. |
| 279 | out += tight || (previous === "paragraph" && node.type === "paragraph") ? "\n" : "\n\n"; |
| 280 | } |
| 281 | out += text; |
| 282 | previous = node.type; |
| 283 | } |
| 284 | return out; |
| 285 | } |
| 286 | |
| 287 | /** The composer's document as the Markdown a message is sent as. */ |
| 288 | export function docToMarkdown(doc: DocNode): string { |
| 289 | return blocksMarkdown(doc.content ?? []) |
| 290 | .replace(/^(?:[ \t]*\n)+/, "") |
| 291 | .replace(/(?:\n[ \t]*)+$/, ""); |
| 292 | } |