| 1 | /** |
| 2 | * A page's Markdown, derived from its Yjs document: what search indexes, |
| 3 | * agents read, export writes and the read view renders. Pure. |
| 4 | * |
| 5 | * The document is BlockNote's, as y-prosemirror stores it: the fragment |
| 6 | * holds one `blockGroup` of `blockContainer`s (each with an `id`), each |
| 7 | * holding one content node (`paragraph`, `heading`, `codeBlock`, ...) and, |
| 8 | * for nested blocks, a `blockGroup` of children. Inline text is an |
| 9 | * `XmlText` whose formatting attributes are the marks (`bold: {}`, |
| 10 | * `link: { href }`, `comment--<hash>: {...}`); inline nodes (mentions, |
| 11 | * dates) are `XmlElement`s between texts. |
| 12 | * |
| 13 | * g1t's own blocks (the editor's schema, apps/web components/docs): |
| 14 | * `callout` (`kind`: info | warning | success | danger), `mermaid` |
| 15 | * (`code`), `math` (`expression`), `embed` (`kind`, `title`, `url`), and |
| 16 | * inline `mention` (`kind`: user | agent | page; `id`: a person's |
| 17 | * username, an agent's id or a page's id; `name`; `href` for a page), |
| 18 | * `date` (`date`) and `citation` (`repo`, `path`, `kind`, `label`, `ref`: |
| 19 | * code the page cites, src/citations.ts). |
| 20 | */ |
| 21 | import type { DocCitation } from "@g1t/contracts"; |
| 22 | import * as Y from "yjs"; |
| 23 | |
| 24 | import { citationMarkdown, cleanCitation } from "./citations.ts"; |
| 25 | |
| 26 | export type Outline = { id: string; type: string; level: number | null; markdown: string }; |
| 27 | |
| 28 | /** The callout kinds, as GitHub's alert names. */ |
| 29 | export const CALLOUT_ALERTS: Record<string, string> = { info: "NOTE", warning: "WARNING", success: "TIP", danger: "CAUTION" }; |
| 30 | |
| 31 | /** The top-level blocks: the containers in the fragment's first block group. */ |
| 32 | export function topContainers(fragment: Y.XmlFragment): Y.XmlElement[] { |
| 33 | const group = fragment.toArray().find((n): n is Y.XmlElement => n instanceof Y.XmlElement && n.nodeName === "blockGroup"); |
| 34 | if (!group) return []; |
| 35 | return group.toArray().filter((n): n is Y.XmlElement => n instanceof Y.XmlElement && n.nodeName === "blockContainer"); |
| 36 | } |
| 37 | |
| 38 | /** A container's content node and its children's group. */ |
| 39 | export function partsOf(container: Y.XmlElement): { content: Y.XmlElement | null; children: Y.XmlElement[] } { |
| 40 | let content: Y.XmlElement | null = null; |
| 41 | let children: Y.XmlElement[] = []; |
| 42 | for (const node of container.toArray()) { |
| 43 | if (!(node instanceof Y.XmlElement)) continue; |
| 44 | if (node.nodeName === "blockGroup") children = node.toArray().filter((n): n is Y.XmlElement => n instanceof Y.XmlElement && n.nodeName === "blockContainer"); |
| 45 | else if (!content) content = node; |
| 46 | } |
| 47 | return { content, children }; |
| 48 | } |
| 49 | |
| 50 | type Delta = { insert: string | object; attributes?: Record<string, unknown> }; |
| 51 | |
| 52 | /** A text run with its marks applied, whitespace kept outside the markers. */ |
| 53 | function marked(text: string, attrs: Record<string, unknown> | undefined): string { |
| 54 | if (!attrs || !text) return text; |
| 55 | const keys = Object.keys(attrs).map((k) => k.replace(/--[a-zA-Z0-9+/=]{8}$/, "")); |
| 56 | if (attrs.code !== undefined && attrs.code !== null) { |
| 57 | const ticks = text.includes("`") ? "``" : "`"; |
| 58 | return `${ticks}${text}${ticks}`; |
| 59 | } |
| 60 | const lead = /^\s*/.exec(text)![0]; |
| 61 | const trail = /\s*$/.exec(text)![0]; |
| 62 | let core = text.slice(lead.length, text.length - trail.length); |
| 63 | if (!core) return text; |
| 64 | if (keys.includes("bold")) core = `**${core}**`; |
| 65 | if (keys.includes("italic")) core = `_${core}_`; |
| 66 | if (keys.includes("strike")) core = `~~${core}~~`; |
| 67 | if (keys.includes("underline")) core = `<u>${core}</u>`; |
| 68 | const link = attrs.link as { href?: string } | undefined; |
| 69 | if (link?.href) core = `[${core}](${link.href})`; |
| 70 | return lead + core + trail; |
| 71 | } |
| 72 | |
| 73 | /** An inline node (mention, date) as text. */ |
| 74 | function inlineNode(node: Y.XmlElement): string { |
| 75 | const a = node.getAttributes() as Record<string, string | undefined>; |
| 76 | if (node.nodeName === "mention") { |
| 77 | if (a.kind === "page") return a.href ? `[${a.name || "Untitled"}](${a.href})` : `[[${a.name || "Untitled"}]]`; |
| 78 | return `@${a.name ?? ""}`; |
| 79 | } |
| 80 | if (node.nodeName === "date") return a.date ?? ""; |
| 81 | if (node.nodeName === "citation") { |
| 82 | const c = cleanCitation(a as Partial<DocCitation>, "body"); |
| 83 | return c ? citationMarkdown(c) : ""; |
| 84 | } |
| 85 | // An unknown inline node: its text, if any. |
| 86 | return node.toArray().map((c) => (c instanceof Y.XmlText ? c.toString() : "")).join(""); |
| 87 | } |
| 88 | |
| 89 | /** A content node's inline content as Markdown. */ |
| 90 | export function inlineMarkdown(node: Y.XmlElement): string { |
| 91 | let out = ""; |
| 92 | for (const child of node.toArray()) { |
| 93 | if (child instanceof Y.XmlText) { |
| 94 | for (const d of child.toDelta() as Delta[]) { |
| 95 | if (typeof d.insert === "string") out += marked(d.insert, d.attributes); |
| 96 | } |
| 97 | } else if (child instanceof Y.XmlElement) { |
| 98 | out += inlineNode(child); |
| 99 | } |
| 100 | } |
| 101 | // Hard breaks inside a paragraph. |
| 102 | return out.replace(/\n/g, "\\\n"); |
| 103 | } |
| 104 | |
| 105 | /** A content node's plain text. */ |
| 106 | export function plainText(node: Y.XmlElement): string { |
| 107 | let out = ""; |
| 108 | for (const child of node.toArray()) { |
| 109 | if (child instanceof Y.XmlText) out += child.toString().replace(/<[^>]+>/g, ""); |
| 110 | else if (child instanceof Y.XmlElement) out += inlineNode(child).replace(/\[([^\]]*)\]\([^)]*\)/g, "$1"); |
| 111 | } |
| 112 | return out; |
| 113 | } |
| 114 | |
| 115 | function codeText(node: Y.XmlElement): string { |
| 116 | return node |
| 117 | .toArray() |
| 118 | .map((c) => (c instanceof Y.XmlText ? (c.toDelta() as Delta[]).map((d) => (typeof d.insert === "string" ? d.insert : "")).join("") : "")) |
| 119 | .join(""); |
| 120 | } |
| 121 | |
| 122 | function fence(body: string, info: string): string { |
| 123 | const ticks = body.includes("```") ? "````" : "```"; |
| 124 | return `${ticks}${info}\n${body}\n${ticks}`; |
| 125 | } |
| 126 | |
| 127 | function table(node: Y.XmlElement): string { |
| 128 | const rows = node |
| 129 | .toArray() |
| 130 | .filter((r): r is Y.XmlElement => r instanceof Y.XmlElement && r.nodeName === "tableRow") |
| 131 | .map((row) => |
| 132 | row |
| 133 | .toArray() |
| 134 | .filter((c): c is Y.XmlElement => c instanceof Y.XmlElement) |
| 135 | .map((cell) => |
| 136 | cell |
| 137 | .toArray() |
| 138 | .filter((p): p is Y.XmlElement => p instanceof Y.XmlElement) |
| 139 | .map((p) => inlineMarkdown(p).replace(/\\\n/g, " ")) |
| 140 | .join(" ") |
| 141 | .replace(/\|/g, "\\|"), |
| 142 | ), |
| 143 | ); |
| 144 | if (!rows.length) return ""; |
| 145 | const width = Math.max(...rows.map((r) => r.length)); |
| 146 | const line = (cells: string[]) => `| ${Array.from({ length: width }, (_, i) => cells[i] ?? "").join(" | ")} |`; |
| 147 | return [line(rows[0]!), `| ${Array.from({ length: width }, () => "---").join(" | ")} |`, ...rows.slice(1).map(line)].join("\n"); |
| 148 | } |
| 149 | |
| 150 | const LIST = new Set(["bulletListItem", "numberedListItem", "checkListItem"]); |
| 151 | |
| 152 | function indent(text: string, by: string): string { |
| 153 | return text |
| 154 | .split("\n") |
| 155 | .map((l) => (l ? by + l : l)) |
| 156 | .join("\n"); |
| 157 | } |
| 158 | |
| 159 | /** |
| 160 | * One block (and its children) as Markdown. `number` is a numbered item's |
| 161 | * place in its run of numbered items. |
| 162 | */ |
| 163 | export function blockMarkdown(container: Y.XmlElement, number = 1): string { |
| 164 | const { content, children } = partsOf(container); |
| 165 | if (!content) return ""; |
| 166 | const a = content.getAttributes() as Record<string, unknown>; |
| 167 | const kids = () => blocksMarkdown(children); |
| 168 | const nested = (prefix: string) => { |
| 169 | const body = kids(); |
| 170 | return body ? `\n${indent(body, " ".repeat(prefix.length))}` : ""; |
| 171 | }; |
| 172 | switch (content.nodeName) { |
| 173 | case "heading": { |
| 174 | const level = Math.min(Math.max(Number(a.level) || 1, 1), 6); |
| 175 | const own = `${"#".repeat(level)} ${inlineMarkdown(content)}`; |
| 176 | const body = kids(); |
| 177 | return body ? `${own}\n\n${body}` : own; |
| 178 | } |
| 179 | case "bulletListItem": |
| 180 | return `- ${inlineMarkdown(content)}${nested("- ")}`; |
| 181 | case "numberedListItem": { |
| 182 | const prefix = `${number}. `; |
| 183 | return `${prefix}${inlineMarkdown(content)}${nested(prefix)}`; |
| 184 | } |
| 185 | case "checkListItem": { |
| 186 | const checked = a.checked === true || a.checked === "true"; |
| 187 | return `- [${checked ? "x" : " "}] ${inlineMarkdown(content)}${nested("- ")}`; |
| 188 | } |
| 189 | case "toggleListItem": { |
| 190 | const body = kids(); |
| 191 | return `<details>\n<summary>${inlineMarkdown(content)}</summary>\n${body ? `\n${body}\n` : ""}\n</details>`; |
| 192 | } |
| 193 | case "quote": |
| 194 | return `${indent(inlineMarkdown(content), "> ").replace(/^$/gm, ">")}${children.length ? `\n\n${kids()}` : ""}`; |
| 195 | case "callout": { |
| 196 | const alert = CALLOUT_ALERTS[String(a.kind ?? "info")] ?? "NOTE"; |
| 197 | return `> [!${alert}]\n${indent(inlineMarkdown(content) || " ", "> ")}${children.length ? `\n\n${kids()}` : ""}`; |
| 198 | } |
| 199 | case "codeBlock": |
| 200 | return fence(codeText(content), String(a.language ?? "") === "text" ? "" : String(a.language ?? "")); |
| 201 | case "mermaid": |
| 202 | return fence(String(a.code ?? ""), "mermaid"); |
| 203 | case "math": |
| 204 | return `$$\n${String(a.expression ?? "")}\n$$`; |
| 205 | case "divider": |
| 206 | return "---"; |
| 207 | case "pageBreak": |
| 208 | return "---"; |
| 209 | case "image": { |
| 210 | const url = String(a.url ?? ""); |
| 211 | if (!url) return ""; |
| 212 | return ``; |
| 213 | } |
| 214 | case "video": |
| 215 | case "audio": |
| 216 | case "file": { |
| 217 | const url = String(a.url ?? ""); |
| 218 | if (!url) return ""; |
| 219 | return `[${String(a.name || a.caption || url)}](${url})`; |
| 220 | } |
| 221 | case "embed": { |
| 222 | const url = String(a.url ?? ""); |
| 223 | const title = String(a.title ?? "") || url; |
| 224 | return url ? `[${title}](${url})` : title; |
| 225 | } |
| 226 | case "table": |
| 227 | return table(content); |
| 228 | case "paragraph": |
| 229 | default: { |
| 230 | const own = inlineMarkdown(content); |
| 231 | const body = kids(); |
| 232 | if (!own) return body; |
| 233 | return body ? `${own}\n\n${body}` : own; |
| 234 | } |
| 235 | } |
| 236 | } |
| 237 | |
| 238 | /** Sibling blocks as Markdown: list items close together, other blocks a blank line apart. */ |
| 239 | export function blocksMarkdown(containers: Y.XmlElement[]): string { |
| 240 | const parts: string[] = []; |
| 241 | let prevType: string | null = null; |
| 242 | let number = 0; |
| 243 | for (const container of containers) { |
| 244 | const type = partsOf(container).content?.nodeName ?? ""; |
| 245 | number = type === "numberedListItem" ? (prevType === "numberedListItem" ? number + 1 : Number(partsOf(container).content?.getAttribute("start")) || 1) : 0; |
| 246 | const text = blockMarkdown(container, number); |
| 247 | if (text === "" && type === "paragraph") { |
| 248 | prevType = type; |
| 249 | continue; |
| 250 | } |
| 251 | // One list: items of the same kind (bullets and to-dos mix), close together. |
| 252 | const family = (t: string) => (t === "numberedListItem" ? "numbered" : "bullet"); |
| 253 | const tight = prevType !== null && LIST.has(type) && LIST.has(prevType) && family(type) === family(prevType); |
| 254 | parts.push((parts.length ? (tight ? "\n" : "\n\n") : "") + text); |
| 255 | prevType = type; |
| 256 | } |
| 257 | return parts.join(""); |
| 258 | } |
| 259 | |
| 260 | /** The whole document as Markdown. */ |
| 261 | export function documentMarkdown(fragment: Y.XmlFragment): string { |
| 262 | const text = blocksMarkdown(topContainers(fragment)).trim(); |
| 263 | return text ? `${text}\n` : ""; |
| 264 | } |
| 265 | |
| 266 | /** The top-level blocks, as agents see them. */ |
| 267 | export function outline(fragment: Y.XmlFragment): Outline[] { |
| 268 | const out: Outline[] = []; |
| 269 | const tops = topContainers(fragment); |
| 270 | let number = 0; |
| 271 | let prev: string | null = null; |
| 272 | for (const container of tops) { |
| 273 | const content = partsOf(container).content; |
| 274 | const type = content?.nodeName ?? "paragraph"; |
| 275 | number = type === "numberedListItem" ? (prev === "numberedListItem" ? number + 1 : 1) : 0; |
| 276 | prev = type; |
| 277 | out.push({ |
| 278 | id: String(container.getAttribute("id") ?? ""), |
| 279 | type, |
| 280 | level: type === "heading" ? Number(content?.getAttribute("level")) || 1 : null, |
| 281 | markdown: blockMarkdown(container, number), |
| 282 | }); |
| 283 | } |
| 284 | return out; |
| 285 | } |
| 286 | |
| 287 | /** Plain text for search: Markdown without its markup. */ |
| 288 | export function searchText(markdown: string): string { |
| 289 | return markdown |
| 290 | .replace(/```[^\n]*\n/g, "") |
| 291 | .replace(/```/g, "") |
| 292 | .replace(/!\[([^\]]*)\]\([^)]*\)/g, "$1") |
| 293 | .replace(/\[([^\]]*)\]\([^)]*\)/g, "$1") |
| 294 | .replace(/<\/?(details|summary|u)>/g, " ") |
| 295 | .replace(/^\s{0,3}(#{1,6}|>|[-*+]|\d+\.)\s+(\[[ x]\]\s+)?/gm, "") |
| 296 | .replace(/\[!(NOTE|WARNING|TIP|CAUTION|IMPORTANT)\]/g, "") |
| 297 | .replace(/[*_~`|]+/g, " ") |
| 298 | .replace(/[ \t]+/g, " ") |
| 299 | .replace(/\n{2,}/g, "\n") |
| 300 | .trim(); |
| 301 | } |
| 302 | |
| 303 | /** The first words of a page, for cards. */ |
| 304 | export function excerpt(markdown: string, max = 180): string { |
| 305 | const text = searchText(markdown).replace(/\s+/g, " ").trim(); |
| 306 | return text.length > max ? `${text.slice(0, max - 1).trimEnd()}…` : text; |
| 307 | } |
| 308 | |
| 309 | /** The people (usernames, lowercased) and agents (agent ids) a document mentions. */ |
| 310 | export function mentionedIds(fragment: Y.XmlFragment): { users: string[]; agents: string[] } { |
| 311 | const users = new Set<string>(); |
| 312 | const agents = new Set<string>(); |
| 313 | const walk = (node: Y.XmlElement | Y.XmlFragment) => { |
| 314 | for (const child of node.toArray()) { |
| 315 | if (!(child instanceof Y.XmlElement)) continue; |
| 316 | if (child.nodeName === "mention") { |
| 317 | const kind = child.getAttribute("kind"); |
| 318 | const id = String(child.getAttribute("id") ?? ""); |
| 319 | if (kind === "user" && id) users.add(id.toLowerCase()); |
| 320 | if (kind === "agent" && id) agents.add(id); |
| 321 | } else walk(child); |
| 322 | } |
| 323 | }; |
| 324 | walk(fragment); |
| 325 | return { users: [...users], agents: [...agents] }; |
| 326 | } |
| 327 | |
| 328 | /** The citation chips in a document, as their attributes say (src/citations.ts cleans them). */ |
| 329 | export function citationNodes(fragment: Y.XmlFragment): Partial<DocCitation>[] { |
| 330 | const out: Partial<DocCitation>[] = []; |
| 331 | const walk = (node: Y.XmlElement | Y.XmlFragment) => { |
| 332 | for (const child of node.toArray()) { |
| 333 | if (!(child instanceof Y.XmlElement)) continue; |
| 334 | if (child.nodeName === "citation") out.push(child.getAttributes() as Partial<DocCitation>); |
| 335 | else walk(child); |
| 336 | } |
| 337 | }; |
| 338 | walk(fragment); |
| 339 | return out; |
| 340 | } |