Pick any line to see why it is the way it is: the commit, the pull request and issue it came from, and what the agent was thinking.
| The artifacts service is services/artifacts, the Worker g1t-artifacts, bound as ARTIFACTS by the API, the site and the agents; its live rooms move to it with a Durable Object transfer from g1t-docs-service, and its database, bucket, indexes and queue keep their names. The git store's binding and settings are GITSTORE, its ops scripts gitstore-*, and workflow run artifacts keep their compatible API under run_artifacts modules. The deploy tool puts a Worker that has never deployed before the Workers in its stage that bind to it, and the deploy guide gives the cutover runbook. | 1 | /** |
| 2 | * Passages: a page's Markdown (or a repository's docs file) split into | |
| 3 | * sections an agent can be handed whole, for the semantic index | |
| 4 | * (src/indexer.ts) and recall. Pure. | |
| 5 | * | |
| 6 | * A passage is one section under its heading path ("Runbook › Rollback"), | |
| 7 | * between MIN_CHARS and MAX_CHARS where it can be: long sections split on | |
| 8 | * paragraph boundaries (a fenced code block stays one paragraph), tiny ones | |
| 9 | * join the next. What is embedded is the passage with its document's title | |
| 10 | * and heading in front (`embedText`); what an agent reads is the passage. | |
| 11 | */ | |
| 12 | ||
| 13 | /** A passage shorter than this joins the next one. */ | |
| 14 | export const MIN_CHARS = 300; | |
| 15 | /** A passage longer than this splits on paragraphs. */ | |
| 16 | export const MAX_CHARS = 1500; | |
| 17 | /** The most passages one document keeps; past it the rest goes unindexed (words still find the page). */ | |
| 18 | export const MAX_CHUNKS = 150; | |
| 19 | /** The most text embedded for one passage: the model reads 512 tokens. */ | |
| 20 | export const EMBED_CHARS = 2000; | |
| 21 | /** Between heading levels in a passage's heading path. */ | |
| 22 | export const HEADING_SEPARATOR = " › "; | |
| 23 | ||
| 24 | export type Chunk = { | |
| 25 | seq: number; | |
| 26 | /** The heading path the passage sits under, or null before the first heading. */ | |
| 27 | heading: string | null; | |
| 28 | /** The passage as Markdown. */ | |
| 29 | text: string; | |
| 30 | }; | |
| 31 | ||
| 32 | type Section = { heading: string | null; lines: string[] }; | |
| 33 | ||
| 34 | const FENCE = /^\s{0,3}(`{3,}|~{3,})/; | |
| 35 | const HEADING = /^\s{0,3}(#{1,6})\s+(.+?)\s*#*\s*$/; | |
| 36 | ||
| 37 | /** A heading's text as plain words: no emphasis, code ticks or links. */ | |
| 38 | function headingText(raw: string): string { | |
| 39 | return raw | |
| 40 | .replace(/!\[([^\]]*)\]\([^)]*\)/g, "$1") | |
| 41 | .replace(/\[([^\]]*)\]\([^)]*\)/g, "$1") | |
| 42 | .replace(/[*_`~]+/g, "") | |
| 43 | .replace(/\s+/g, " ") | |
| 44 | .trim(); | |
| 45 | } | |
| 46 | ||
| 47 | /** Drops YAML front matter, as repository docs files often have. */ | |
| 48 | function withoutFrontMatter(markdown: string): string { | |
| 49 | const front = /^---\r?\n[\s\S]*?\r?\n---[ \t]*(\r?\n|$)/.exec(markdown); | |
| 50 | return front ? markdown.slice(front[0].length) : markdown; | |
| 51 | } | |
| 52 | ||
| 53 | /** The document split at its headings, each section with its heading path. */ | |
| 54 | function sections(markdown: string, title: string): Section[] { | |
| 55 | const lines = withoutFrontMatter(markdown).replace(/\r\n?/g, "\n").split("\n"); | |
| 56 | const out: Section[] = [{ heading: null, lines: [] }]; | |
| 57 | const path: { level: number; text: string }[] = []; | |
| 58 | let fence: string | null = null; | |
| 59 | let first = true; | |
| 60 | const wantTitle = headingText(title).toLowerCase(); | |
| 61 | for (const line of lines) { | |
| 62 | const f = FENCE.exec(line); | |
| 63 | if (fence) { | |
| 64 | if (f && f[1]![0] === fence[0] && f[1]!.length >= fence.length && line.trim() === f[1]) fence = null; | |
| 65 | out[out.length - 1]!.lines.push(line); | |
| 66 | continue; | |
| 67 | } | |
| 68 | if (f) { | |
| 69 | fence = f[1]!; | |
| 70 | out[out.length - 1]!.lines.push(line); | |
| 71 | continue; | |
| 72 | } | |
| 73 | const h = HEADING.exec(line); | |
| 74 | if (!h) { | |
| 75 | if (line.trim()) first = false; | |
| 76 | out[out.length - 1]!.lines.push(line); | |
| 77 | continue; | |
| 78 | } | |
| 79 | const level = h[1]!.length; | |
| 80 | const text = headingText(h[2]!); | |
| 81 | // A file's leading `# Title` is its title, not a section of it. | |
| 82 | if (first && level === 1 && text.toLowerCase() === wantTitle) { | |
| 83 | first = false; | |
| 84 | continue; | |
| 85 | } | |
| 86 | first = false; | |
| 87 | while (path.length && path[path.length - 1]!.level >= level) path.pop(); | |
| 88 | path.push({ level, text }); | |
| 89 | out.push({ heading: path.map((p) => p.text).filter(Boolean).join(HEADING_SEPARATOR) || null, lines: [] }); | |
| 90 | } | |
| 91 | return out; | |
| 92 | } | |
| 93 | ||
| 94 | /** Paragraphs: runs of lines between blank lines, with a fenced block kept whole. */ | |
| 95 | function paragraphs(lines: string[]): string[] { | |
| 96 | const out: string[] = []; | |
| 97 | let current: string[] = []; | |
| 98 | let fence: string | null = null; | |
| 99 | const flush = () => { | |
| 100 | const text = current.join("\n").trim(); | |
| 101 | if (text) out.push(text); | |
| 102 | current = []; | |
| 103 | }; | |
| 104 | for (const line of lines) { | |
| 105 | const f = FENCE.exec(line); | |
| 106 | if (fence) { | |
| 107 | current.push(line); | |
| 108 | if (f && f[1]![0] === fence[0] && f[1]!.length >= fence.length && line.trim() === f[1]) fence = null; | |
| 109 | continue; | |
| 110 | } | |
| 111 | if (f) { | |
| 112 | fence = f[1]!; | |
| 113 | current.push(line); | |
| 114 | continue; | |
| 115 | } | |
| 116 | if (!line.trim()) flush(); | |
| 117 | else current.push(line); | |
| 118 | } | |
| 119 | flush(); | |
| 120 | return out; | |
| 121 | } | |
| 122 | ||
| 123 | /** A paragraph longer than MAX_CHARS, cut at sentence ends or spaces. */ | |
| 124 | function cut(text: string): string[] { | |
| 125 | const out: string[] = []; | |
| 126 | let rest = text; | |
| 127 | while (rest.length > MAX_CHARS) { | |
| 128 | const window = rest.slice(0, MAX_CHARS); | |
| 129 | let at = Math.max(window.lastIndexOf(". "), window.lastIndexOf(".\n"), window.lastIndexOf("\n")); | |
| 130 | if (at < MAX_CHARS / 2) at = window.lastIndexOf(" "); | |
| 131 | if (at < MAX_CHARS / 2) at = MAX_CHARS - 1; | |
| 132 | out.push(rest.slice(0, at + 1).trim()); | |
| 133 | rest = rest.slice(at + 1).trim(); | |
| 134 | } | |
| 135 | if (rest) out.push(rest); | |
| 136 | return out; | |
| 137 | } | |
| 138 | ||
| 139 | /** One section's passages: its paragraphs packed up to MAX_CHARS. */ | |
| 140 | function pack(paras: string[]): string[] { | |
| 141 | const out: string[] = []; | |
| 142 | let current = ""; | |
| 143 | for (const para of paras.flatMap(cut)) { | |
| 144 | if (!current) current = para; | |
| 145 | else if (current.length + 2 + para.length <= MAX_CHARS) current += `\n\n${para}`; | |
| 146 | else { | |
| 147 | out.push(current); | |
| 148 | current = para; | |
| 149 | } | |
| 150 | } | |
| 151 | if (current) out.push(current); | |
| 152 | return out; | |
| 153 | } | |
| 154 | ||
| 155 | /** | |
| 156 | * A document's passages, in order. `title` is the page's or file's title: | |
| 157 | * not part of any passage, but a file's leading `# Title` is dropped as it. | |
| 158 | */ | |
| 159 | export function chunkMarkdown(markdown: string, title = ""): Chunk[] { | |
| 160 | const pieces: { heading: string | null; text: string }[] = []; | |
| 161 | for (const section of sections(String(markdown ?? ""), title)) { | |
| 162 | for (const text of pack(paragraphs(section.lines))) pieces.push({ heading: section.heading, text }); | |
| 163 | // A heading with nothing under it but subsections still names them: nothing to keep alone. | |
| 164 | } | |
| 165 | // Tiny passages join the next one (its heading written in), while that fits. | |
| 166 | const merged: { heading: string | null; text: string }[] = []; | |
| 167 | for (const piece of pieces) { | |
| 168 | const last = merged[merged.length - 1]; | |
| 169 | if (last && last.text.length < MIN_CHARS) { | |
| 170 | const joined = piece.heading && piece.heading !== last.heading ? `${last.text}\n\n**${piece.heading.split(HEADING_SEPARATOR).pop()}**\n\n${piece.text}` : `${last.text}\n\n${piece.text}`; | |
| 171 | if (joined.length <= MAX_CHARS) { | |
| 172 | last.text = joined; | |
| 173 | if (!last.heading) last.heading = piece.heading; | |
| 174 | continue; | |
| 175 | } | |
| 176 | } | |
| 177 | merged.push({ ...piece }); | |
| 178 | } | |
| 179 | // A tiny last passage joins the one before it, when that fits. | |
| 180 | if (merged.length > 1) { | |
| 181 | const last = merged[merged.length - 1]!; | |
| 182 | const before = merged[merged.length - 2]!; | |
| 183 | if (last.text.length < MIN_CHARS && before.text.length + last.text.length + 40 <= MAX_CHARS) { | |
| 184 | before.text = last.heading && last.heading !== before.heading ? `${before.text}\n\n**${last.heading.split(HEADING_SEPARATOR).pop()}**\n\n${last.text}` : `${before.text}\n\n${last.text}`; | |
| 185 | merged.pop(); | |
| 186 | } | |
| 187 | } | |
| 188 | return merged.slice(0, MAX_CHUNKS).map((piece, seq) => ({ seq, heading: piece.heading, text: piece.text })); | |
| 189 | } | |
| 190 | ||
| 191 | /** What is embedded for a passage: its document's title and heading in front, so a passage alone still says what it is about. */ | |
| 192 | export function embedText(title: string, chunk: Pick<Chunk, "heading" | "text">): string { | |
| 193 | const head = [title.trim(), chunk.heading ?? ""].filter(Boolean).join(HEADING_SEPARATOR); | |
| 194 | return (head ? `${head}\n\n${chunk.text}` : chunk.text).slice(0, EMBED_CHARS); | |
| 195 | } | |
| 196 | ||
| 197 | /** A short, stable hash (hex) of a string: whether a passage changed since it was embedded. Not for security. */ | |
| 198 | export function textHash(text: string): string { | |
| 199 | let h1 = 0xdeadbeef; | |
| 200 | let h2 = 0x41c6ce57; | |
| 201 | for (let i = 0; i < text.length; i++) { | |
| 202 | const c = text.charCodeAt(i); | |
| 203 | h1 = Math.imul(h1 ^ c, 2654435761); | |
| 204 | h2 = Math.imul(h2 ^ c, 1597334677); | |
| 205 | } | |
| 206 | h1 = Math.imul(h1 ^ (h1 >>> 16), 2246822507) ^ Math.imul(h2 ^ (h2 >>> 13), 3266489909); | |
| 207 | h2 = Math.imul(h2 ^ (h2 >>> 16), 2246822507) ^ Math.imul(h1 ^ (h1 >>> 13), 3266489909); | |
| 208 | return (h2 >>> 0).toString(16).padStart(8, "0") + (h1 >>> 0).toString(16).padStart(8, "0"); | |
| 209 | } | |
| 210 | ||
| 211 | /** A repository docs file's id in the index (`rf_` and 32 hex): short enough for a vector id, the same each time. */ | |
| 212 | export function repoFileId(spaceId: string, path: string): string { | |
| 213 | return `rf_${textHash(`${spaceId}\n${path}`)}${textHash(`${path}\n${spaceId}`)}`; | |
| 214 | } | |
| 215 | ||
| 216 | /** A passage's id, in D1 and in the vector index: `<page or file id>:<seq>`. */ | |
| 217 | export function chunkId(docId: string, seq: number): string { | |
| 218 | return `${docId}:${seq}`; | |
| 219 | } |