Skip to content
219 linesCodeBlameRaw

Pick any line to see why it is the way it is: the commit, the pull request and issue it came from, and what the agent was thinking.

The artifacts service is services/artifacts, the Worker g1t-artifacts, bound as ARTIFACTS by the API, the site and the agents; its live rooms move to it with a Durable Object transfer from g1t-docs-service, and its database, bucket, indexes and queue keep their names. The git store's binding and settings are GITSTORE, its ops scripts gitstore-*, and workflow run artifacts keep their compatible API under run_artifacts modules. The deploy tool puts a Worker that has never deployed before the Workers in its stage that bind to it, and the deploy guide gives the cutover runbook.1/**
2 * Passages: a page's Markdown (or a repository's docs file) split into
3 * sections an agent can be handed whole, for the semantic index
4 * (src/indexer.ts) and recall. Pure.
5 *
6 * A passage is one section under its heading path ("Runbook › Rollback"),
7 * between MIN_CHARS and MAX_CHARS where it can be: long sections split on
8 * paragraph boundaries (a fenced code block stays one paragraph), tiny ones
9 * join the next. What is embedded is the passage with its document's title
10 * and heading in front (`embedText`); what an agent reads is the passage.
11 */
12
13/** A passage shorter than this joins the next one. */
14export const MIN_CHARS = 300;
15/** A passage longer than this splits on paragraphs. */
16export const MAX_CHARS = 1500;
17/** The most passages one document keeps; past it the rest goes unindexed (words still find the page). */
18export const MAX_CHUNKS = 150;
19/** The most text embedded for one passage: the model reads 512 tokens. */
20export const EMBED_CHARS = 2000;
21/** Between heading levels in a passage's heading path. */
22export const HEADING_SEPARATOR = " › ";
23
24export type Chunk = {
25 seq: number;
26 /** The heading path the passage sits under, or null before the first heading. */
27 heading: string | null;
28 /** The passage as Markdown. */
29 text: string;
30};
31
32type Section = { heading: string | null; lines: string[] };
33
34const FENCE = /^\s{0,3}(`{3,}|~{3,})/;
35const HEADING = /^\s{0,3}(#{1,6})\s+(.+?)\s*#*\s*$/;
36
37/** A heading's text as plain words: no emphasis, code ticks or links. */
38function headingText(raw: string): string {
39 return raw
40 .replace(/!\[([^\]]*)\]\([^)]*\)/g, "$1")
41 .replace(/\[([^\]]*)\]\([^)]*\)/g, "$1")
42 .replace(/[*_`~]+/g, "")
43 .replace(/\s+/g, " ")
44 .trim();
45}
46
47/** Drops YAML front matter, as repository docs files often have. */
48function withoutFrontMatter(markdown: string): string {
49 const front = /^---\r?\n[\s\S]*?\r?\n---[ \t]*(\r?\n|$)/.exec(markdown);
50 return front ? markdown.slice(front[0].length) : markdown;
51}
52
53/** The document split at its headings, each section with its heading path. */
54function sections(markdown: string, title: string): Section[] {
55 const lines = withoutFrontMatter(markdown).replace(/\r\n?/g, "\n").split("\n");
56 const out: Section[] = [{ heading: null, lines: [] }];
57 const path: { level: number; text: string }[] = [];
58 let fence: string | null = null;
59 let first = true;
60 const wantTitle = headingText(title).toLowerCase();
61 for (const line of lines) {
62 const f = FENCE.exec(line);
63 if (fence) {
64 if (f && f[1]![0] === fence[0] && f[1]!.length >= fence.length && line.trim() === f[1]) fence = null;
65 out[out.length - 1]!.lines.push(line);
66 continue;
67 }
68 if (f) {
69 fence = f[1]!;
70 out[out.length - 1]!.lines.push(line);
71 continue;
72 }
73 const h = HEADING.exec(line);
74 if (!h) {
75 if (line.trim()) first = false;
76 out[out.length - 1]!.lines.push(line);
77 continue;
78 }
79 const level = h[1]!.length;
80 const text = headingText(h[2]!);
81 // A file's leading `# Title` is its title, not a section of it.
82 if (first && level === 1 && text.toLowerCase() === wantTitle) {
83 first = false;
84 continue;
85 }
86 first = false;
87 while (path.length && path[path.length - 1]!.level >= level) path.pop();
88 path.push({ level, text });
89 out.push({ heading: path.map((p) => p.text).filter(Boolean).join(HEADING_SEPARATOR) || null, lines: [] });
90 }
91 return out;
92}
93
94/** Paragraphs: runs of lines between blank lines, with a fenced block kept whole. */
95function paragraphs(lines: string[]): string[] {
96 const out: string[] = [];
97 let current: string[] = [];
98 let fence: string | null = null;
99 const flush = () => {
100 const text = current.join("\n").trim();
101 if (text) out.push(text);
102 current = [];
103 };
104 for (const line of lines) {
105 const f = FENCE.exec(line);
106 if (fence) {
107 current.push(line);
108 if (f && f[1]![0] === fence[0] && f[1]!.length >= fence.length && line.trim() === f[1]) fence = null;
109 continue;
110 }
111 if (f) {
112 fence = f[1]!;
113 current.push(line);
114 continue;
115 }
116 if (!line.trim()) flush();
117 else current.push(line);
118 }
119 flush();
120 return out;
121}
122
123/** A paragraph longer than MAX_CHARS, cut at sentence ends or spaces. */
124function cut(text: string): string[] {
125 const out: string[] = [];
126 let rest = text;
127 while (rest.length > MAX_CHARS) {
128 const window = rest.slice(0, MAX_CHARS);
129 let at = Math.max(window.lastIndexOf(". "), window.lastIndexOf(".\n"), window.lastIndexOf("\n"));
130 if (at < MAX_CHARS / 2) at = window.lastIndexOf(" ");
131 if (at < MAX_CHARS / 2) at = MAX_CHARS - 1;
132 out.push(rest.slice(0, at + 1).trim());
133 rest = rest.slice(at + 1).trim();
134 }
135 if (rest) out.push(rest);
136 return out;
137}
138
139/** One section's passages: its paragraphs packed up to MAX_CHARS. */
140function pack(paras: string[]): string[] {
141 const out: string[] = [];
142 let current = "";
143 for (const para of paras.flatMap(cut)) {
144 if (!current) current = para;
145 else if (current.length + 2 + para.length <= MAX_CHARS) current += `\n\n${para}`;
146 else {
147 out.push(current);
148 current = para;
149 }
150 }
151 if (current) out.push(current);
152 return out;
153}
154
155/**
156 * A document's passages, in order. `title` is the page's or file's title:
157 * not part of any passage, but a file's leading `# Title` is dropped as it.
158 */
159export function chunkMarkdown(markdown: string, title = ""): Chunk[] {
160 const pieces: { heading: string | null; text: string }[] = [];
161 for (const section of sections(String(markdown ?? ""), title)) {
162 for (const text of pack(paragraphs(section.lines))) pieces.push({ heading: section.heading, text });
163 // A heading with nothing under it but subsections still names them: nothing to keep alone.
164 }
165 // Tiny passages join the next one (its heading written in), while that fits.
166 const merged: { heading: string | null; text: string }[] = [];
167 for (const piece of pieces) {
168 const last = merged[merged.length - 1];
169 if (last && last.text.length < MIN_CHARS) {
170 const joined = piece.heading && piece.heading !== last.heading ? `${last.text}\n\n**${piece.heading.split(HEADING_SEPARATOR).pop()}**\n\n${piece.text}` : `${last.text}\n\n${piece.text}`;
171 if (joined.length <= MAX_CHARS) {
172 last.text = joined;
173 if (!last.heading) last.heading = piece.heading;
174 continue;
175 }
176 }
177 merged.push({ ...piece });
178 }
179 // A tiny last passage joins the one before it, when that fits.
180 if (merged.length > 1) {
181 const last = merged[merged.length - 1]!;
182 const before = merged[merged.length - 2]!;
183 if (last.text.length < MIN_CHARS && before.text.length + last.text.length + 40 <= MAX_CHARS) {
184 before.text = last.heading && last.heading !== before.heading ? `${before.text}\n\n**${last.heading.split(HEADING_SEPARATOR).pop()}**\n\n${last.text}` : `${before.text}\n\n${last.text}`;
185 merged.pop();
186 }
187 }
188 return merged.slice(0, MAX_CHUNKS).map((piece, seq) => ({ seq, heading: piece.heading, text: piece.text }));
189}
190
191/** What is embedded for a passage: its document's title and heading in front, so a passage alone still says what it is about. */
192export function embedText(title: string, chunk: Pick<Chunk, "heading" | "text">): string {
193 const head = [title.trim(), chunk.heading ?? ""].filter(Boolean).join(HEADING_SEPARATOR);
194 return (head ? `${head}\n\n${chunk.text}` : chunk.text).slice(0, EMBED_CHARS);
195}
196
197/** A short, stable hash (hex) of a string: whether a passage changed since it was embedded. Not for security. */
198export function textHash(text: string): string {
199 let h1 = 0xdeadbeef;
200 let h2 = 0x41c6ce57;
201 for (let i = 0; i < text.length; i++) {
202 const c = text.charCodeAt(i);
203 h1 = Math.imul(h1 ^ c, 2654435761);
204 h2 = Math.imul(h2 ^ c, 1597334677);
205 }
206 h1 = Math.imul(h1 ^ (h1 >>> 16), 2246822507) ^ Math.imul(h2 ^ (h2 >>> 13), 3266489909);
207 h2 = Math.imul(h2 ^ (h2 >>> 16), 2246822507) ^ Math.imul(h1 ^ (h1 >>> 13), 3266489909);
208 return (h2 >>> 0).toString(16).padStart(8, "0") + (h1 >>> 0).toString(16).padStart(8, "0");
209}
210
211/** A repository docs file's id in the index (`rf_` and 32 hex): short enough for a vector id, the same each time. */
212export function repoFileId(spaceId: string, path: string): string {
213 return `rf_${textHash(`${spaceId}\n${path}`)}${textHash(`${path}\n${spaceId}`)}`;
214}
215
216/** A passage's id, in D1 and in the vector index: `<page or file id>:<seq>`. */
217export function chunkId(docId: string, seq: number): string {
218 return `${docId}:${seq}`;
219}