g1t/services/context/src/extract.ts
Pick any line to see why it is the way it is: the commit, the pull request and issue it came from, and what the agent was thinking.
| Agents get guardrails, run credentials, an audit log, a context hub, repository instructions and mentions; security upkeep; snake_case API | 1 | /** |
| 2 | * What a project's files say about it, read without running anything: the | |
| 3 | * languages and packages in its manifests, the APIs it exposes, its docs, | |
| 4 | * whether its workflows run tests, who owns it, and short facts and | |
| 5 | * conventions worth remembering. Pure: given a file's path and text, the | |
| 6 | * same facts every time, so a file whose blob has not changed is never read | |
| 7 | * again. | |
| 8 | */ | |
| 9 | ||
| 10 | import { parse as parseYaml } from "yaml"; | |
| 11 | ||
| 12 | export type Ecosystem = "npm" | "cargo" | "go" | "pypi"; | |
| 13 | ||
| 14 | export type PackageFact = { | |
| 15 | ecosystem: Ecosystem; | |
| 16 | name: string; | |
| 17 | version: string | null; | |
| 18 | /** Its dependencies' names, at most `MAX_DEPENDENCIES`. */ | |
| 19 | dependencies: string[]; | |
| 20 | /** For npm: the scripts people run. */ | |
| 21 | scripts?: Record<string, string>; | |
| 22 | /** For a monorepo's root: where its packages are. */ | |
| 23 | members?: string[]; | |
| 24 | }; | |
| 25 | ||
| 26 | export type ApiFact = { | |
| 27 | /** `openapi` for a described HTTP API, `worker` for a Worker's routes. */ | |
| 28 | kind: "openapi" | "worker"; | |
| 29 | name: string; | |
| 30 | summary: string | null; | |
| 31 | /** `GET /users`, or a route pattern such as `api.example.com/*`. */ | |
| 32 | routes: string[]; | |
| 33 | }; | |
| 34 | ||
| 35 | export type DocFact = { | |
| 36 | path: string; | |
| 37 | title: string; | |
| 38 | summary: string | null; | |
| 39 | /** The doc in pieces of at most `CHUNK_CHARS`, each starting with its heading. */ | |
| 40 | chunks: string[]; | |
| 41 | /** README, AGENTS.md (or CLAUDE.md), CONTRIBUTING, or another doc. */ | |
| 42 | role: "readme" | "agents" | "contributing" | "doc"; | |
| 43 | }; | |
| 44 | ||
| 45 | /** Something worth remembering, as a memory candidate. */ | |
| 46 | export type Hint = { | |
| 47 | kind: "fact" | "convention" | "decision" | "gotcha"; | |
| 48 | text: string; | |
| 49 | /** 0 to 1. A doc's hint at or above 0.85 is kept without review. */ | |
| 50 | confidence: number; | |
| 51 | evidence: string; | |
| 52 | }; | |
| 53 | ||
| 54 | export type FileFacts = { | |
| 55 | languages: string[]; | |
| 56 | packages: PackageFact[]; | |
| 57 | apis: ApiFact[]; | |
| 58 | doc: DocFact | null; | |
| 59 | /** A workflow that runs tests. */ | |
| 60 | tests: boolean; | |
| 61 | /** Usernames this file names as owners. */ | |
| 62 | owners: string[]; | |
| 63 | hints: Hint[]; | |
| 64 | }; | |
| 65 | ||
| 66 | /** What extraction knows besides the file itself. */ | |
| 67 | export type ExtractContext = { | |
| 68 | /** The project's name, for writing hints. */ | |
| 69 | project: string; | |
| 70 | /** The names of the files beside the manifest: lockfiles, tsconfig.json. */ | |
| 71 | siblings: string[]; | |
| 72 | }; | |
| 73 | ||
| 74 | export const MAX_DEPENDENCIES = 60; | |
| 75 | export const CHUNK_CHARS = 1500; | |
| 76 | export const MAX_CHUNKS = 12; | |
| 77 | const MAX_ROUTES = 40; | |
| 78 | const MAX_HINTS_PER_FILE = 12; | |
| 79 | const MAX_HINT_CHARS = 300; | |
| 80 | ||
| 81 | const EMPTY: FileFacts = { languages: [], packages: [], apis: [], doc: null, tests: false, owners: [], hints: [] }; | |
| 82 | ||
| 83 | const DOC_NAME = /^(readme|agents|claude|contributing)(\.(md|markdown|txt))?$/i; | |
| 84 | const OPENAPI_NAME = /^(openapi|swagger)\.(json|ya?ml)$/i; | |
| 85 | const MANIFEST_NAMES = new Set([ | |
| 86 | "package.json", | |
| 87 | "Cargo.toml", | |
| 88 | "go.mod", | |
| 89 | "pyproject.toml", | |
| 90 | "requirements.txt", | |
| 91 | "wrangler.jsonc", | |
| 92 | "wrangler.json", | |
| 93 | "wrangler.toml", | |
| 94 | "CODEOWNERS", | |
| 95 | ]); | |
| 96 | ||
| 97 | /** The base name of a path. */ | |
| 98 | function base(path: string): string { | |
| 99 | return path.slice(path.lastIndexOf("/") + 1); | |
| 100 | } | |
| 101 | ||
| 102 | /** | |
| 103 | * Whether a file at `path` (relative to the project's root) is one this | |
| 104 | * reads: a manifest, an API description, a doc, a workflow or ownership. | |
| 105 | */ | |
| 106 | export function interesting(path: string): boolean { | |
| 107 | const name = base(path); | |
| 108 | const dir = path.slice(0, Math.max(0, path.lastIndexOf("/"))); | |
| 109 | if (dir === "" && (MANIFEST_NAMES.has(name) || DOC_NAME.test(name) || OPENAPI_NAME.test(name))) return true; | |
| 110 | if ((dir === "docs" || dir === "doc" || dir === "runbooks" || dir === "docs/runbooks") && /\.(md|markdown)$/i.test(name)) return true; | |
| 111 | if ((dir === ".g1t/workflows" || dir === ".github/workflows") && /\.ya?ml$/i.test(name)) return true; | |
| 112 | if (path === ".g1t/project.yml" || path === ".github/CODEOWNERS" || path === "docs/CODEOWNERS") return true; | |
| 113 | if (dir === "" && OPENAPI_NAME.test(name)) return true; | |
| 114 | return false; | |
| 115 | } | |
| 116 | ||
| 117 | /** One line, at most `max` characters. */ | |
| 118 | export function clip(text: string, max = MAX_HINT_CHARS): string { | |
| 119 | const line = text.replace(/\s+/g, " ").trim(); | |
| 120 | return line.length <= max ? line : `${line.slice(0, max - 1).replace(/\s+\S*$/, "")}…`; | |
| 121 | } | |
| 122 | ||
| 123 | /** JSON with comments and trailing commas, as wrangler.jsonc and tsconfig.json allow. */ | |
| 124 | export function parseJsonc(text: string): unknown { | |
| 125 | let out = ""; | |
| 126 | let inString = false; | |
| 127 | for (let i = 0; i < text.length; i++) { | |
| 128 | const c = text[i]; | |
| 129 | if (inString) { | |
| 130 | out += c; | |
| 131 | if (c === "\\") { | |
| 132 | out += text[++i] ?? ""; | |
| 133 | } else if (c === '"') { | |
| 134 | inString = false; | |
| 135 | } | |
| 136 | continue; | |
| 137 | } | |
| 138 | if (c === '"') { | |
| 139 | inString = true; | |
| 140 | out += c; | |
| 141 | } else if (c === "/" && text[i + 1] === "/") { | |
| 142 | while (i < text.length && text[i] !== "\n") i++; | |
| 143 | out += "\n"; | |
| 144 | } else if (c === "/" && text[i + 1] === "*") { | |
| 145 | i += 2; | |
| 146 | while (i < text.length && !(text[i] === "*" && text[i + 1] === "/")) i++; | |
| 147 | i++; | |
| 148 | } else { | |
| 149 | out += c; | |
| 150 | } | |
| 151 | } | |
| 152 | return JSON.parse(out.replace(/,(\s*[}\]])/g, "$1")); | |
| 153 | } | |
| 154 | ||
| 155 | type Toml = Record<string, Record<string, unknown>>; | |
| 156 | ||
| 157 | /** | |
| 158 | * The little of TOML that manifests use: `[sections]`, `key = "string"`, | |
| 159 | * numbers, booleans, one-line and multi-line string arrays, and inline | |
| 160 | * tables kept as text. Keys outside a section are under "". | |
| 161 | */ | |
| 162 | export function parseToml(text: string): Toml { | |
| 163 | const out: Toml = { "": {} }; | |
| 164 | let section = ""; | |
| 165 | const lines = text.split(/\r?\n/); | |
| 166 | for (let i = 0; i < lines.length; i++) { | |
| 167 | let line = lines[i].replace(/(^|\s)#.*$/, "").trim(); | |
| 168 | if (!line) continue; | |
| 169 | const header = /^\[\[?([^\]]+)\]\]?$/.exec(line); | |
| 170 | if (header) { | |
| 171 | section = header[1].trim().replace(/"/g, ""); | |
| 172 | out[section] ??= {}; | |
| 173 | continue; | |
| 174 | } | |
| 175 | const eq = line.indexOf("="); | |
| 176 | if (eq < 0) continue; | |
| 177 | const key = line.slice(0, eq).trim().replace(/"/g, ""); | |
| 178 | let value = line.slice(eq + 1).trim(); | |
| 179 | // A multi-line array: gather until its closing bracket. | |
| 180 | if (value.startsWith("[") && !value.includes("]")) { | |
| 181 | while (i + 1 < lines.length && !value.includes("]")) value += " " + lines[++i].replace(/#.*$/, "").trim(); | |
| 182 | } | |
| 183 | out[section] ??= {}; | |
| 184 | out[section][key] = tomlValue(value); | |
| 185 | } | |
| 186 | return out; | |
| 187 | } | |
| 188 | ||
| 189 | function tomlValue(value: string): unknown { | |
| 190 | if (value.startsWith('"') || value.startsWith("'")) return value.slice(1, value.lastIndexOf(value[0])); | |
| 191 | if (value === "true" || value === "false") return value === "true"; | |
| 192 | if (value.startsWith("[")) { | |
| 193 | return [...value.matchAll(/"([^"]*)"|'([^']*)'/g)].map((m) => m[1] ?? m[2]); | |
| 194 | } | |
| 195 | if (/^-?\d+(\.\d+)?$/.test(value)) return Number(value); | |
| 196 | return value; | |
| 197 | } | |
| 198 | ||
| 199 | function str(value: unknown): string | null { | |
| 200 | return typeof value === "string" && value.trim() ? value.trim() : null; | |
| 201 | } | |
| 202 | ||
| 203 | function keys(value: unknown): string[] { | |
| 204 | return value && typeof value === "object" && !Array.isArray(value) ? Object.keys(value) : []; | |
| 205 | } | |
| 206 | ||
| 207 | /** The package manager a JavaScript project's lockfile says it uses. */ | |
| 208 | export function packageManager(siblings: string[]): "pnpm" | "yarn" | "bun" | "npm" | null { | |
| 209 | const has = (name: string) => siblings.includes(name); | |
| 210 | if (has("pnpm-lock.yaml")) return "pnpm"; | |
| 211 | if (has("yarn.lock")) return "yarn"; | |
| 212 | if (has("bun.lockb") || has("bun.lock")) return "bun"; | |
| 213 | if (has("package-lock.json")) return "npm"; | |
| 214 | return null; | |
| 215 | } | |
| 216 | ||
| 217 | function npm(path: string, text: string, ctx: ExtractContext): FileFacts { | |
| 218 | const json = JSON.parse(text) as Record<string, unknown>; | |
| 219 | const scripts = (json.scripts && typeof json.scripts === "object" ? json.scripts : {}) as Record<string, string>; | |
| 220 | const deps = [...keys(json.dependencies), ...keys(json.devDependencies)]; | |
| 221 | const typescript = deps.includes("typescript") || ctx.siblings.includes("tsconfig.json"); | |
| 222 | const workspaces = Array.isArray(json.workspaces) | |
| 223 | ? (json.workspaces as unknown[]).filter((w): w is string => typeof w === "string") | |
| 224 | : Array.isArray((json.workspaces as { packages?: unknown })?.packages) | |
| 225 | ? ((json.workspaces as { packages: unknown[] }).packages.filter((w): w is string => typeof w === "string")) | |
| 226 | : []; | |
| 227 | const pm = packageManager(ctx.siblings); | |
| 228 | const run = (script: string) => (script === "test" ? `${pm ?? "npm"} test` : `${pm ?? "npm"} run ${script}`); | |
| 229 | const hints: Hint[] = []; | |
| 230 | if (pm) { | |
| 231 | const others = ["npm", "pnpm", "yarn", "bun"].filter((other) => other !== pm).join(" or "); | |
| 232 | hints.push({ | |
| 233 | kind: "convention", | |
| 234 | text: `${ctx.project} uses ${pm} (its lockfile is committed): install with \`${pm} install\`, not ${others}.`, | |
| 235 | confidence: 0.9, | |
| 236 | evidence: `${path} beside its ${pm} lockfile`, | |
| 237 | }); | |
| 238 | } | |
| 239 | for (const script of ["test", "typecheck", "lint", "build"]) { | |
| 240 | const command = str(scripts[script]); | |
| 241 | if (!command) continue; | |
| 242 | hints.push({ | |
| 243 | kind: "fact", | |
| 244 | text: clip(`In ${ctx.project}, \`${run(script)}\` runs ${script === "test" ? "the tests" : `the ${script}`}: \`${command}\`.`), | |
| 245 | confidence: 0.9, | |
| 246 | evidence: `${path} scripts.${script}`, | |
| 247 | }); | |
| 248 | } | |
| 249 | if (workspaces.length > 0) { | |
| 250 | hints.push({ | |
| 251 | kind: "fact", | |
| 252 | text: clip(`${ctx.project} is a monorepo; its packages are in ${workspaces.join(", ")}.`), | |
| 253 | confidence: 0.9, | |
| 254 | evidence: `${path} workspaces`, | |
| 255 | }); | |
| 256 | } | |
| 257 | const name = str(json.name); | |
| 258 | return { | |
| 259 | ...EMPTY, | |
| 260 | languages: typescript ? ["TypeScript"] : ["JavaScript"], | |
| 261 | packages: name | |
| 262 | ? [ | |
| 263 | { | |
| 264 | ecosystem: "npm", | |
| 265 | name, | |
| 266 | version: str(json.version), | |
| 267 | dependencies: deps.slice(0, MAX_DEPENDENCIES), | |
| 268 | scripts: Object.fromEntries(Object.entries(scripts).filter(([, v]) => typeof v === "string").slice(0, 20)), | |
| 269 | ...(workspaces.length ? { members: workspaces } : {}), | |
| 270 | }, | |
| 271 | ] | |
| 272 | : [], | |
| 273 | hints, | |
| 274 | }; | |
| 275 | } | |
| 276 | ||
| 277 | function cargo(path: string, text: string, ctx: ExtractContext): FileFacts { | |
| 278 | const toml = parseToml(text); | |
| 279 | const name = str(toml.package?.name); | |
| 280 | const members = Array.isArray(toml.workspace?.members) ? (toml.workspace.members as string[]) : []; | |
| 281 | const deps = [...keys(toml.dependencies), ...keys(toml["dev-dependencies"]), ...keys(toml["workspace.dependencies"])]; | |
| 282 | const hints: Hint[] = [ | |
| 283 | { | |
| 284 | kind: "fact", | |
| 285 | text: clip( | |
| 286 | members.length | |
| 287 | ? `${ctx.project} is a Cargo workspace (${members.join(", ")}); \`cargo test\` runs its tests.` | |
| 288 | : `${ctx.project} is written in Rust; \`cargo test\` runs its tests.`, | |
| 289 | ), | |
| 290 | confidence: 0.9, | |
| 291 | evidence: path, | |
| 292 | }, | |
| 293 | ]; | |
| 294 | return { | |
| 295 | ...EMPTY, | |
| 296 | languages: ["Rust"], | |
| 297 | packages: name | |
| 298 | ? [{ ecosystem: "cargo", name, version: str(toml.package?.version), dependencies: [...new Set(deps)].slice(0, MAX_DEPENDENCIES) }] | |
| 299 | : members.length | |
| 300 | ? [{ ecosystem: "cargo", name: `${ctx.project} (workspace)`, version: null, dependencies: [...new Set(deps)].slice(0, MAX_DEPENDENCIES), members }] | |
| 301 | : [], | |
| 302 | hints, | |
| 303 | }; | |
| 304 | } | |
| 305 | ||
| 306 | function goMod(path: string, text: string, ctx: ExtractContext): FileFacts { | |
| 307 | const module = /^module\s+(\S+)/m.exec(text)?.[1] ?? null; | |
| 308 | const version = /^go\s+(\S+)/m.exec(text)?.[1] ?? null; | |
| 309 | const requires = [...text.matchAll(/^\s*(?:require\s+)?([\w.-]+\.[\w.-]+\/\S+)\s+v[\w.+-]+/gm)].map((m) => m[1]); | |
| 310 | return { | |
| 311 | ...EMPTY, | |
| 312 | languages: ["Go"], | |
| 313 | packages: module ? [{ ecosystem: "go", name: module, version, dependencies: requires.slice(0, MAX_DEPENDENCIES) }] : [], | |
| 314 | hints: [ | |
| 315 | { | |
| 316 | kind: "fact", | |
| 317 | text: clip(`${ctx.project} is a Go module${module ? ` (${module}${version ? `, go ${version}` : ""})` : ""}; \`go test ./...\` runs its tests.`), | |
| 318 | confidence: 0.9, | |
| 319 | evidence: path, | |
| 320 | }, | |
| 321 | ], | |
| 322 | }; | |
| 323 | } | |
| 324 | ||
| 325 | function python(path: string, text: string, ctx: ExtractContext): FileFacts { | |
| 326 | let name: string | null = null; | |
| 327 | let version: string | null = null; | |
| 328 | let deps: string[] = []; | |
| 329 | if (base(path) === "pyproject.toml") { | |
| 330 | const toml = parseToml(text); | |
| 331 | name = str(toml.project?.name) ?? str(toml["tool.poetry"]?.name); | |
| 332 | version = str(toml.project?.version) ?? str(toml["tool.poetry"]?.version); | |
| 333 | deps = [ | |
| 334 | ...(Array.isArray(toml.project?.dependencies) ? (toml.project.dependencies as string[]) : []), | |
| 335 | ...keys(toml["tool.poetry.dependencies"]), | |
| 336 | ]; | |
| 337 | } else { | |
| 338 | deps = text.split(/\r?\n/).map((line) => line.replace(/#.*$/, "").trim()).filter((line) => line && !line.startsWith("-")); | |
| 339 | } | |
| 340 | deps = deps.map((dep) => dep.split(/[<>=!~;\[\s]/)[0]).filter(Boolean); | |
| 341 | const hints: Hint[] = deps.includes("pytest") | |
| 342 | ? [{ kind: "fact", text: `In ${ctx.project}, \`pytest\` runs the tests.`, confidence: 0.85, evidence: path }] | |
| 343 | : []; | |
| 344 | return { | |
| 345 | ...EMPTY, | |
| 346 | languages: ["Python"], | |
| 347 | packages: name ? [{ ecosystem: "pypi", name, version, dependencies: deps.slice(0, MAX_DEPENDENCIES) }] : [], | |
| 348 | hints, | |
| 349 | }; | |
| 350 | } | |
| 351 | ||
| 352 | function wrangler(path: string, text: string): FileFacts { | |
| 353 | let config: Record<string, unknown> = {}; | |
| 354 | if (path.endsWith(".toml")) { | |
| 355 | const toml = parseToml(text); | |
| 356 | config = { ...toml[""], routes: Array.isArray(toml[""].routes) ? toml[""].routes : [] }; | |
| 357 | if (toml.routes) config.routes = [toml.routes.pattern]; | |
| 358 | } else { | |
| 359 | config = parseJsonc(text) as Record<string, unknown>; | |
| 360 | } | |
| 361 | const routes: string[] = []; | |
| 362 | const add = (value: unknown) => { | |
| 363 | const pattern = typeof value === "string" ? value : str((value as { pattern?: unknown })?.pattern); | |
| 364 | if (pattern) routes.push(pattern); | |
| 365 | }; | |
| 366 | if (Array.isArray(config.routes)) config.routes.forEach(add); | |
| 367 | add(config.route); | |
| 368 | const name = str(config.name); | |
| 369 | if (!name) return EMPTY; | |
| 370 | return { | |
| 371 | ...EMPTY, | |
| 372 | apis: [ | |
| 373 | { | |
| 374 | kind: "worker", | |
| 375 | name, | |
| 376 | summary: routes.length ? `Served at ${routes.slice(0, 5).join(", ")}` : "A Worker reached through service bindings", | |
| 377 | routes: routes.slice(0, MAX_ROUTES), | |
| 378 | }, | |
| 379 | ], | |
| 380 | }; | |
| 381 | } | |
| 382 | ||
| 383 | const METHODS = ["get", "post", "put", "patch", "delete", "head", "options"]; | |
| 384 | ||
| 385 | function openapi(path: string, text: string): FileFacts { | |
| 386 | const spec = (/\.json$/i.test(path) ? JSON.parse(text) : parseYaml(text)) as { | |
| 387 | info?: { title?: unknown; description?: unknown; version?: unknown }; | |
| 388 | paths?: Record<string, Record<string, unknown>>; | |
| 389 | }; | |
| 390 | const routes: string[] = []; | |
| 391 | for (const [route, operations] of Object.entries(spec?.paths ?? {})) { | |
| 392 | for (const method of Object.keys(operations ?? {})) { | |
| 393 | if (METHODS.includes(method)) routes.push(`${method.toUpperCase()} ${route}`); | |
| 394 | } | |
| 395 | } | |
| 396 | const title = str(spec?.info?.title) ?? base(path); | |
| 397 | return { | |
| 398 | ...EMPTY, | |
| 399 | apis: [ | |
| 400 | { | |
| 401 | kind: "openapi", | |
| 402 | name: title, | |
| 403 | summary: clip(str(spec?.info?.description) ?? `${routes.length} operations${str(spec?.info?.version) ? `, version ${spec.info!.version}` : ""}`), | |
| 404 | routes: routes.slice(0, MAX_ROUTES), | |
| 405 | }, | |
| 406 | ], | |
| 407 | }; | |
| 408 | } | |
| 409 | ||
| 410 | /** A doc split at its headings into pieces of at most `CHUNK_CHARS`. */ | |
| 411 | export function chunk(text: string, title: string): string[] { | |
| 412 | const sections: string[] = []; | |
| 413 | let current = ""; | |
| 414 | for (const line of text.split(/\r?\n/)) { | |
| 415 | if (/^#{1,3}\s/.test(line) && current.trim()) { | |
| 416 | sections.push(current.trim()); | |
| 417 | current = ""; | |
| 418 | } | |
| 419 | current += line + "\n"; | |
| 420 | } | |
| 421 | if (current.trim()) sections.push(current.trim()); | |
| 422 | const pieces: string[] = []; | |
| 423 | for (const section of sections) { | |
| 424 | const heading = /^#{1,3}\s+(.*)$/m.exec(section)?.[1]?.trim() ?? title; | |
| 425 | for (let at = 0; at < section.length && pieces.length < MAX_CHUNKS; at += CHUNK_CHARS) { | |
| 426 | const body = section.slice(at, at + CHUNK_CHARS); | |
| 427 | pieces.push(at === 0 ? body : `${heading} (continued)\n${body}`); | |
| 428 | } | |
| 429 | } | |
| 430 | return pieces.slice(0, MAX_CHUNKS); | |
| 431 | } | |
| 432 | ||
| 433 | const COMMAND = /^\s*(?:\$\s*)?((?:npm|pnpm|yarn|bun|npx|cargo|go|make|pytest|python3?|poetry|uv|docker|wrangler|just|mix|bundle|rake|gradle|\.\/gradlew|mvn|dotnet)\b[^\n]{0,160})$/; | |
| 434 | const SETUP_HEADING = /\b(develop|development|getting started|setup|set up|install|build|test|testing|contribut|running|run locally|local)\b/i; | |
| 435 | const CONVENTION_HEADING = /\b(convention|guideline|rules|style|standards|gotcha|pitfall|caveat|known issue|do not|don't)\b/i; | |
| 436 | const GOTCHA = /\b(never|don't|do not|must not|careful|beware|gotcha|warning|avoid)\b/i; | |
| 437 | ||
| 438 | /** What a doc says worth remembering: commands in its setup sections, and its conventions. */ | |
| 439 | function docHints(path: string, text: string, role: DocFact["role"], ctx: ExtractContext): Hint[] { | |
| 440 | const hints: Hint[] = []; | |
| 441 | const agents = role === "agents"; | |
| 442 | let heading = ""; | |
| 443 | let fenced = false; | |
| 444 | for (const raw of text.split(/\r?\n/)) { | |
| 445 | const line = raw.trimEnd(); | |
| 446 | if (/^\s*(```|~~~)/.test(line)) { | |
| 447 | fenced = !fenced; | |
| 448 | continue; | |
| 449 | } | |
| 450 | const h = /^#{1,6}\s+(.*)$/.exec(line); | |
| 451 | if (h && !fenced) { | |
| 452 | heading = h[1].trim(); | |
| 453 | continue; | |
| 454 | } | |
| 455 | if (fenced) { | |
| 456 | const command = COMMAND.exec(line)?.[1]; | |
| 457 | if (command && (agents || SETUP_HEADING.test(heading))) { | |
| 458 | const what = heading ? heading.replace(/[:.]$/, "").toLowerCase() : "work on it"; | |
| 459 | hints.push({ | |
| 460 | kind: "fact", | |
| 461 | text: clip(`In ${ctx.project}, for ${what}: \`${command.trim()}\`.`), | |
| 462 | confidence: agents ? 0.9 : 0.7, | |
| 463 | evidence: `${path}${heading ? ` (${heading})` : ""}`, | |
| 464 | }); | |
| 465 | } | |
| 466 | continue; | |
| 467 | } | |
| 468 | const bullet = /^\s*[-*+]\s+(.*)$/.exec(line)?.[1]?.trim(); | |
| 469 | if (!bullet || bullet.length < 20 || bullet.length > MAX_HINT_CHARS) continue; | |
| 470 | // A bullet that is only a link is a table of contents. | |
| 471 | if (/^\[[^\]]*\]\([^)]*\)\.?$/.test(bullet)) continue; | |
| 472 | if (agents || CONVENTION_HEADING.test(heading)) { | |
| 473 | hints.push({ | |
| 474 | kind: GOTCHA.test(bullet) ? "gotcha" : "convention", | |
| 475 | text: clip(bullet.replace(/\*\*/g, "")), | |
| 476 | confidence: agents ? 0.9 : 0.7, | |
| 477 | evidence: `${path}${heading ? ` (${heading})` : ""}`, | |
| 478 | }); | |
| 479 | } | |
| 480 | } | |
| 481 | // The same command under two headings is one hint. | |
| 482 | const seen = new Set<string>(); | |
| 483 | return hints.filter((hint) => !seen.has(hint.text) && seen.add(hint.text)).slice(0, MAX_HINTS_PER_FILE); | |
| 484 | } | |
| 485 | ||
| 486 | function doc(path: string, text: string, ctx: ExtractContext): FileFacts { | |
| 487 | const name = base(path); | |
| 488 | const role: DocFact["role"] = /^readme/i.test(name) | |
| 489 | ? "readme" | |
| 490 | : /^(agents|claude)/i.test(name) | |
| 491 | ? "agents" | |
| 492 | : /^contributing/i.test(name) | |
| 493 | ? "contributing" | |
| 494 | : "doc"; | |
| 495 | const title = /^#\s+(.*)$/m.exec(text)?.[1]?.trim() ?? name.replace(/\.(md|markdown|txt)$/i, ""); | |
| 496 | const paragraph = text | |
| 497 | .split(/\r?\n\s*\r?\n/) | |
| 498 | .map((block) => block.trim()) | |
| 499 | .find((block) => block && !block.startsWith("#") && !block.startsWith("```") && !block.startsWith("<") && !block.startsWith("![")); | |
| 500 | return { | |
| 501 | ...EMPTY, | |
| 502 | doc: { path, title: clip(title, 120), summary: paragraph ? clip(paragraph) : null, chunks: chunk(text, title), role }, | |
| 503 | hints: docHints(path, text, role, ctx), | |
| 504 | }; | |
| 505 | } | |
| 506 | ||
| 507 | function owners(path: string, text: string): FileFacts { | |
| 508 | const names = new Set<string>(); | |
| 509 | if (path.endsWith("project.yml")) { | |
| 510 | const parsed = parseYaml(text) as { owners?: unknown } | null; | |
| 511 | const list = Array.isArray(parsed?.owners) ? parsed.owners : typeof parsed?.owners === "string" ? [parsed.owners] : []; | |
| 512 | for (const owner of list) if (typeof owner === "string") names.add(owner.replace(/^@/, "").trim()); | |
| 513 | } else { | |
| 514 | for (const line of text.split(/\r?\n/)) { | |
| 515 | if (line.trim().startsWith("#")) continue; | |
| 516 | for (const match of line.matchAll(/@([A-Za-z0-9][A-Za-z0-9_.-]*)(?=\s|$)/g)) names.add(match[1]); | |
| 517 | } | |
| 518 | } | |
| 519 | return { ...EMPTY, owners: [...names].filter(Boolean).slice(0, 20) }; | |
| 520 | } | |
| 521 | ||
| 522 | const TESTS = /\b(test|tests|pytest|vitest|jest|mocha|cargo test|go test|npm test|playwright|cypress)\b/i; | |
| 523 | ||
| 524 | /** The facts in one file. A file that cannot be parsed says nothing. */ | |
| 525 | export function extract(path: string, text: string, ctx: ExtractContext): FileFacts { | |
| 526 | const name = base(path); | |
| 527 | try { | |
| 528 | if (path.startsWith(".g1t/workflows/") || path.startsWith(".github/workflows/")) { | |
| 529 | return { ...EMPTY, tests: TESTS.test(text) }; | |
| 530 | } | |
| 531 | if (path === ".g1t/project.yml" || name === "CODEOWNERS") return owners(path, text); | |
| 532 | if (name === "package.json") return npm(path, text, ctx); | |
| 533 | if (name === "Cargo.toml") return cargo(path, text, ctx); | |
| 534 | if (name === "go.mod") return goMod(path, text, ctx); | |
| 535 | if (name === "pyproject.toml" || name === "requirements.txt") return python(path, text, ctx); | |
| 536 | if (/^wrangler\.(jsonc|json|toml)$/.test(name)) return wrangler(path, text); | |
| 537 | if (OPENAPI_NAME.test(name)) return openapi(path, text); | |
| 538 | if (/\.(md|markdown|txt)$/i.test(name) || DOC_NAME.test(name)) return doc(path, text, ctx); | |
| 539 | } catch { | |
| 540 | return EMPTY; | |
| 541 | } | |
| 542 | return EMPTY; | |
| 543 | } |