g1t/services/context/src/extract.ts
| 1 | /** |
| 2 | * What a project's files say about it, read without running anything: the |
| 3 | * languages and packages in its manifests, the APIs it exposes, its docs, |
| 4 | * whether its workflows run tests, who owns it, and short facts and |
| 5 | * conventions worth remembering. Pure: given a file's path and text, the |
| 6 | * same facts every time, so a file whose blob has not changed is never read |
| 7 | * again. |
| 8 | */ |
| 9 | |
| 10 | import { parse as parseYaml } from "yaml"; |
| 11 | |
| 12 | export type Ecosystem = "npm" | "cargo" | "go" | "pypi"; |
| 13 | |
| 14 | export type PackageFact = { |
| 15 | ecosystem: Ecosystem; |
| 16 | name: string; |
| 17 | version: string | null; |
| 18 | /** Its dependencies' names, at most `MAX_DEPENDENCIES`. */ |
| 19 | dependencies: string[]; |
| 20 | /** For npm: the scripts people run. */ |
| 21 | scripts?: Record<string, string>; |
| 22 | /** For a monorepo's root: where its packages are. */ |
| 23 | members?: string[]; |
| 24 | }; |
| 25 | |
| 26 | export type ApiFact = { |
| 27 | /** `openapi` for a described HTTP API, `worker` for a Worker's routes. */ |
| 28 | kind: "openapi" | "worker"; |
| 29 | name: string; |
| 30 | summary: string | null; |
| 31 | /** `GET /users`, or a route pattern such as `api.example.com/*`. */ |
| 32 | routes: string[]; |
| 33 | }; |
| 34 | |
| 35 | export type DocFact = { |
| 36 | path: string; |
| 37 | title: string; |
| 38 | summary: string | null; |
| 39 | /** The doc in pieces of at most `CHUNK_CHARS`, each starting with its heading. */ |
| 40 | chunks: string[]; |
| 41 | /** README, AGENTS.md (or CLAUDE.md), CONTRIBUTING, or another doc. */ |
| 42 | role: "readme" | "agents" | "contributing" | "doc"; |
| 43 | }; |
| 44 | |
| 45 | /** Something worth remembering, as a memory candidate. */ |
| 46 | export type Hint = { |
| 47 | kind: "fact" | "convention" | "decision" | "gotcha"; |
| 48 | text: string; |
| 49 | /** 0 to 1. A doc's hint at or above 0.85 is kept without review. */ |
| 50 | confidence: number; |
| 51 | evidence: string; |
| 52 | }; |
| 53 | |
| 54 | export type FileFacts = { |
| 55 | languages: string[]; |
| 56 | packages: PackageFact[]; |
| 57 | apis: ApiFact[]; |
| 58 | doc: DocFact | null; |
| 59 | /** A workflow that runs tests. */ |
| 60 | tests: boolean; |
| 61 | /** Usernames this file names as owners. */ |
| 62 | owners: string[]; |
| 63 | hints: Hint[]; |
| 64 | }; |
| 65 | |
| 66 | /** What extraction knows besides the file itself. */ |
| 67 | export type ExtractContext = { |
| 68 | /** The project's name, for writing hints. */ |
| 69 | project: string; |
| 70 | /** The names of the files beside the manifest: lockfiles, tsconfig.json. */ |
| 71 | siblings: string[]; |
| 72 | }; |
| 73 | |
| 74 | export const MAX_DEPENDENCIES = 60; |
| 75 | export const CHUNK_CHARS = 1500; |
| 76 | export const MAX_CHUNKS = 12; |
| 77 | const MAX_ROUTES = 40; |
| 78 | const MAX_HINTS_PER_FILE = 12; |
| 79 | const MAX_HINT_CHARS = 300; |
| 80 | |
| 81 | const EMPTY: FileFacts = { languages: [], packages: [], apis: [], doc: null, tests: false, owners: [], hints: [] }; |
| 82 | |
| 83 | const DOC_NAME = /^(readme|agents|claude|contributing)(\.(md|markdown|txt))?$/i; |
| 84 | const OPENAPI_NAME = /^(openapi|swagger)\.(json|ya?ml)$/i; |
| 85 | const MANIFEST_NAMES = new Set([ |
| 86 | "package.json", |
| 87 | "Cargo.toml", |
| 88 | "go.mod", |
| 89 | "pyproject.toml", |
| 90 | "requirements.txt", |
| 91 | "wrangler.jsonc", |
| 92 | "wrangler.json", |
| 93 | "wrangler.toml", |
| 94 | "CODEOWNERS", |
| 95 | ]); |
| 96 | |
| 97 | /** The base name of a path. */ |
| 98 | function base(path: string): string { |
| 99 | return path.slice(path.lastIndexOf("/") + 1); |
| 100 | } |
| 101 | |
| 102 | /** |
| 103 | * Whether a file at `path` (relative to the project's root) is one this |
| 104 | * reads: a manifest, an API description, a doc, a workflow or ownership. |
| 105 | */ |
| 106 | export function interesting(path: string): boolean { |
| 107 | const name = base(path); |
| 108 | const dir = path.slice(0, Math.max(0, path.lastIndexOf("/"))); |
| 109 | if (dir === "" && (MANIFEST_NAMES.has(name) || DOC_NAME.test(name) || OPENAPI_NAME.test(name))) return true; |
| 110 | if ((dir === "docs" || dir === "doc" || dir === "runbooks" || dir === "docs/runbooks") && /\.(md|markdown)$/i.test(name)) return true; |
| 111 | if ((dir === ".g1t/workflows" || dir === ".github/workflows") && /\.ya?ml$/i.test(name)) return true; |
| 112 | if (path === ".g1t/project.yml" || path === ".github/CODEOWNERS" || path === "docs/CODEOWNERS") return true; |
| 113 | if (dir === "" && OPENAPI_NAME.test(name)) return true; |
| 114 | return false; |
| 115 | } |
| 116 | |
| 117 | /** One line, at most `max` characters. */ |
| 118 | export function clip(text: string, max = MAX_HINT_CHARS): string { |
| 119 | const line = text.replace(/\s+/g, " ").trim(); |
| 120 | return line.length <= max ? line : `${line.slice(0, max - 1).replace(/\s+\S*$/, "")}…`; |
| 121 | } |
| 122 | |
| 123 | /** JSON with comments and trailing commas, as wrangler.jsonc and tsconfig.json allow. */ |
| 124 | export function parseJsonc(text: string): unknown { |
| 125 | let out = ""; |
| 126 | let inString = false; |
| 127 | for (let i = 0; i < text.length; i++) { |
| 128 | const c = text[i]; |
| 129 | if (inString) { |
| 130 | out += c; |
| 131 | if (c === "\\") { |
| 132 | out += text[++i] ?? ""; |
| 133 | } else if (c === '"') { |
| 134 | inString = false; |
| 135 | } |
| 136 | continue; |
| 137 | } |
| 138 | if (c === '"') { |
| 139 | inString = true; |
| 140 | out += c; |
| 141 | } else if (c === "/" && text[i + 1] === "/") { |
| 142 | while (i < text.length && text[i] !== "\n") i++; |
| 143 | out += "\n"; |
| 144 | } else if (c === "/" && text[i + 1] === "*") { |
| 145 | i += 2; |
| 146 | while (i < text.length && !(text[i] === "*" && text[i + 1] === "/")) i++; |
| 147 | i++; |
| 148 | } else { |
| 149 | out += c; |
| 150 | } |
| 151 | } |
| 152 | return JSON.parse(out.replace(/,(\s*[}\]])/g, "$1")); |
| 153 | } |
| 154 | |
| 155 | type Toml = Record<string, Record<string, unknown>>; |
| 156 | |
| 157 | /** |
| 158 | * The little of TOML that manifests use: `[sections]`, `key = "string"`, |
| 159 | * numbers, booleans, one-line and multi-line string arrays, and inline |
| 160 | * tables kept as text. Keys outside a section are under "". |
| 161 | */ |
| 162 | export function parseToml(text: string): Toml { |
| 163 | const out: Toml = { "": {} }; |
| 164 | let section = ""; |
| 165 | const lines = text.split(/\r?\n/); |
| 166 | for (let i = 0; i < lines.length; i++) { |
| 167 | let line = lines[i].replace(/(^|\s)#.*$/, "").trim(); |
| 168 | if (!line) continue; |
| 169 | const header = /^\[\[?([^\]]+)\]\]?$/.exec(line); |
| 170 | if (header) { |
| 171 | section = header[1].trim().replace(/"/g, ""); |
| 172 | out[section] ??= {}; |
| 173 | continue; |
| 174 | } |
| 175 | const eq = line.indexOf("="); |
| 176 | if (eq < 0) continue; |
| 177 | const key = line.slice(0, eq).trim().replace(/"/g, ""); |
| 178 | let value = line.slice(eq + 1).trim(); |
| 179 | // A multi-line array: gather until its closing bracket. |
| 180 | if (value.startsWith("[") && !value.includes("]")) { |
| 181 | while (i + 1 < lines.length && !value.includes("]")) value += " " + lines[++i].replace(/#.*$/, "").trim(); |
| 182 | } |
| 183 | out[section] ??= {}; |
| 184 | out[section][key] = tomlValue(value); |
| 185 | } |
| 186 | return out; |
| 187 | } |
| 188 | |
| 189 | function tomlValue(value: string): unknown { |
| 190 | if (value.startsWith('"') || value.startsWith("'")) return value.slice(1, value.lastIndexOf(value[0])); |
| 191 | if (value === "true" || value === "false") return value === "true"; |
| 192 | if (value.startsWith("[")) { |
| 193 | return [...value.matchAll(/"([^"]*)"|'([^']*)'/g)].map((m) => m[1] ?? m[2]); |
| 194 | } |
| 195 | if (/^-?\d+(\.\d+)?$/.test(value)) return Number(value); |
| 196 | return value; |
| 197 | } |
| 198 | |
| 199 | function str(value: unknown): string | null { |
| 200 | return typeof value === "string" && value.trim() ? value.trim() : null; |
| 201 | } |
| 202 | |
| 203 | function keys(value: unknown): string[] { |
| 204 | return value && typeof value === "object" && !Array.isArray(value) ? Object.keys(value) : []; |
| 205 | } |
| 206 | |
| 207 | /** The package manager a JavaScript project's lockfile says it uses. */ |
| 208 | export function packageManager(siblings: string[]): "pnpm" | "yarn" | "bun" | "npm" | null { |
| 209 | const has = (name: string) => siblings.includes(name); |
| 210 | if (has("pnpm-lock.yaml")) return "pnpm"; |
| 211 | if (has("yarn.lock")) return "yarn"; |
| 212 | if (has("bun.lockb") || has("bun.lock")) return "bun"; |
| 213 | if (has("package-lock.json")) return "npm"; |
| 214 | return null; |
| 215 | } |
| 216 | |
| 217 | function npm(path: string, text: string, ctx: ExtractContext): FileFacts { |
| 218 | const json = JSON.parse(text) as Record<string, unknown>; |
| 219 | const scripts = (json.scripts && typeof json.scripts === "object" ? json.scripts : {}) as Record<string, string>; |
| 220 | const deps = [...keys(json.dependencies), ...keys(json.devDependencies)]; |
| 221 | const typescript = deps.includes("typescript") || ctx.siblings.includes("tsconfig.json"); |
| 222 | const workspaces = Array.isArray(json.workspaces) |
| 223 | ? (json.workspaces as unknown[]).filter((w): w is string => typeof w === "string") |
| 224 | : Array.isArray((json.workspaces as { packages?: unknown })?.packages) |
| 225 | ? ((json.workspaces as { packages: unknown[] }).packages.filter((w): w is string => typeof w === "string")) |
| 226 | : []; |
| 227 | const pm = packageManager(ctx.siblings); |
| 228 | const run = (script: string) => (script === "test" ? `${pm ?? "npm"} test` : `${pm ?? "npm"} run ${script}`); |
| 229 | const hints: Hint[] = []; |
| 230 | if (pm) { |
| 231 | const others = ["npm", "pnpm", "yarn", "bun"].filter((other) => other !== pm).join(" or "); |
| 232 | hints.push({ |
| 233 | kind: "convention", |
| 234 | text: `${ctx.project} uses ${pm} (its lockfile is committed): install with \`${pm} install\`, not ${others}.`, |
| 235 | confidence: 0.9, |
| 236 | evidence: `${path} beside its ${pm} lockfile`, |
| 237 | }); |
| 238 | } |
| 239 | for (const script of ["test", "typecheck", "lint", "build"]) { |
| 240 | const command = str(scripts[script]); |
| 241 | if (!command) continue; |
| 242 | hints.push({ |
| 243 | kind: "fact", |
| 244 | text: clip(`In ${ctx.project}, \`${run(script)}\` runs ${script === "test" ? "the tests" : `the ${script}`}: \`${command}\`.`), |
| 245 | confidence: 0.9, |
| 246 | evidence: `${path} scripts.${script}`, |
| 247 | }); |
| 248 | } |
| 249 | if (workspaces.length > 0) { |
| 250 | hints.push({ |
| 251 | kind: "fact", |
| 252 | text: clip(`${ctx.project} is a monorepo; its packages are in ${workspaces.join(", ")}.`), |
| 253 | confidence: 0.9, |
| 254 | evidence: `${path} workspaces`, |
| 255 | }); |
| 256 | } |
| 257 | const name = str(json.name); |
| 258 | return { |
| 259 | ...EMPTY, |
| 260 | languages: typescript ? ["TypeScript"] : ["JavaScript"], |
| 261 | packages: name |
| 262 | ? [ |
| 263 | { |
| 264 | ecosystem: "npm", |
| 265 | name, |
| 266 | version: str(json.version), |
| 267 | dependencies: deps.slice(0, MAX_DEPENDENCIES), |
| 268 | scripts: Object.fromEntries(Object.entries(scripts).filter(([, v]) => typeof v === "string").slice(0, 20)), |
| 269 | ...(workspaces.length ? { members: workspaces } : {}), |
| 270 | }, |
| 271 | ] |
| 272 | : [], |
| 273 | hints, |
| 274 | }; |
| 275 | } |
| 276 | |
| 277 | function cargo(path: string, text: string, ctx: ExtractContext): FileFacts { |
| 278 | const toml = parseToml(text); |
| 279 | const name = str(toml.package?.name); |
| 280 | const members = Array.isArray(toml.workspace?.members) ? (toml.workspace.members as string[]) : []; |
| 281 | const deps = [...keys(toml.dependencies), ...keys(toml["dev-dependencies"]), ...keys(toml["workspace.dependencies"])]; |
| 282 | const hints: Hint[] = [ |
| 283 | { |
| 284 | kind: "fact", |
| 285 | text: clip( |
| 286 | members.length |
| 287 | ? `${ctx.project} is a Cargo workspace (${members.join(", ")}); \`cargo test\` runs its tests.` |
| 288 | : `${ctx.project} is written in Rust; \`cargo test\` runs its tests.`, |
| 289 | ), |
| 290 | confidence: 0.9, |
| 291 | evidence: path, |
| 292 | }, |
| 293 | ]; |
| 294 | return { |
| 295 | ...EMPTY, |
| 296 | languages: ["Rust"], |
| 297 | packages: name |
| 298 | ? [{ ecosystem: "cargo", name, version: str(toml.package?.version), dependencies: [...new Set(deps)].slice(0, MAX_DEPENDENCIES) }] |
| 299 | : members.length |
| 300 | ? [{ ecosystem: "cargo", name: `${ctx.project} (workspace)`, version: null, dependencies: [...new Set(deps)].slice(0, MAX_DEPENDENCIES), members }] |
| 301 | : [], |
| 302 | hints, |
| 303 | }; |
| 304 | } |
| 305 | |
| 306 | function goMod(path: string, text: string, ctx: ExtractContext): FileFacts { |
| 307 | const module = /^module\s+(\S+)/m.exec(text)?.[1] ?? null; |
| 308 | const version = /^go\s+(\S+)/m.exec(text)?.[1] ?? null; |
| 309 | const requires = [...text.matchAll(/^\s*(?:require\s+)?([\w.-]+\.[\w.-]+\/\S+)\s+v[\w.+-]+/gm)].map((m) => m[1]); |
| 310 | return { |
| 311 | ...EMPTY, |
| 312 | languages: ["Go"], |
| 313 | packages: module ? [{ ecosystem: "go", name: module, version, dependencies: requires.slice(0, MAX_DEPENDENCIES) }] : [], |
| 314 | hints: [ |
| 315 | { |
| 316 | kind: "fact", |
| 317 | text: clip(`${ctx.project} is a Go module${module ? ` (${module}${version ? `, go ${version}` : ""})` : ""}; \`go test ./...\` runs its tests.`), |
| 318 | confidence: 0.9, |
| 319 | evidence: path, |
| 320 | }, |
| 321 | ], |
| 322 | }; |
| 323 | } |
| 324 | |
| 325 | function python(path: string, text: string, ctx: ExtractContext): FileFacts { |
| 326 | let name: string | null = null; |
| 327 | let version: string | null = null; |
| 328 | let deps: string[] = []; |
| 329 | if (base(path) === "pyproject.toml") { |
| 330 | const toml = parseToml(text); |
| 331 | name = str(toml.project?.name) ?? str(toml["tool.poetry"]?.name); |
| 332 | version = str(toml.project?.version) ?? str(toml["tool.poetry"]?.version); |
| 333 | deps = [ |
| 334 | ...(Array.isArray(toml.project?.dependencies) ? (toml.project.dependencies as string[]) : []), |
| 335 | ...keys(toml["tool.poetry.dependencies"]), |
| 336 | ]; |
| 337 | } else { |
| 338 | deps = text.split(/\r?\n/).map((line) => line.replace(/#.*$/, "").trim()).filter((line) => line && !line.startsWith("-")); |
| 339 | } |
| 340 | deps = deps.map((dep) => dep.split(/[<>=!~;\[\s]/)[0]).filter(Boolean); |
| 341 | const hints: Hint[] = deps.includes("pytest") |
| 342 | ? [{ kind: "fact", text: `In ${ctx.project}, \`pytest\` runs the tests.`, confidence: 0.85, evidence: path }] |
| 343 | : []; |
| 344 | return { |
| 345 | ...EMPTY, |
| 346 | languages: ["Python"], |
| 347 | packages: name ? [{ ecosystem: "pypi", name, version, dependencies: deps.slice(0, MAX_DEPENDENCIES) }] : [], |
| 348 | hints, |
| 349 | }; |
| 350 | } |
| 351 | |
| 352 | function wrangler(path: string, text: string): FileFacts { |
| 353 | let config: Record<string, unknown> = {}; |
| 354 | if (path.endsWith(".toml")) { |
| 355 | const toml = parseToml(text); |
| 356 | config = { ...toml[""], routes: Array.isArray(toml[""].routes) ? toml[""].routes : [] }; |
| 357 | if (toml.routes) config.routes = [toml.routes.pattern]; |
| 358 | } else { |
| 359 | config = parseJsonc(text) as Record<string, unknown>; |
| 360 | } |
| 361 | const routes: string[] = []; |
| 362 | const add = (value: unknown) => { |
| 363 | const pattern = typeof value === "string" ? value : str((value as { pattern?: unknown })?.pattern); |
| 364 | if (pattern) routes.push(pattern); |
| 365 | }; |
| 366 | if (Array.isArray(config.routes)) config.routes.forEach(add); |
| 367 | add(config.route); |
| 368 | const name = str(config.name); |
| 369 | if (!name) return EMPTY; |
| 370 | return { |
| 371 | ...EMPTY, |
| 372 | apis: [ |
| 373 | { |
| 374 | kind: "worker", |
| 375 | name, |
| 376 | summary: routes.length ? `Served at ${routes.slice(0, 5).join(", ")}` : "A Worker reached through service bindings", |
| 377 | routes: routes.slice(0, MAX_ROUTES), |
| 378 | }, |
| 379 | ], |
| 380 | }; |
| 381 | } |
| 382 | |
| 383 | const METHODS = ["get", "post", "put", "patch", "delete", "head", "options"]; |
| 384 | |
| 385 | function openapi(path: string, text: string): FileFacts { |
| 386 | const spec = (/\.json$/i.test(path) ? JSON.parse(text) : parseYaml(text)) as { |
| 387 | info?: { title?: unknown; description?: unknown; version?: unknown }; |
| 388 | paths?: Record<string, Record<string, unknown>>; |
| 389 | }; |
| 390 | const routes: string[] = []; |
| 391 | for (const [route, operations] of Object.entries(spec?.paths ?? {})) { |
| 392 | for (const method of Object.keys(operations ?? {})) { |
| 393 | if (METHODS.includes(method)) routes.push(`${method.toUpperCase()} ${route}`); |
| 394 | } |
| 395 | } |
| 396 | const title = str(spec?.info?.title) ?? base(path); |
| 397 | return { |
| 398 | ...EMPTY, |
| 399 | apis: [ |
| 400 | { |
| 401 | kind: "openapi", |
| 402 | name: title, |
| 403 | summary: clip(str(spec?.info?.description) ?? `${routes.length} operations${str(spec?.info?.version) ? `, version ${spec.info!.version}` : ""}`), |
| 404 | routes: routes.slice(0, MAX_ROUTES), |
| 405 | }, |
| 406 | ], |
| 407 | }; |
| 408 | } |
| 409 | |
| 410 | /** A doc split at its headings into pieces of at most `CHUNK_CHARS`. */ |
| 411 | export function chunk(text: string, title: string): string[] { |
| 412 | const sections: string[] = []; |
| 413 | let current = ""; |
| 414 | for (const line of text.split(/\r?\n/)) { |
| 415 | if (/^#{1,3}\s/.test(line) && current.trim()) { |
| 416 | sections.push(current.trim()); |
| 417 | current = ""; |
| 418 | } |
| 419 | current += line + "\n"; |
| 420 | } |
| 421 | if (current.trim()) sections.push(current.trim()); |
| 422 | const pieces: string[] = []; |
| 423 | for (const section of sections) { |
| 424 | const heading = /^#{1,3}\s+(.*)$/m.exec(section)?.[1]?.trim() ?? title; |
| 425 | for (let at = 0; at < section.length && pieces.length < MAX_CHUNKS; at += CHUNK_CHARS) { |
| 426 | const body = section.slice(at, at + CHUNK_CHARS); |
| 427 | pieces.push(at === 0 ? body : `${heading} (continued)\n${body}`); |
| 428 | } |
| 429 | } |
| 430 | return pieces.slice(0, MAX_CHUNKS); |
| 431 | } |
| 432 | |
| 433 | const COMMAND = /^\s*(?:\$\s*)?((?:npm|pnpm|yarn|bun|npx|cargo|go|make|pytest|python3?|poetry|uv|docker|wrangler|just|mix|bundle|rake|gradle|\.\/gradlew|mvn|dotnet)\b[^\n]{0,160})$/; |
| 434 | const SETUP_HEADING = /\b(develop|development|getting started|setup|set up|install|build|test|testing|contribut|running|run locally|local)\b/i; |
| 435 | const CONVENTION_HEADING = /\b(convention|guideline|rules|style|standards|gotcha|pitfall|caveat|known issue|do not|don't)\b/i; |
| 436 | const GOTCHA = /\b(never|don't|do not|must not|careful|beware|gotcha|warning|avoid)\b/i; |
| 437 | |
| 438 | /** What a doc says worth remembering: commands in its setup sections, and its conventions. */ |
| 439 | function docHints(path: string, text: string, role: DocFact["role"], ctx: ExtractContext): Hint[] { |
| 440 | const hints: Hint[] = []; |
| 441 | const agents = role === "agents"; |
| 442 | let heading = ""; |
| 443 | let fenced = false; |
| 444 | for (const raw of text.split(/\r?\n/)) { |
| 445 | const line = raw.trimEnd(); |
| 446 | if (/^\s*(```|~~~)/.test(line)) { |
| 447 | fenced = !fenced; |
| 448 | continue; |
| 449 | } |
| 450 | const h = /^#{1,6}\s+(.*)$/.exec(line); |
| 451 | if (h && !fenced) { |
| 452 | heading = h[1].trim(); |
| 453 | continue; |
| 454 | } |
| 455 | if (fenced) { |
| 456 | const command = COMMAND.exec(line)?.[1]; |
| 457 | if (command && (agents || SETUP_HEADING.test(heading))) { |
| 458 | const what = heading ? heading.replace(/[:.]$/, "").toLowerCase() : "work on it"; |
| 459 | hints.push({ |
| 460 | kind: "fact", |
| 461 | text: clip(`In ${ctx.project}, for ${what}: \`${command.trim()}\`.`), |
| 462 | confidence: agents ? 0.9 : 0.7, |
| 463 | evidence: `${path}${heading ? ` (${heading})` : ""}`, |
| 464 | }); |
| 465 | } |
| 466 | continue; |
| 467 | } |
| 468 | const bullet = /^\s*[-*+]\s+(.*)$/.exec(line)?.[1]?.trim(); |
| 469 | if (!bullet || bullet.length < 20 || bullet.length > MAX_HINT_CHARS) continue; |
| 470 | // A bullet that is only a link is a table of contents. |
| 471 | if (/^\[[^\]]*\]\([^)]*\)\.?$/.test(bullet)) continue; |
| 472 | if (agents || CONVENTION_HEADING.test(heading)) { |
| 473 | hints.push({ |
| 474 | kind: GOTCHA.test(bullet) ? "gotcha" : "convention", |
| 475 | text: clip(bullet.replace(/\*\*/g, "")), |
| 476 | confidence: agents ? 0.9 : 0.7, |
| 477 | evidence: `${path}${heading ? ` (${heading})` : ""}`, |
| 478 | }); |
| 479 | } |
| 480 | } |
| 481 | // The same command under two headings is one hint. |
| 482 | const seen = new Set<string>(); |
| 483 | return hints.filter((hint) => !seen.has(hint.text) && seen.add(hint.text)).slice(0, MAX_HINTS_PER_FILE); |
| 484 | } |
| 485 | |
| 486 | function doc(path: string, text: string, ctx: ExtractContext): FileFacts { |
| 487 | const name = base(path); |
| 488 | const role: DocFact["role"] = /^readme/i.test(name) |
| 489 | ? "readme" |
| 490 | : /^(agents|claude)/i.test(name) |
| 491 | ? "agents" |
| 492 | : /^contributing/i.test(name) |
| 493 | ? "contributing" |
| 494 | : "doc"; |
| 495 | const title = /^#\s+(.*)$/m.exec(text)?.[1]?.trim() ?? name.replace(/\.(md|markdown|txt)$/i, ""); |
| 496 | const paragraph = text |
| 497 | .split(/\r?\n\s*\r?\n/) |
| 498 | .map((block) => block.trim()) |
| 499 | .find((block) => block && !block.startsWith("#") && !block.startsWith("```") && !block.startsWith("<") && !block.startsWith("![")); |
| 500 | return { |
| 501 | ...EMPTY, |
| 502 | doc: { path, title: clip(title, 120), summary: paragraph ? clip(paragraph) : null, chunks: chunk(text, title), role }, |
| 503 | hints: docHints(path, text, role, ctx), |
| 504 | }; |
| 505 | } |
| 506 | |
| 507 | function owners(path: string, text: string): FileFacts { |
| 508 | const names = new Set<string>(); |
| 509 | if (path.endsWith("project.yml")) { |
| 510 | const parsed = parseYaml(text) as { owners?: unknown } | null; |
| 511 | const list = Array.isArray(parsed?.owners) ? parsed.owners : typeof parsed?.owners === "string" ? [parsed.owners] : []; |
| 512 | for (const owner of list) if (typeof owner === "string") names.add(owner.replace(/^@/, "").trim()); |
| 513 | } else { |
| 514 | for (const line of text.split(/\r?\n/)) { |
| 515 | if (line.trim().startsWith("#")) continue; |
| 516 | for (const match of line.matchAll(/@([A-Za-z0-9][A-Za-z0-9_.-]*)(?=\s|$)/g)) names.add(match[1]); |
| 517 | } |
| 518 | } |
| 519 | return { ...EMPTY, owners: [...names].filter(Boolean).slice(0, 20) }; |
| 520 | } |
| 521 | |
| 522 | const TESTS = /\b(test|tests|pytest|vitest|jest|mocha|cargo test|go test|npm test|playwright|cypress)\b/i; |
| 523 | |
| 524 | /** The facts in one file. A file that cannot be parsed says nothing. */ |
| 525 | export function extract(path: string, text: string, ctx: ExtractContext): FileFacts { |
| 526 | const name = base(path); |
| 527 | try { |
| 528 | if (path.startsWith(".g1t/workflows/") || path.startsWith(".github/workflows/")) { |
| 529 | return { ...EMPTY, tests: TESTS.test(text) }; |
| 530 | } |
| 531 | if (path === ".g1t/project.yml" || name === "CODEOWNERS") return owners(path, text); |
| 532 | if (name === "package.json") return npm(path, text, ctx); |
| 533 | if (name === "Cargo.toml") return cargo(path, text, ctx); |
| 534 | if (name === "go.mod") return goMod(path, text, ctx); |
| 535 | if (name === "pyproject.toml" || name === "requirements.txt") return python(path, text, ctx); |
| 536 | if (/^wrangler\.(jsonc|json|toml)$/.test(name)) return wrangler(path, text); |
| 537 | if (OPENAPI_NAME.test(name)) return openapi(path, text); |
| 538 | if (/\.(md|markdown|txt)$/i.test(name) || DOC_NAME.test(name)) return doc(path, text, ctx); |
| 539 | } catch { |
| 540 | return EMPTY; |
| 541 | } |
| 542 | return EMPTY; |
| 543 | } |