Skip to content

g1t/services/runner/src/model-env.ts

635 lines26,508 bytesCodeBlame
1/**
2 * The kinds of work a g1t agent does, as model routes and billing name
3 * them. A workspace routes each to a provider (Integrations → Models).
4 */
5export type AgentTask = "implement" | "review" | "update" | "plan";
6
7/**
8 * What the router routes: every kind of agent job. Revising a change and
9 * answering a question are routed on their own, and go to the workspace's
10 * `implement` route (`taskOf`).
11 */
12export type JobKind = AgentTask | "revise" | "answer";
13
14/** The route and the bill a job goes on. */
15export function taskOf(kind: JobKind): AgentTask {
16 return kind === "revise" || kind === "answer" ? "implement" : kind;
17}
18
19/**
20 * How capable, and how costly, a model is: `small` (fast and cheap, for
21 * work a smaller model does as well), `large` (the standard, most
22 * changes) and `frontier` (the most capable, for hard work only).
23 */
24export type Tier = "small" | "large" | "frontier";
25
26/** Cheapest first. */
27export const TIERS: Tier[] = ["small", "large", "frontier"];
28
29/** Per million tokens, in US dollars: what the provider lists. */
30export type TokenPrice = { input: number; output: number; cacheRead: number; cacheWrite: number };
31
32/** Where one kind of work goes: what people see, and what is sent. */
33export type ModelRoute = {
34 /** The model's public name, e.g. `Claude Sonnet 5.5`. */
35 modelName: string;
36 /** The identifier sent to the provider. */
37 model: string;
38 /**
39 * The provider's list price, for estimates (the savings report, routing
40 * by cost). Never what anyone is charged: runs are charged what AI
41 * Gateway priced them at.
42 */
43 price?: TokenPrice;
44 /**
45 * What the model can do, from the catalogue (`effort`, `thinking`…);
46 * absent when the route came from configuration, which says nothing.
47 */
48 capabilities?: string[];
49};
50
51/**
52 * How hard the model thinks before it answers, on models that take it
53 * (Claude Haiku 5.5 and later): more effort, more thinking tokens.
54 */
55export type Effort = "low" | "medium" | "high" | "xhigh" | "max";
56
57export const EFFORTS: Effort[] = ["low", "medium", "high", "xhigh", "max"];
58
59/** How a job's rule decides: a tier, or `change` to size the change it reads. */
60export type TaskRule = Tier | "change";
61
62/**
63 * Learning from a repository's own runs: of its last `window` runs of the
64 * same kind, a cheaper tier that finished at least `stepDownAt` of at
65 * least `minRuns` takes the work; a tier that finished less than
66 * `stepUpAt` of at least `minRuns` hands it up.
67 */
68export type Learning = { window: number; minRuns: number; stepDownAt: number; stepUpAt: number };
69
70/**
71 * g1t's routing policy. Nobody assigning an agent has to pick a model:
72 * "Auto" decides here, by the work. Staff choose the model behind each
73 * tier, the background model and each job's tier and effort in sudo
74 * (billing's `model_defaults`, applied by `withDefaults`); the rest of the
75 * policy (labels, change sizes, learning) is `AGENT_ROUTING` in
76 * wrangler.jsonc, which is also the whole of it when billing cannot be
77 * read. Never in code.
78 */
79export type AgentRouting = {
80 /** The model behind each tier: the catalogue. */
81 tiers: Record<Tier, ModelRoute>;
82 /** The tier each kind of job starts from, or `change` to size it. */
83 tasks: Record<JobKind, TaskRule>;
84 /**
85 * The effort each kind of job runs at, whatever tier it lands on; left
86 * out, the harness's own default. Only on g1t's tiers: a route that
87 * names its own model is sent as it is.
88 */
89 effort: Partial<Record<JobKind, Effort>>;
90 /** The largest change `change` sends to the small tier. */
91 smallChange: { files: number; lines: number };
92 /** A change larger than this (either) is reviewed on the frontier tier. */
93 largeChange: { files: number; lines: number };
94 /** Issue labels that keep work off the small tier. */
95 largeLabels: string[];
96 /** Issue labels that send work to the frontier tier. */
97 frontierLabels: string[];
98 /** Issue labels that let changes and answers start on the small tier. */
99 smallLabels: string[];
100 /** Failed attempts at the same work, in a row, before the frontier tier. */
101 frontierAfter: number;
102 learning: Learning;
103 /**
104 * The harness's own small background tasks (titles, summaries): the
105 * small tier's model when absent.
106 */
107 background?: ModelRoute;
108 /**
109 * Why a tier's model is not the one staff chose (it was retired, or has
110 * no price): said in the run's reason line.
111 */
112 notes?: Partial<Record<Tier, string>>;
113};
114
115/** What a change is, as far as routing cares. */
116export type ChangeSize = {
117 files: number;
118 /** Lines added and removed. */
119 lines: number;
120 /**
121 * What it touches that runs, configures or guards things: CI, secrets,
122 * infrastructure, ownership (work's confidence.rs `sensitive`).
123 */
124 sensitive: string[];
125};
126
127/** One past run of the same kind of job in the repository, for learning. */
128export type PastOutcome = {
129 /** The tier it ran on, when it ran on one of g1t's. */
130 tier: Tier | null;
131 /** It finished, and did not leave a change g1t had low confidence in. */
132 ok: boolean;
133};
134
135/** What g1t knows about one piece of work when it routes it. */
136export type RouteSignals = {
137 /** The change the work reads; null or absent when g1t does not know it. */
138 change?: ChangeSize | null;
139 /** Labels on the issue the work is for. */
140 labels?: string[];
141 /** The last attempt at the same work failed. Same as `failures: 1`. */
142 retry?: boolean;
143 /** Failed attempts at the same work, in a row, most recent last. */
144 failures?: number;
145 /** The last attempt finished, but left a change g1t has low confidence in. */
146 lowConfidence?: boolean;
147 /** Recent runs of the same kind in this repository, newest first. */
148 history?: PastOutcome[];
149 /** A tier the workspace chose for this work instead of Auto. */
150 chosen?: Tier | null;
151};
152
153/** The router's answer: the tier, and why, in one line people can read. */
154export type Routed = {
155 tier: Tier;
156 /** E.g. `Used a fast model (Claude Haiku 5.5): small change, 3 files and 80 lines.` */
157 reason: string;
158};
159
160/** The routing g1t ships with, for whatever the configuration leaves out. */
161export const DEFAULT_ROUTING: AgentRouting = {
162 tiers: {
163 // Prompts up to 100,000 tokens; past that, five times as much.
164 small: {
165 modelName: "Claude Haiku 5.5",
166 model: "claude-haiku-5-5",
167 price: { input: 0.1, output: 0.5, cacheRead: 0.01, cacheWrite: 0.125 },
168 },
169 large: {
170 modelName: "Claude Sonnet 5.5",
171 model: "claude-sonnet-5-5",
172 price: { input: 2, output: 10, cacheRead: 0.1, cacheWrite: 2.5 },
173 },
174 frontier: {
175 modelName: "Claude Opus 5.5",
176 model: "claude-opus-5-5",
177 price: { input: 4, output: 20, cacheRead: 0.2, cacheWrite: 5 },
178 },
179 },
180 // Plans start on the fast model thinking hard, and go up a tier when
181 // one fails or leaves low confidence, as any work does.
182 tasks: { implement: "large", revise: "large", answer: "small", review: "change", update: "small", plan: "small" },
183 effort: { plan: "high", answer: "medium", update: "low" },
184 smallChange: { files: 10, lines: 200 },
185 largeChange: { files: 60, lines: 3000 },
186 largeLabels: ["security"],
187 frontierLabels: ["architecture"],
188 smallLabels: ["documentation", "docs", "typo"],
189 frontierAfter: 2,
190 learning: { window: 20, minRuns: 5, stepDownAt: 0.9, stepUpAt: 0.5 },
191};
192
193function isTier(value: unknown): value is Tier {
194 return value === "small" || value === "large" || value === "frontier";
195}
196
197/**
198 * The routing in `AGENT_ROUTING`, with anything it leaves out taken from
199 * `DEFAULT_ROUTING`. An unset or unreadable value is the default, and so is
200 * any single rule that names no tier.
201 */
202export function parseRouting(json: string | undefined): AgentRouting {
203 let given: Partial<AgentRouting> = {};
204 try {
205 const parsed: unknown = json ? JSON.parse(json) : {};
206 given = parsed && typeof parsed === "object" && !Array.isArray(parsed) ? (parsed as Partial<AgentRouting>) : {};
207 } catch {
208 console.log("AGENT_ROUTING is not JSON; using the default routing");
209 }
210 const tasks = { ...DEFAULT_ROUTING.tasks };
211 for (const [kind, rule] of Object.entries(given.tasks ?? {})) {
212 if (kind in tasks && (isTier(rule) || rule === "change")) tasks[kind as JobKind] = rule;
213 }
214 const effort = { ...DEFAULT_ROUTING.effort };
215 for (const [kind, level] of Object.entries(given.effort ?? {})) {
216 if (kind in tasks && EFFORTS.includes(level as Effort)) effort[kind as JobKind] = level as Effort;
217 }
218 const tiers = { ...DEFAULT_ROUTING.tiers };
219 for (const tier of TIERS) {
220 const route = given.tiers?.[tier];
221 if (route && typeof route.model === "string" && route.model) {
222 // A model named without a price has none: estimates leave it out
223 // rather than price it as another model.
224 tiers[tier] = { modelName: route.modelName || route.model, model: route.model, ...(route.price ? { price: route.price } : {}) };
225 }
226 }
227 const labels = (list: unknown, fallback: string[]) =>
228 Array.isArray(list) ? list.filter((label): label is string => typeof label === "string") : fallback;
229 return {
230 tiers,
231 tasks,
232 effort,
233 smallChange: { ...DEFAULT_ROUTING.smallChange, ...given.smallChange },
234 largeChange: { ...DEFAULT_ROUTING.largeChange, ...given.largeChange },
235 largeLabels: labels(given.largeLabels, DEFAULT_ROUTING.largeLabels),
236 frontierLabels: labels(given.frontierLabels, DEFAULT_ROUTING.frontierLabels),
237 smallLabels: labels(given.smallLabels, DEFAULT_ROUTING.smallLabels),
238 frontierAfter: typeof given.frontierAfter === "number" && given.frontierAfter >= 1 ? given.frontierAfter : DEFAULT_ROUTING.frontierAfter,
239 learning: { ...DEFAULT_ROUTING.learning, ...given.learning },
240 };
241}
242
243/** How each tier is named to people. */
244export const TIER_LABEL: Record<Tier, { noun: string; used: string }> = {
245 small: { noun: "fast", used: "Used a fast model" },
246 large: { noun: "standard", used: "Used the standard model" },
247 frontier: { noun: "most capable", used: "Used the most capable model" },
248};
249
250const up = (tier: Tier): Tier => TIERS[Math.min(TIERS.indexOf(tier) + 1, TIERS.length - 1)];
251const down = (tier: Tier): Tier => TIERS[Math.max(TIERS.indexOf(tier) - 1, 0)];
252const rank = (tier: Tier) => TIERS.indexOf(tier);
253
254function plural(n: number, one: string, many: string): string {
255 return `${n} ${n === 1 ? one : many}`;
256}
257
258/** How a tier did in the repository's recent runs of the same kind. */
259export function record(history: PastOutcome[], tier: Tier, window: number): { runs: number; ok: number } {
260 const recent = history.slice(0, window).filter((run) => run.tier === tier);
261 return { runs: recent.length, ok: recent.filter((run) => run.ok).length };
262}
263
264/**
265 * Where one job runs, and why. In order:
266 *
267 * 1. A tier the workspace chose for this work is used as chosen.
268 * 2. The job's rule gives the starting tier: a fixed tier, or for
269 * `change`, the change's size (small and touching nothing sensitive:
270 * small; larger than `largeChange`: frontier; unknown or anything
271 * else: large). Issue labels move it: `frontierLabels` to the frontier,
272 * `largeLabels` off the small tier, `smallLabels` let a change or an
273 * answer start small.
274 * 3. Escalation: `frontierAfter` failures in a row go to the frontier; one
275 * failure, or a last attempt that left low confidence, one tier up.
276 * 4. Otherwise, learning from the repository's own runs of the same kind:
277 * one tier down when the cheaper tier finished nearly all of its recent
278 * ones (never for sensitive or labelled work), one tier up when this
279 * tier failed half of its own.
280 */
281export function route(kind: JobKind, signals: RouteSignals, routing: AgentRouting = DEFAULT_ROUTING): Routed {
282 const say = (tier: Tier, why: string): Routed => ({
283 tier,
284 reason: `${TIER_LABEL[tier].used} (${routing.tiers[tier].modelName}): ${why}.${routing.notes?.[tier] ? ` ${routing.notes[tier]}` : ""}`,
285 });
286 if (signals.chosen && isTier(signals.chosen)) {
287 return say(signals.chosen, `the workspace chose the ${TIER_LABEL[signals.chosen].noun} model for this work`);
288 }
289
290 const labels = new Set((signals.labels ?? []).map((label) => label.toLowerCase()));
291 const has = (list: string[]) => list.find((label) => labels.has(label.toLowerCase()));
292 const rule = routing.tasks[kind] ?? "large";
293 let tier: Tier;
294 let why: string;
295 // Sensitive or labelled work is never stepped down by learning.
296 let pinned = false;
297 if (rule === "change") {
298 const change = signals.change;
299 if (!change || change.files === 0) {
300 [tier, why] = ["large", "the change's size is not known"];
301 } else if (change.sensitive.length > 0) {
302 [tier, why, pinned] = ["large", `it touches ${change.sensitive.join(", ")}`, true];
303 } else if (change.files > routing.largeChange.files || change.lines > routing.largeChange.lines) {
304 [tier, why] = ["frontier", `large change, ${plural(change.files, "file", "files")} and ${plural(change.lines, "line", "lines")}`];
305 } else if (change.files <= routing.smallChange.files && change.lines <= routing.smallChange.lines) {
306 [tier, why] = ["small", `small change, ${plural(change.files, "file", "files")} and ${plural(change.lines, "line", "lines")}`];
307 } else {
308 [tier, why] = ["large", `a change of ${plural(change.files, "file", "files")} and ${plural(change.lines, "line", "lines")}`];
309 }
310 } else {
311 tier = rule;
312 why = DEFAULT_WHY[kind];
313 }
314 const frontierLabel = has(routing.frontierLabels);
315 const largeLabel = has(routing.largeLabels);
316 const smallLabel = has(routing.smallLabels);
317 if (frontierLabel) {
318 [tier, why, pinned] = ["frontier", `the issue is labelled ${frontierLabel}`, true];
319 } else if (largeLabel && tier === "small") {
320 [tier, why, pinned] = ["large", `the issue is labelled ${largeLabel}`, true];
321 } else if (largeLabel) {
322 pinned = true;
323 } else if (smallLabel && tier === "large" && (kind === "implement" || kind === "revise" || kind === "answer")) {
324 [tier, why] = ["small", `the issue is labelled ${smallLabel}`];
325 }
326
327 const failures = Math.max(signals.failures ?? 0, signals.retry ? 1 : 0);
328 if (failures >= routing.frontierAfter) {
329 return say("frontier", `the last ${plural(failures, "attempt", "attempts")} at this work failed`);
330 }
331 if (failures > 0) {
332 return tier === "frontier" ? say(tier, `${why}; the last attempt failed`) : say(up(tier), "the last attempt at this work failed");
333 }
334 if (signals.lowConfidence) {
335 return tier === "frontier" ? say(tier, why) : say(up(tier), "the last attempt left a change g1t was not confident in");
336 }
337
338 const history = signals.history ?? [];
339 const { window, minRuns, stepDownAt, stepUpAt } = routing.learning;
340 const here = record(history, tier, window);
341 if (here.runs >= minRuns && here.ok / here.runs < stepUpAt && tier !== "frontier") {
342 const failed = here.runs - here.ok;
343 return say(up(tier), `the ${TIER_LABEL[tier].noun} model failed ${failed} of its last ${here.runs} runs like this here`);
344 }
345 if (!pinned && tier !== "small") {
346 const cheaper = record(history, down(tier), window);
347 if (cheaper.runs >= minRuns && cheaper.ok / cheaper.runs >= stepDownAt) {
348 return say(down(tier), `it finished ${cheaper.ok} of its last ${cheaper.runs} runs like this here`);
349 }
350 }
351 return say(tier, why);
352}
353
354/** Why each kind of job starts where it does, when nothing else decides. */
355const DEFAULT_WHY: Record<JobKind, string> = {
356 implement: "making a change",
357 revise: "revising a change",
358 answer: "answering a question",
359 review: "reviewing a change",
360 update: "catching up with the base branch",
361 plan: "planning work",
362};
363
364/**
365 * The tier one piece of work runs on: `route`'s tier, for callers that
366 * need no reason.
367 */
368export function chooseTier(kind: JobKind, signals: RouteSignals, routing: AgentRouting = DEFAULT_ROUTING): Tier {
369 return route(kind, signals, routing).tier;
370}
371
372/** The tier a model ran as, by its public name or id; null when none of g1t's. */
373export function tierOfModel(model: string | null | undefined, routing: AgentRouting): Tier | null {
374 if (!model) return null;
375 return TIERS.find((tier) => routing.tiers[tier].modelName === model || routing.tiers[tier].model === model) ?? null;
376}
377
378/** The settings that decide where model requests go. */
379export type ModelRouting = {
380 /**
381 * The provider's key. Not needed when the gateway holds it and requests
382 * authenticate to the gateway instead.
383 */
384 ANTHROPIC_API_KEY?: string;
385 /** A Cloudflare AI Gateway id; empty sends requests to the provider directly. */
386 AI_GATEWAY_ID: string;
387 CLOUDFLARE_ACCOUNT_ID: string;
388 /** Authenticates to the gateway, if it requires it. */
389 AI_GATEWAY_TOKEN?: string;
390};
391
392/**
393 * What a run is for, attached to each of its requests at the gateway.
394 * `session` is the run's id there: billing finds the run's requests by it
395 * and settles the run to what the gateway priced them at.
396 */
397export type RunTags = { repo: string; pull: number; session?: string };
398
399/**
400 * A session id for a run that goes straight to the gateway (no model
401 * proxy): `rs_` and 24 hex characters, which billing's log filter needs
402 * no escaping for.
403 */
404export function gatewaySession(): string {
405 const bytes = crypto.getRandomValues(new Uint8Array(12));
406 return `rs_${Array.from(bytes, (b) => b.toString(16).padStart(2, "0")).join("")}`;
407}
408
409/** Whether there is a way to reach a model at all. */
410export function canReachModel(env: ModelRouting): boolean {
411 return Boolean(env.ANTHROPIC_API_KEY || (env.AI_GATEWAY_ID && env.AI_GATEWAY_TOKEN));
412}
413
414/**
415 * The model variables of a run on g1t's hosted models: the tier's model
416 * for the work, and the small tier's for the harness's own small tasks.
417 */
418export function tierVars(routing: AgentRouting, tier: Tier): Record<string, string> {
419 const route = routing.tiers[tier];
420 const background = (routing.background ?? routing.tiers.small).model;
421 return {
422 ANTHROPIC_MODEL: route.model,
423 // Recorded at the top of the session, so anyone can see what ran.
424 AGENT_MODEL_NAME: route.modelName,
425 ANTHROPIC_DEFAULT_HAIKU_MODEL: background,
426 ANTHROPIC_SMALL_FAST_MODEL: background,
427 };
428}
429
430/**
431 * The effort a job runs at on a tier: the job's, unless the catalogue says
432 * the tier's model takes no effort level (Claude Haiku 4.5), when the
433 * harness's own default is left.
434 */
435export function effortFor(routing: AgentRouting, kind: JobKind, tier: Tier): Effort | undefined {
436 const effort = routing.effort[kind];
437 const capabilities = routing.tiers[tier].capabilities;
438 if (effort && capabilities && !capabilities.includes("effort")) return undefined;
439 return effort;
440}
441
442/** What `model_defaults` gives: billing's `ModelDefaults`, as far as routing reads it. */
443export type StoredDefaults = {
444 models: {
445 purpose: string;
446 model: {
447 model: string;
448 name: string;
449 inputMicros: number;
450 outputMicros: number;
451 cacheReadMicros: number;
452 cacheWriteMicros: number;
453 } | null;
454 capabilities: string[];
455 note: string | null;
456 }[];
457 jobs: { kind: string; tier: string; effort: string | null }[];
458};
459
460const TIER_PURPOSE: Record<Tier, string> = { small: "tier_small", large: "tier_large", frontier: "tier_frontier" };
461
462/**
463 * `routing` with staff's defaults from billing's catalogue on top: the
464 * model behind each tier (named as the catalogue names it, priced from
465 * it), the background model, and each job's starting tier and effort. A
466 * purpose billing left out, or could not find any model for, keeps what
467 * `routing` had; a fallback's note is said in the run's reason line.
468 */
469export function withDefaults(routing: AgentRouting, stored: StoredDefaults): AgentRouting {
470 const perMillion = (micros: number) => Math.max(0, micros) / 1_000_000;
471 const routeOf = (purpose: string): { route: ModelRoute; note: string | null } | null => {
472 const found = stored.models.find((entry) => entry.purpose === purpose);
473 if (!found?.model?.model) return null;
474 const { model } = found;
475 return {
476 route: {
477 modelName: model.name || model.model,
478 model: model.model,
479 price: {
480 input: perMillion(model.inputMicros),
481 output: perMillion(model.outputMicros),
482 cacheRead: perMillion(model.cacheReadMicros),
483 cacheWrite: perMillion(model.cacheWriteMicros),
484 },
485 capabilities: found.capabilities,
486 },
487 note: found.note,
488 };
489 };
490 const tiers = { ...routing.tiers };
491 const notes: Partial<Record<Tier, string>> = { ...routing.notes };
492 for (const tier of TIERS) {
493 const found = routeOf(TIER_PURPOSE[tier]);
494 if (!found) continue;
495 tiers[tier] = found.route;
496 if (found.note) notes[tier] = found.note;
497 else delete notes[tier];
498 }
499 const tasks = { ...routing.tasks };
500 const effort = { ...routing.effort };
501 for (const job of stored.jobs) {
502 if (!(job.kind in tasks)) continue;
503 const kind = job.kind as JobKind;
504 // Only a review is sized by the change it reads.
505 if (isTier(job.tier) || (job.tier === "change" && kind === "review")) tasks[kind] = job.tier as TaskRule;
506 if (job.effort && EFFORTS.includes(job.effort as Effort)) effort[kind] = job.effort as Effort;
507 else delete effort[kind];
508 }
509 const background = routeOf("background")?.route ?? routing.background;
510 return { ...routing, tiers, tasks, effort, notes, ...(background ? { background } : {}) };
511}
512
513/**
514 * The routing to use now: staff's defaults from billing, read at most once
515 * a minute per isolate, on top of `AGENT_ROUTING`. When billing cannot be
516 * read, `AGENT_ROUTING` (and `DEFAULT_ROUTING` under it) alone, and billing
517 * is asked again ten seconds later.
518 */
519export function routingReader(ttlMs = 60_000, retryMs = 10_000) {
520 let cached: { value: StoredDefaults | null; until: number } | null = null;
521 return async (configured: string | undefined, read: () => Promise<StoredDefaults>, now = Date.now()): Promise<AgentRouting> => {
522 const base = parseRouting(configured);
523 if (!cached || cached.until <= now) {
524 try {
525 cached = { value: await read(), until: now + ttlMs };
526 } catch (error) {
527 console.log(`model defaults unreadable, using AGENT_ROUTING: ${error instanceof Error ? error.message : String(error)}`);
528 cached = { value: null, until: now + retryMs };
529 }
530 }
531 return cached.value ? withDefaults(base, cached.value) : base;
532 };
533}
534
535/** Where the sandbox sends model requests, and what it sends with them. */
536export function modelEnv(
537 env: ModelRouting,
538 routing: AgentRouting,
539 task: AgentTask,
540 tier: Tier,
541 tags: RunTags,
542): Record<string, string> {
543 const vars = tierVars(routing, tier);
544 if (env.ANTHROPIC_API_KEY) vars.ANTHROPIC_API_KEY = env.ANTHROPIC_API_KEY;
545 if (!env.AI_GATEWAY_ID) return vars;
546
547 vars.ANTHROPIC_BASE_URL = `https://gateway.ai.cloudflare.com/v1/${env.CLOUDFLARE_ACCOUNT_ID}/${env.AI_GATEWAY_ID}/anthropic`;
548 // The gateway logs these with every request, so spend and failures can
549 // be read per kind of work, tier, repository and pull request; and by
550 // the run's session, which billing settles the run's charge by.
551 const headers = [`cf-aig-metadata: ${JSON.stringify({ task, tier, ...tags })}`];
552 if (env.AI_GATEWAY_TOKEN) {
553 vars.AI_GATEWAY_TOKEN = env.AI_GATEWAY_TOKEN;
554 headers.push(`cf-aig-authorization: Bearer ${env.AI_GATEWAY_TOKEN}`);
555 // With the provider's key stored in the gateway, the sandbox never
556 // holds it. The harness still wants the variable set.
557 vars.ANTHROPIC_API_KEY ??= env.AI_GATEWAY_TOKEN;
558 }
559 vars.ANTHROPIC_CUSTOM_HEADERS = headers.join("\n");
560 return vars;
561}
562
563/** Lines added and removed across a change's files. */
564export function changeSize(files: { additions: number; deletions: number }[], sensitive: string[]): ChangeSize {
565 return {
566 files: files.length,
567 lines: files.reduce((sum, file) => sum + file.additions + file.deletions, 0),
568 sensitive,
569 };
570}
571
572/** A past run of the same work, as the work service lists it. */
573export type PastAttempt = {
574 status: string;
575 halted?: string | null;
576 title?: string | null;
577 /** The pull request or issue it was for. */
578 number?: number | null;
579 /** The model it ran on, by its public name. */
580 model?: string | null;
581 confidence?: { level: string } | null;
582};
583
584/**
585 * Whether the latest attempt at the same work failed: it failed, or g1t
586 * stopped it at a cap of its guardrails. A person stopping it is not a
587 * failure. `title` narrows it to the same plan, whose runs have no pull
588 * request to tell them apart.
589 */
590export function lastAttemptFailed(newestFirst: PastAttempt[], title?: string): boolean {
591 const last = newestFirst[0];
592 if (!last) return false;
593 if (title !== undefined && (last.title ?? "").trim() !== title.trim()) return false;
594 return last.status === "failed" || (last.status === "stopped" && Boolean(last.halted));
595}
596
597/** Whether a past run failed: it failed, or g1t stopped it at a cap of its guardrails. */
598function failed(run: PastAttempt): boolean {
599 return run.status === "failed" || (run.status === "stopped" && Boolean(run.halted));
600}
601
602/**
603 * How many of the latest attempts at the same work failed in a row. A
604 * finished one, or a person stopping one, ends the count. `title` narrows
605 * it to the same plan, as for `lastAttemptFailed`.
606 */
607export function failuresInARow(newestFirst: PastAttempt[], title?: string): number {
608 let count = 0;
609 for (const run of newestFirst) {
610 if (title !== undefined && (run.title ?? "").trim() !== title.trim()) break;
611 if (!failed(run)) break;
612 count += 1;
613 }
614 return count;
615}
616
617/** Whether the latest attempt finished but left a change g1t was not confident in. */
618export function leftLowConfidence(newestFirst: PastAttempt[]): boolean {
619 const last = newestFirst[0];
620 return Boolean(last && last.status === "succeeded" && last.confidence?.level === "low");
621}
622
623/**
624 * The repository's recent runs of one kind, as learning reads them: the
625 * tier each ran on, and whether it did the work. Runs still going say
626 * nothing yet, and a person stopping one is not the model's failure.
627 */
628export function outcomesOf(newestFirst: PastAttempt[], routing: AgentRouting): PastOutcome[] {
629 return newestFirst
630 .filter((run) => run.status === "succeeded" || failed(run))
631 .map((run) => ({
632 tier: tierOfModel(run.model, routing),
633 ok: run.status === "succeeded" && run.confidence?.level !== "low",
634 }));
635}