Skip to content

g1t/services/runner/src/model-env.test.ts

327 lines17,127 bytesCodeBlame
1import assert from "node:assert/strict";
2import { createServer } from "node:http";
3import { test } from "node:test";
4
5import {
6 type AgentRouting,
7 type ChangeSize,
8 DEFAULT_ROUTING,
9 canReachModel,
10 changeSize,
11 chooseTier,
12 failuresInARow,
13 gatewaySession,
14 lastAttemptFailed,
15 leftLowConfidence,
16 modelEnv,
17 outcomesOf,
18 parseRouting,
19 route,
20 taskOf,
21 tierOfModel,
22} from "./model-env.ts";
23
24const routes: AgentRouting = {
25 ...DEFAULT_ROUTING,
26 tiers: {
27 small: { modelName: "Claude Haiku 4.5", model: "claude-haiku-4-5-20251001" },
28 large: { modelName: "Claude Sonnet 5.5", model: "claude-sonnet-5-5" },
29 frontier: { modelName: "Claude Opus 5.5", model: "claude-opus-5-5" },
30 },
31};
32const tags = { repo: "acme/site", pull: 12 };
33const direct = { ANTHROPIC_API_KEY: "sk-test", AI_GATEWAY_ID: "", CLOUDFLARE_ACCOUNT_ID: "acct" };
34
35/** `ANTHROPIC_CUSTOM_HEADERS` as the harness reads it: one header per line. */
36function customHeaders(vars: Record<string, string>): Record<string, string> {
37 return Object.fromEntries(
38 vars.ANTHROPIC_CUSTOM_HEADERS.split("\n").map((line) => {
39 const at = line.indexOf(": ");
40 return [line.slice(0, at), line.slice(at + 2)];
41 }),
42 );
43}
44
45const small: ChangeSize = { files: 3, lines: 80, sensitive: [] };
46
47test("Auto starts each kind of job on its tier: fast for catching up, answering and plans, standard for changes", () => {
48 assert.equal(chooseTier("update", {}, routes), "small");
49 assert.equal(chooseTier("answer", {}, routes), "small");
50 assert.equal(chooseTier("implement", {}, routes), "large");
51 assert.equal(chooseTier("revise", {}, routes), "large");
52 assert.equal(chooseTier("implement", { change: small }, routes), "large");
53 assert.equal(chooseTier("plan", {}, routes), "small");
54 assert.equal(chooseTier("plan", { labels: ["architecture"] }, routes), "frontier");
55});
56
57test("revising and answering go to the workspace's implement route and bill", () => {
58 assert.equal(taskOf("revise"), "implement");
59 assert.equal(taskOf("answer"), "implement");
60 assert.equal(taskOf("review"), "review");
61 assert.equal(taskOf("plan"), "plan");
62});
63
64test("a review is sized by its change: fast when small and safe, most capable when large", () => {
65 assert.equal(chooseTier("review", { change: small }, routes), "small");
66 assert.equal(chooseTier("review", { change: { ...small, lines: 200, files: 10 } }, routes), "small");
67 assert.equal(chooseTier("review", { change: { ...small, lines: 201 } }, routes), "large");
68 assert.equal(chooseTier("review", { change: { ...small, files: 11 } }, routes), "large");
69 assert.equal(chooseTier("review", { change: { ...small, sensitive: ["CI workflows"] } }, routes), "large");
70 assert.equal(chooseTier("review", { change: small, labels: ["Security"] }, routes), "large");
71 assert.equal(chooseTier("review", { change: small, labels: ["docs"] }, routes), "small");
72 assert.equal(chooseTier("review", { change: { ...small, files: 61 } }, routes), "frontier");
73 assert.equal(chooseTier("review", { change: { ...small, lines: 3001 } }, routes), "frontier");
74});
75
76test("a review of a change g1t cannot size runs on the standard tier", () => {
77 assert.equal(chooseTier("review", {}, routes), "large");
78 assert.equal(chooseTier("review", { change: null }, routes), "large");
79 assert.equal(chooseTier("review", { change: { files: 0, lines: 0, sensitive: [] } }, routes), "large");
80});
81
82test("labels move work: architecture to the most capable, documentation to the fast tier", () => {
83 assert.equal(chooseTier("implement", { labels: ["Architecture"] }, routes), "frontier");
84 assert.equal(chooseTier("review", { change: small, labels: ["architecture"] }, routes), "frontier");
85 assert.equal(chooseTier("implement", { labels: ["docs"] }, routes), "small");
86 assert.equal(chooseTier("answer", { labels: ["typo"] }, routes), "small");
87 // A small label never takes a review or a plan down.
88 assert.equal(chooseTier("plan", { labels: ["docs"] }, routes), "small");
89 assert.equal(chooseTier("plan", { labels: ["security"] }, routes), "large");
90 // Security outranks documentation.
91 assert.equal(chooseTier("implement", { labels: ["docs", "security"] }, routes), "large");
92});
93
94test("a failed attempt goes one tier up, and repeated failures to the most capable", () => {
95 assert.equal(chooseTier("update", { retry: true }, routes), "large");
96 assert.equal(chooseTier("review", { change: small, retry: true }, routes), "large");
97 assert.equal(chooseTier("implement", { failures: 1 }, routes), "frontier");
98 assert.equal(chooseTier("update", { failures: 2 }, routes), "frontier");
99 assert.equal(chooseTier("plan", { failures: 1 }, routes), "large");
100 const why = route("update", { failures: 2 }, routes).reason;
101 assert.equal(why, "Used the most capable model (Claude Opus 5.5): the last 2 attempts at this work failed.");
102});
103
104test("a change left at low confidence sends the next attempt one tier up", () => {
105 assert.equal(chooseTier("revise", { lowConfidence: true }, routes), "frontier");
106 assert.equal(chooseTier("update", { lowConfidence: true }, routes), "large");
107});
108
109test("a tier the workspace chose wins over Auto", () => {
110 const chosen = route("review", { change: small, failures: 3, chosen: "large" }, routes);
111 assert.equal(chosen.tier, "large");
112 assert.equal(chosen.reason, "Used the standard model (Claude Sonnet 5.5): the workspace chose the standard model for this work.");
113});
114
115const ok = (tier: "small" | "large" | "frontier", n: number) => Array.from({ length: n }, () => ({ tier, ok: true }));
116const bad = (tier: "small" | "large" | "frontier", n: number) => Array.from({ length: n }, () => ({ tier, ok: false }));
117
118test("a repository whose fast runs finish nearly always steps the work down", () => {
119 const history = [...ok("small", 9), ...bad("small", 1)];
120 const routed = route("implement", { history }, routes);
121 assert.equal(routed.tier, "small");
122 assert.equal(routed.reason, "Used a fast model (Claude Haiku 4.5): it finished 9 of its last 10 runs like this here.");
123 // Too few runs to tell, or not quite enough of them finished: no change.
124 assert.equal(chooseTier("implement", { history: ok("small", 4) }, routes), "large");
125 assert.equal(chooseTier("implement", { history: [...ok("small", 8), ...bad("small", 2)] }, routes), "large");
126 // Plans step down the same way.
127 assert.equal(chooseTier("plan", { history: ok("small", 6) }, routes), "small");
128});
129
130test("learning never steps down sensitive or labelled work, nor a retry", () => {
131 const history = ok("small", 10);
132 assert.equal(chooseTier("review", { change: { ...small, files: 20, sensitive: ["secrets"] }, history }, routes), "large");
133 assert.equal(chooseTier("implement", { labels: ["security"], history }, routes), "large");
134 assert.equal(chooseTier("implement", { failures: 1, history }, routes), "frontier");
135});
136
137test("a tier that fails half its runs in a repository hands the work up", () => {
138 const history = [...bad("large", 3), ...ok("large", 2)];
139 const routed = route("implement", { history }, routes);
140 assert.equal(routed.tier, "frontier");
141 assert.equal(routed.reason, "Used the most capable model (Claude Opus 5.5): the standard model failed 3 of its last 5 runs like this here.");
142});
143
144test("past runs are read by the tier they ran on and whether they did the work", () => {
145 const runs = [
146 { status: "running", model: "Claude Haiku 4.5" },
147 { status: "succeeded", model: "Claude Haiku 4.5" },
148 { status: "succeeded", model: "Claude Sonnet 5.5", confidence: { level: "low" } },
149 { status: "failed", model: "Claude Opus 5.5" },
150 { status: "stopped", halted: null, model: "Claude Haiku 4.5" },
151 { status: "stopped", halted: "budget", model: "claude-haiku-4-5-20251001" },
152 { status: "succeeded", model: "Claude Sonnet 5.5, through Acme" },
153 ];
154 assert.deepEqual(outcomesOf(runs, routes), [
155 { tier: "small", ok: true },
156 { tier: "large", ok: false },
157 { tier: "frontier", ok: false },
158 { tier: "small", ok: false },
159 { tier: null, ok: true },
160 ]);
161 assert.equal(tierOfModel("Claude Opus 5.5", routes), "frontier");
162 assert.equal(tierOfModel(null, routes), null);
163});
164
165test("failures are counted in a row, newest first, and confidence is read from the last finished one", () => {
166 assert.equal(failuresInARow([]), 0);
167 assert.equal(failuresInARow([{ status: "failed" }, { status: "failed" }, { status: "succeeded" }, { status: "failed" }]), 2);
168 assert.equal(failuresInARow([{ status: "stopped", halted: "time" }, { status: "failed" }]), 2);
169 assert.equal(failuresInARow([{ status: "stopped", halted: null }, { status: "failed" }]), 0);
170 assert.equal(failuresInARow([{ status: "failed", title: "Add search" }, { status: "failed", title: "Other" }], "Add search"), 1);
171 assert.equal(leftLowConfidence([{ status: "succeeded", confidence: { level: "low" } }]), true);
172 assert.equal(leftLowConfidence([{ status: "succeeded", confidence: { level: "medium" } }]), false);
173 assert.equal(leftLowConfidence([]), false);
174});
175
176test("a retry is the same work again after its latest attempt failed", () => {
177 assert.equal(lastAttemptFailed([]), false);
178 assert.equal(lastAttemptFailed([{ status: "failed" }, { status: "succeeded" }]), true);
179 assert.equal(lastAttemptFailed([{ status: "succeeded" }, { status: "failed" }]), false);
180 assert.equal(lastAttemptFailed([{ status: "stopped", halted: "budget" }]), true);
181 assert.equal(lastAttemptFailed([{ status: "stopped", halted: null }]), false);
182 assert.equal(lastAttemptFailed([{ status: "running" }]), false);
183 assert.equal(lastAttemptFailed([{ status: "failed", title: "Add search" }], "Add search "), true);
184 assert.equal(lastAttemptFailed([{ status: "failed", title: "Add search" }], "Add billing"), false);
185});
186
187test("the configuration decides the catalogue, the rules and the limits", () => {
188 const parsed = parseRouting(
189 JSON.stringify({
190 tiers: { small: { modelName: "Small", model: "small-1" }, frontier: { model: "" } },
191 tasks: { update: "large", plan: "huge", answer: "change" },
192 smallChange: { lines: 50 },
193 frontierLabels: ["hard"],
194 learning: { minRuns: 3 },
195 }),
196 );
197 assert.deepEqual(parsed.tiers.small, { modelName: "Small", model: "small-1" });
198 assert.deepEqual(parsed.tiers.large, DEFAULT_ROUTING.tiers.large);
199 // A tier with no model keeps the default.
200 assert.deepEqual(parsed.tiers.frontier, DEFAULT_ROUTING.tiers.frontier);
201 assert.equal(chooseTier("update", {}, parsed), "large");
202 // A rule that names no tier keeps the default.
203 assert.equal(chooseTier("plan", {}, parsed), "small");
204 assert.equal(parsed.tasks.answer, "change");
205 assert.equal(chooseTier("review", { change: { ...small, lines: 51 } }, parsed), "large");
206 assert.equal(chooseTier("review", { change: { ...small, files: 10, lines: 50 } }, parsed), "small");
207 assert.equal(chooseTier("implement", { labels: ["hard"] }, parsed), "frontier");
208 assert.equal(parsed.learning.minRuns, 3);
209 assert.equal(parsed.learning.window, DEFAULT_ROUTING.learning.window);
210 assert.deepEqual(parseRouting(undefined), DEFAULT_ROUTING);
211 assert.deepEqual(parseRouting("not json"), DEFAULT_ROUTING);
212 assert.deepEqual(parseRouting("[1]"), DEFAULT_ROUTING);
213});
214
215test("the routing the runner ships with is the default, written as configuration", async () => {
216 const { readFile } = await import("node:fs/promises");
217 const config = await readFile(new URL("../wrangler.jsonc", import.meta.url), "utf8");
218 const line = config.split("\n").find((l) => l.trim().startsWith('"AGENT_ROUTING"'));
219 assert.ok(line, "wrangler.jsonc sets AGENT_ROUTING");
220 const value = JSON.parse(line.trim().replace(/^"AGENT_ROUTING":\s*/, "").replace(/,$/, ""));
221 assert.deepEqual(parseRouting(value), DEFAULT_ROUTING);
222});
223
224test("a change's size is its files and the lines added and removed", () => {
225 assert.deepEqual(
226 changeSize([{ additions: 10, deletions: 2 }, { additions: 0, deletions: 5 }], ["secrets"]),
227 { files: 2, lines: 17, sensitive: ["secrets"] },
228 );
229});
230
231test("the tier decides the model, and the harness's small tasks use the small tier", () => {
232 const large = modelEnv(direct, routes, "implement", "large", tags);
233 assert.equal(large.ANTHROPIC_MODEL, "claude-sonnet-5-5");
234 assert.equal(large.AGENT_MODEL_NAME, "Claude Sonnet 5.5");
235 assert.equal(large.ANTHROPIC_SMALL_FAST_MODEL, "claude-haiku-4-5-20251001");
236 assert.equal(large.ANTHROPIC_DEFAULT_HAIKU_MODEL, "claude-haiku-4-5-20251001");
237 const review = modelEnv(direct, routes, "review", "small", tags);
238 assert.equal(review.ANTHROPIC_MODEL, "claude-haiku-4-5-20251001");
239 assert.equal(review.AGENT_MODEL_NAME, "Claude Haiku 4.5");
240});
241
242test("without a gateway, requests go to the provider directly", () => {
243 const vars = modelEnv(direct, routes, "implement", "large", tags);
244 assert.equal(vars.ANTHROPIC_BASE_URL, undefined);
245 assert.equal(vars.ANTHROPIC_CUSTOM_HEADERS, undefined);
246 assert.equal(vars.ANTHROPIC_API_KEY, "sk-test");
247});
248
249test("with a gateway, requests go through it and say what they are for", () => {
250 const vars = modelEnv({ ...direct, AI_GATEWAY_ID: "g1t" }, routes, "review", "large", tags);
251 assert.equal(vars.ANTHROPIC_BASE_URL, "https://gateway.ai.cloudflare.com/v1/acct/g1t/anthropic");
252 assert.deepEqual(customHeaders(vars), {
253 "cf-aig-metadata": '{"task":"review","tier":"large","repo":"acme/site","pull":12}',
254 });
255});
256
257test("a run straight to the gateway carries its session, so billing can settle it", () => {
258 const session = gatewaySession();
259 assert.match(session, /^rs_[0-9a-f]{24}$/);
260 assert.notEqual(gatewaySession(), session);
261 const vars = modelEnv({ ...direct, AI_GATEWAY_ID: "g1t" }, routes, "implement", "small", { ...tags, session });
262 const metadata = JSON.parse(customHeaders(vars)["cf-aig-metadata"]);
263 assert.equal(metadata.session, session);
264 // The gateway keeps at most five metadata entries.
265 assert.ok(Object.keys(metadata).length <= 5);
266});
267
268test("an authenticated gateway is sent its token", () => {
269 const vars = modelEnv({ ...direct, AI_GATEWAY_ID: "g1t", AI_GATEWAY_TOKEN: "tok" }, routes, "implement", "large", tags);
270 assert.equal(customHeaders(vars)["cf-aig-authorization"], "Bearer tok");
271 assert.equal(vars.ANTHROPIC_API_KEY, "sk-test");
272});
273
274test("when the gateway holds the provider's key, the sandbox never gets it", () => {
275 const gatewayOnly = { AI_GATEWAY_ID: "g1t", AI_GATEWAY_TOKEN: "tok", CLOUDFLARE_ACCOUNT_ID: "acct" };
276 assert.equal(canReachModel(gatewayOnly), true);
277 assert.equal(modelEnv(gatewayOnly, routes, "implement", "large", tags).ANTHROPIC_API_KEY, "tok");
278 assert.equal(canReachModel({ AI_GATEWAY_ID: "g1t", CLOUDFLARE_ACCOUNT_ID: "acct" }), false);
279 assert.equal(canReachModel(direct), true);
280});
281
282// A stand-in for the gateway: what a sandbox's agent sends, given these
283// variables, arrives at the gateway's path with both credentials.
284test("a request built from these variables reaches a gateway as expected", async () => {
285 const seen: { url?: string; key?: string; gateway?: string; metadata?: string; model?: string } = {};
286 const server = createServer((request, response) => {
287 let body = "";
288 request.on("data", (chunk) => (body += chunk));
289 request.on("end", () => {
290 seen.url = request.url;
291 seen.key = request.headers["x-api-key"] as string;
292 seen.gateway = request.headers["cf-aig-authorization"] as string;
293 seen.metadata = request.headers["cf-aig-metadata"] as string;
294 seen.model = JSON.parse(body).model;
295 response.setHeader("content-type", "application/json");
296 response.end(JSON.stringify({ type: "message", content: [] }));
297 });
298 });
299 await new Promise<void>((resolve) => server.listen(0, resolve));
300 const { port } = server.address() as { port: number };
301
302 const vars = modelEnv({ ...direct, AI_GATEWAY_ID: "g1t", AI_GATEWAY_TOKEN: "tok" }, routes, "implement", "large", tags);
303 // Same path as the real gateway, on the stand-in's address.
304 const base = vars.ANTHROPIC_BASE_URL.replace("https://gateway.ai.cloudflare.com", `http://localhost:${port}`);
305 await fetch(`${base}/v1/messages`, {
306 method: "POST",
307 headers: { "x-api-key": vars.ANTHROPIC_API_KEY, ...customHeaders(vars), "content-type": "application/json" },
308 body: JSON.stringify({ model: vars.ANTHROPIC_MODEL, max_tokens: 1, messages: [] }),
309 });
310 server.close();
311
312 assert.deepEqual(seen, {
313 url: "/v1/acct/g1t/anthropic/v1/messages",
314 key: "sk-test",
315 gateway: "Bearer tok",
316 metadata: '{"task":"implement","tier":"large","repo":"acme/site","pull":12}',
317 model: "claude-sonnet-5-5",
318 });
319});
320
321test("Each kind of job has its effort: plans think hard, answers less, catching up least", () => {
322 assert.deepEqual(DEFAULT_ROUTING.effort, { plan: "high", answer: "medium", update: "low" });
323 const parsed = parseRouting(JSON.stringify({ effort: { implement: "xhigh", plan: "enormous", nonsense: "low" } }));
324 assert.equal(parsed.effort.implement, "xhigh");
325 assert.equal(parsed.effort.plan, "high");
326 assert.equal((parsed.effort as Record<string, string>).nonsense, undefined);
327});