Skip to content

g1t/services/runner/src/model-env.test.ts

318 lines16,564 bytesCodeBlame
1import assert from "node:assert/strict";
2import { createServer } from "node:http";
3import { test } from "node:test";
4
5import {
6 type AgentRouting,
7 type ChangeSize,
8 DEFAULT_ROUTING,
9 canReachModel,
10 changeSize,
11 chooseTier,
12 failuresInARow,
13 gatewaySession,
14 lastAttemptFailed,
15 leftLowConfidence,
16 modelEnv,
17 outcomesOf,
18 parseRouting,
19 route,
20 taskOf,
21 tierOfModel,
22} from "./model-env.ts";
23
24const routes: AgentRouting = {
25 ...DEFAULT_ROUTING,
26 tiers: {
27 small: { modelName: "Claude Haiku 4.5", model: "claude-haiku-4-5-20251001" },
28 large: { modelName: "Claude Sonnet 5.5", model: "claude-sonnet-5-5" },
29 frontier: { modelName: "Claude Opus 5.5", model: "claude-opus-5-5" },
30 },
31};
32const tags = { repo: "acme/site", pull: 12 };
33const direct = { ANTHROPIC_API_KEY: "sk-test", AI_GATEWAY_ID: "", CLOUDFLARE_ACCOUNT_ID: "acct" };
34
35/** `ANTHROPIC_CUSTOM_HEADERS` as the harness reads it: one header per line. */
36function customHeaders(vars: Record<string, string>): Record<string, string> {
37 return Object.fromEntries(
38 vars.ANTHROPIC_CUSTOM_HEADERS.split("\n").map((line) => {
39 const at = line.indexOf(": ");
40 return [line.slice(0, at), line.slice(at + 2)];
41 }),
42 );
43}
44
45const small: ChangeSize = { files: 3, lines: 80, sensitive: [] };
46
47test("Auto starts each kind of job on its tier: fast for catching up and answering, standard for changes and plans", () => {
48 assert.equal(chooseTier("update", {}, routes), "small");
49 assert.equal(chooseTier("answer", {}, routes), "small");
50 assert.equal(chooseTier("implement", {}, routes), "large");
51 assert.equal(chooseTier("revise", {}, routes), "large");
52 assert.equal(chooseTier("implement", { change: small }, routes), "large");
53 assert.equal(chooseTier("plan", {}, routes), "large");
54 assert.equal(chooseTier("plan", { labels: ["architecture"] }, routes), "frontier");
55});
56
57test("revising and answering go to the workspace's implement route and bill", () => {
58 assert.equal(taskOf("revise"), "implement");
59 assert.equal(taskOf("answer"), "implement");
60 assert.equal(taskOf("review"), "review");
61 assert.equal(taskOf("plan"), "plan");
62});
63
64test("a review is sized by its change: fast when small and safe, most capable when large", () => {
65 assert.equal(chooseTier("review", { change: small }, routes), "small");
66 assert.equal(chooseTier("review", { change: { ...small, lines: 200, files: 10 } }, routes), "small");
67 assert.equal(chooseTier("review", { change: { ...small, lines: 201 } }, routes), "large");
68 assert.equal(chooseTier("review", { change: { ...small, files: 11 } }, routes), "large");
69 assert.equal(chooseTier("review", { change: { ...small, sensitive: ["CI workflows"] } }, routes), "large");
70 assert.equal(chooseTier("review", { change: small, labels: ["Security"] }, routes), "large");
71 assert.equal(chooseTier("review", { change: small, labels: ["docs"] }, routes), "small");
72 assert.equal(chooseTier("review", { change: { ...small, files: 61 } }, routes), "frontier");
73 assert.equal(chooseTier("review", { change: { ...small, lines: 3001 } }, routes), "frontier");
74});
75
76test("a review of a change g1t cannot size runs on the standard tier", () => {
77 assert.equal(chooseTier("review", {}, routes), "large");
78 assert.equal(chooseTier("review", { change: null }, routes), "large");
79 assert.equal(chooseTier("review", { change: { files: 0, lines: 0, sensitive: [] } }, routes), "large");
80});
81
82test("labels move work: architecture to the most capable, documentation to the fast tier", () => {
83 assert.equal(chooseTier("implement", { labels: ["Architecture"] }, routes), "frontier");
84 assert.equal(chooseTier("review", { change: small, labels: ["architecture"] }, routes), "frontier");
85 assert.equal(chooseTier("implement", { labels: ["docs"] }, routes), "small");
86 assert.equal(chooseTier("answer", { labels: ["typo"] }, routes), "small");
87 // A small label never takes a review or a plan down.
88 assert.equal(chooseTier("plan", { labels: ["docs"] }, routes), "large");
89 // Security outranks documentation.
90 assert.equal(chooseTier("implement", { labels: ["docs", "security"] }, routes), "large");
91});
92
93test("a failed attempt goes one tier up, and repeated failures to the most capable", () => {
94 assert.equal(chooseTier("update", { retry: true }, routes), "large");
95 assert.equal(chooseTier("review", { change: small, retry: true }, routes), "large");
96 assert.equal(chooseTier("implement", { failures: 1 }, routes), "frontier");
97 assert.equal(chooseTier("update", { failures: 2 }, routes), "frontier");
98 assert.equal(chooseTier("plan", { failures: 1 }, routes), "frontier");
99 const why = route("update", { failures: 2 }, routes).reason;
100 assert.equal(why, "Used the most capable model (Claude Opus 5.5): the last 2 attempts at this work failed.");
101});
102
103test("a change left at low confidence sends the next attempt one tier up", () => {
104 assert.equal(chooseTier("revise", { lowConfidence: true }, routes), "frontier");
105 assert.equal(chooseTier("update", { lowConfidence: true }, routes), "large");
106});
107
108test("a tier the workspace chose wins over Auto", () => {
109 const chosen = route("review", { change: small, failures: 3, chosen: "large" }, routes);
110 assert.equal(chosen.tier, "large");
111 assert.equal(chosen.reason, "Used the standard model (Claude Sonnet 5.5): the workspace chose the standard model for this work.");
112});
113
114const ok = (tier: "small" | "large" | "frontier", n: number) => Array.from({ length: n }, () => ({ tier, ok: true }));
115const bad = (tier: "small" | "large" | "frontier", n: number) => Array.from({ length: n }, () => ({ tier, ok: false }));
116
117test("a repository whose fast runs finish nearly always steps the work down", () => {
118 const history = [...ok("small", 9), ...bad("small", 1)];
119 const routed = route("implement", { history }, routes);
120 assert.equal(routed.tier, "small");
121 assert.equal(routed.reason, "Used a fast model (Claude Haiku 4.5): it finished 9 of its last 10 runs like this here.");
122 // Too few runs to tell, or not quite enough of them finished: no change.
123 assert.equal(chooseTier("implement", { history: ok("small", 4) }, routes), "large");
124 assert.equal(chooseTier("implement", { history: [...ok("small", 8), ...bad("small", 2)] }, routes), "large");
125 // Plans step down the same way.
126 assert.equal(chooseTier("plan", { history: ok("small", 6) }, routes), "small");
127});
128
129test("learning never steps down sensitive or labelled work, nor a retry", () => {
130 const history = ok("small", 10);
131 assert.equal(chooseTier("review", { change: { ...small, files: 20, sensitive: ["secrets"] }, history }, routes), "large");
132 assert.equal(chooseTier("implement", { labels: ["security"], history }, routes), "large");
133 assert.equal(chooseTier("implement", { failures: 1, history }, routes), "frontier");
134});
135
136test("a tier that fails half its runs in a repository hands the work up", () => {
137 const history = [...bad("large", 3), ...ok("large", 2)];
138 const routed = route("implement", { history }, routes);
139 assert.equal(routed.tier, "frontier");
140 assert.equal(routed.reason, "Used the most capable model (Claude Opus 5.5): the standard model failed 3 of its last 5 runs like this here.");
141});
142
143test("past runs are read by the tier they ran on and whether they did the work", () => {
144 const runs = [
145 { status: "running", model: "Claude Haiku 4.5" },
146 { status: "succeeded", model: "Claude Haiku 4.5" },
147 { status: "succeeded", model: "Claude Sonnet 5.5", confidence: { level: "low" } },
148 { status: "failed", model: "Claude Opus 5.5" },
149 { status: "stopped", halted: null, model: "Claude Haiku 4.5" },
150 { status: "stopped", halted: "budget", model: "claude-haiku-4-5-20251001" },
151 { status: "succeeded", model: "Claude Sonnet 5.5, through Acme" },
152 ];
153 assert.deepEqual(outcomesOf(runs, routes), [
154 { tier: "small", ok: true },
155 { tier: "large", ok: false },
156 { tier: "frontier", ok: false },
157 { tier: "small", ok: false },
158 { tier: null, ok: true },
159 ]);
160 assert.equal(tierOfModel("Claude Opus 5.5", routes), "frontier");
161 assert.equal(tierOfModel(null, routes), null);
162});
163
164test("failures are counted in a row, newest first, and confidence is read from the last finished one", () => {
165 assert.equal(failuresInARow([]), 0);
166 assert.equal(failuresInARow([{ status: "failed" }, { status: "failed" }, { status: "succeeded" }, { status: "failed" }]), 2);
167 assert.equal(failuresInARow([{ status: "stopped", halted: "time" }, { status: "failed" }]), 2);
168 assert.equal(failuresInARow([{ status: "stopped", halted: null }, { status: "failed" }]), 0);
169 assert.equal(failuresInARow([{ status: "failed", title: "Add search" }, { status: "failed", title: "Other" }], "Add search"), 1);
170 assert.equal(leftLowConfidence([{ status: "succeeded", confidence: { level: "low" } }]), true);
171 assert.equal(leftLowConfidence([{ status: "succeeded", confidence: { level: "medium" } }]), false);
172 assert.equal(leftLowConfidence([]), false);
173});
174
175test("a retry is the same work again after its latest attempt failed", () => {
176 assert.equal(lastAttemptFailed([]), false);
177 assert.equal(lastAttemptFailed([{ status: "failed" }, { status: "succeeded" }]), true);
178 assert.equal(lastAttemptFailed([{ status: "succeeded" }, { status: "failed" }]), false);
179 assert.equal(lastAttemptFailed([{ status: "stopped", halted: "budget" }]), true);
180 assert.equal(lastAttemptFailed([{ status: "stopped", halted: null }]), false);
181 assert.equal(lastAttemptFailed([{ status: "running" }]), false);
182 assert.equal(lastAttemptFailed([{ status: "failed", title: "Add search" }], "Add search "), true);
183 assert.equal(lastAttemptFailed([{ status: "failed", title: "Add search" }], "Add billing"), false);
184});
185
186test("the configuration decides the catalogue, the rules and the limits", () => {
187 const parsed = parseRouting(
188 JSON.stringify({
189 tiers: { small: { modelName: "Small", model: "small-1" }, frontier: { model: "" } },
190 tasks: { update: "large", plan: "huge", answer: "change" },
191 smallChange: { lines: 50 },
192 frontierLabels: ["hard"],
193 learning: { minRuns: 3 },
194 }),
195 );
196 assert.deepEqual(parsed.tiers.small, { modelName: "Small", model: "small-1" });
197 assert.deepEqual(parsed.tiers.large, DEFAULT_ROUTING.tiers.large);
198 // A tier with no model keeps the default.
199 assert.deepEqual(parsed.tiers.frontier, DEFAULT_ROUTING.tiers.frontier);
200 assert.equal(chooseTier("update", {}, parsed), "large");
201 // A rule that names no tier keeps the default.
202 assert.equal(chooseTier("plan", {}, parsed), "large");
203 assert.equal(parsed.tasks.answer, "change");
204 assert.equal(chooseTier("review", { change: { ...small, lines: 51 } }, parsed), "large");
205 assert.equal(chooseTier("review", { change: { ...small, files: 10, lines: 50 } }, parsed), "small");
206 assert.equal(chooseTier("implement", { labels: ["hard"] }, parsed), "frontier");
207 assert.equal(parsed.learning.minRuns, 3);
208 assert.equal(parsed.learning.window, DEFAULT_ROUTING.learning.window);
209 assert.deepEqual(parseRouting(undefined), DEFAULT_ROUTING);
210 assert.deepEqual(parseRouting("not json"), DEFAULT_ROUTING);
211 assert.deepEqual(parseRouting("[1]"), DEFAULT_ROUTING);
212});
213
214test("the routing the runner ships with is the default, written as configuration", async () => {
215 const { readFile } = await import("node:fs/promises");
216 const config = await readFile(new URL("../wrangler.jsonc", import.meta.url), "utf8");
217 const line = config.split("\n").find((l) => l.trim().startsWith('"AGENT_ROUTING"'));
218 assert.ok(line, "wrangler.jsonc sets AGENT_ROUTING");
219 const value = JSON.parse(line.trim().replace(/^"AGENT_ROUTING":\s*/, "").replace(/,$/, ""));
220 assert.deepEqual(parseRouting(value), DEFAULT_ROUTING);
221});
222
223test("a change's size is its files and the lines added and removed", () => {
224 assert.deepEqual(
225 changeSize([{ additions: 10, deletions: 2 }, { additions: 0, deletions: 5 }], ["secrets"]),
226 { files: 2, lines: 17, sensitive: ["secrets"] },
227 );
228});
229
230test("the tier decides the model, and the harness's small tasks use the small tier", () => {
231 const large = modelEnv(direct, routes, "implement", "large", tags);
232 assert.equal(large.ANTHROPIC_MODEL, "claude-sonnet-5-5");
233 assert.equal(large.AGENT_MODEL_NAME, "Claude Sonnet 5.5");
234 assert.equal(large.ANTHROPIC_SMALL_FAST_MODEL, "claude-haiku-4-5-20251001");
235 assert.equal(large.ANTHROPIC_DEFAULT_HAIKU_MODEL, "claude-haiku-4-5-20251001");
236 const review = modelEnv(direct, routes, "review", "small", tags);
237 assert.equal(review.ANTHROPIC_MODEL, "claude-haiku-4-5-20251001");
238 assert.equal(review.AGENT_MODEL_NAME, "Claude Haiku 4.5");
239});
240
241test("without a gateway, requests go to the provider directly", () => {
242 const vars = modelEnv(direct, routes, "implement", "large", tags);
243 assert.equal(vars.ANTHROPIC_BASE_URL, undefined);
244 assert.equal(vars.ANTHROPIC_CUSTOM_HEADERS, undefined);
245 assert.equal(vars.ANTHROPIC_API_KEY, "sk-test");
246});
247
248test("with a gateway, requests go through it and say what they are for", () => {
249 const vars = modelEnv({ ...direct, AI_GATEWAY_ID: "g1t" }, routes, "review", "large", tags);
250 assert.equal(vars.ANTHROPIC_BASE_URL, "https://gateway.ai.cloudflare.com/v1/acct/g1t/anthropic");
251 assert.deepEqual(customHeaders(vars), {
252 "cf-aig-metadata": '{"task":"review","tier":"large","repo":"acme/site","pull":12}',
253 });
254});
255
256test("a run straight to the gateway carries its session, so billing can settle it", () => {
257 const session = gatewaySession();
258 assert.match(session, /^rs_[0-9a-f]{24}$/);
259 assert.notEqual(gatewaySession(), session);
260 const vars = modelEnv({ ...direct, AI_GATEWAY_ID: "g1t" }, routes, "implement", "small", { ...tags, session });
261 const metadata = JSON.parse(customHeaders(vars)["cf-aig-metadata"]);
262 assert.equal(metadata.session, session);
263 // The gateway keeps at most five metadata entries.
264 assert.ok(Object.keys(metadata).length <= 5);
265});
266
267test("an authenticated gateway is sent its token", () => {
268 const vars = modelEnv({ ...direct, AI_GATEWAY_ID: "g1t", AI_GATEWAY_TOKEN: "tok" }, routes, "implement", "large", tags);
269 assert.equal(customHeaders(vars)["cf-aig-authorization"], "Bearer tok");
270 assert.equal(vars.ANTHROPIC_API_KEY, "sk-test");
271});
272
273test("when the gateway holds the provider's key, the sandbox never gets it", () => {
274 const gatewayOnly = { AI_GATEWAY_ID: "g1t", AI_GATEWAY_TOKEN: "tok", CLOUDFLARE_ACCOUNT_ID: "acct" };
275 assert.equal(canReachModel(gatewayOnly), true);
276 assert.equal(modelEnv(gatewayOnly, routes, "implement", "large", tags).ANTHROPIC_API_KEY, "tok");
277 assert.equal(canReachModel({ AI_GATEWAY_ID: "g1t", CLOUDFLARE_ACCOUNT_ID: "acct" }), false);
278 assert.equal(canReachModel(direct), true);
279});
280
281// A stand-in for the gateway: what a sandbox's agent sends, given these
282// variables, arrives at the gateway's path with both credentials.
283test("a request built from these variables reaches a gateway as expected", async () => {
284 const seen: { url?: string; key?: string; gateway?: string; metadata?: string; model?: string } = {};
285 const server = createServer((request, response) => {
286 let body = "";
287 request.on("data", (chunk) => (body += chunk));
288 request.on("end", () => {
289 seen.url = request.url;
290 seen.key = request.headers["x-api-key"] as string;
291 seen.gateway = request.headers["cf-aig-authorization"] as string;
292 seen.metadata = request.headers["cf-aig-metadata"] as string;
293 seen.model = JSON.parse(body).model;
294 response.setHeader("content-type", "application/json");
295 response.end(JSON.stringify({ type: "message", content: [] }));
296 });
297 });
298 await new Promise<void>((resolve) => server.listen(0, resolve));
299 const { port } = server.address() as { port: number };
300
301 const vars = modelEnv({ ...direct, AI_GATEWAY_ID: "g1t", AI_GATEWAY_TOKEN: "tok" }, routes, "implement", "large", tags);
302 // Same path as the real gateway, on the stand-in's address.
303 const base = vars.ANTHROPIC_BASE_URL.replace("https://gateway.ai.cloudflare.com", `http://localhost:${port}`);
304 await fetch(`${base}/v1/messages`, {
305 method: "POST",
306 headers: { "x-api-key": vars.ANTHROPIC_API_KEY, ...customHeaders(vars), "content-type": "application/json" },
307 body: JSON.stringify({ model: vars.ANTHROPIC_MODEL, max_tokens: 1, messages: [] }),
308 });
309 server.close();
310
311 assert.deepEqual(seen, {
312 url: "/v1/acct/g1t/anthropic/v1/messages",
313 key: "sk-test",
314 gateway: "Bearer tok",
315 metadata: '{"task":"implement","tier":"large","repo":"acme/site","pull":12}',
316 model: "claude-sonnet-5-5",
317 });
318});