Skip to content
317 linesCodeBlameRaw
1import assert from "node:assert/strict";
2import { createServer } from "node:http";
3import { test } from "node:test";
4
5import {
6 type AgentRouting,
7 type ChangeSize,
8 DEFAULT_ROUTING,
9 canReachModel,
10 changeSize,
11 chooseTier,
12 failuresInARow,
13 gatewaySession,
14 lastAttemptFailed,
15 leftLowConfidence,
16 modelEnv,
17 outcomesOf,
18 parseRouting,
19 route,
20 taskOf,
21 tierOfModel,
22} from "./model-env.ts";
23
24const routes: AgentRouting = {
25 ...DEFAULT_ROUTING,
26 tiers: {
27 small: { modelName: "Claude Haiku 4.5", model: "claude-haiku-4-5-20251001" },
28 large: { modelName: "Claude Sonnet 5.5", model: "claude-sonnet-5-5" },
29 frontier: { modelName: "Claude Opus 5.5", model: "claude-opus-5-5" },
30 },
31};
32const tags = { repo: "acme/site", pull: 12 };
33const direct = { ANTHROPIC_API_KEY: "sk-test", AI_GATEWAY_ID: "", CLOUDFLARE_ACCOUNT_ID: "acct" };
34
35/** `ANTHROPIC_CUSTOM_HEADERS` as the harness reads it: one header per line. */
36function customHeaders(vars: Record<string, string>): Record<string, string> {
37 return Object.fromEntries(
38 vars.ANTHROPIC_CUSTOM_HEADERS.split("\n").map((line) => {
39 const at = line.indexOf(": ");
40 return [line.slice(0, at), line.slice(at + 2)];
41 }),
42 );
43}
44
45const small: ChangeSize = { files: 3, lines: 80, sensitive: [] };
46
47test("Auto starts each kind of job on its tier: fast for catching up and answering, standard for changes, most capable for plans", () => {
48 assert.equal(chooseTier("update", {}, routes), "small");
49 assert.equal(chooseTier("answer", {}, routes), "small");
50 assert.equal(chooseTier("implement", {}, routes), "large");
51 assert.equal(chooseTier("revise", {}, routes), "large");
52 assert.equal(chooseTier("implement", { change: small }, routes), "large");
53 assert.equal(chooseTier("plan", {}, routes), "frontier");
54});
55
56test("revising and answering go to the workspace's implement route and bill", () => {
57 assert.equal(taskOf("revise"), "implement");
58 assert.equal(taskOf("answer"), "implement");
59 assert.equal(taskOf("review"), "review");
60 assert.equal(taskOf("plan"), "plan");
61});
62
63test("a review is sized by its change: fast when small and safe, most capable when large", () => {
64 assert.equal(chooseTier("review", { change: small }, routes), "small");
65 assert.equal(chooseTier("review", { change: { ...small, lines: 200, files: 10 } }, routes), "small");
66 assert.equal(chooseTier("review", { change: { ...small, lines: 201 } }, routes), "large");
67 assert.equal(chooseTier("review", { change: { ...small, files: 11 } }, routes), "large");
68 assert.equal(chooseTier("review", { change: { ...small, sensitive: ["CI workflows"] } }, routes), "large");
69 assert.equal(chooseTier("review", { change: small, labels: ["Security"] }, routes), "large");
70 assert.equal(chooseTier("review", { change: small, labels: ["docs"] }, routes), "small");
71 assert.equal(chooseTier("review", { change: { ...small, files: 61 } }, routes), "frontier");
72 assert.equal(chooseTier("review", { change: { ...small, lines: 3001 } }, routes), "frontier");
73});
74
75test("a review of a change g1t cannot size runs on the standard tier", () => {
76 assert.equal(chooseTier("review", {}, routes), "large");
77 assert.equal(chooseTier("review", { change: null }, routes), "large");
78 assert.equal(chooseTier("review", { change: { files: 0, lines: 0, sensitive: [] } }, routes), "large");
79});
80
81test("labels move work: architecture to the most capable, documentation to the fast tier", () => {
82 assert.equal(chooseTier("implement", { labels: ["Architecture"] }, routes), "frontier");
83 assert.equal(chooseTier("review", { change: small, labels: ["architecture"] }, routes), "frontier");
84 assert.equal(chooseTier("implement", { labels: ["docs"] }, routes), "small");
85 assert.equal(chooseTier("answer", { labels: ["typo"] }, routes), "small");
86 // A small label never takes a review or a plan down.
87 assert.equal(chooseTier("plan", { labels: ["docs"] }, routes), "frontier");
88 // Security outranks documentation.
89 assert.equal(chooseTier("implement", { labels: ["docs", "security"] }, routes), "large");
90});
91
92test("a failed attempt goes one tier up, and repeated failures to the most capable", () => {
93 assert.equal(chooseTier("update", { retry: true }, routes), "large");
94 assert.equal(chooseTier("review", { change: small, retry: true }, routes), "large");
95 assert.equal(chooseTier("implement", { failures: 1 }, routes), "frontier");
96 assert.equal(chooseTier("update", { failures: 2 }, routes), "frontier");
97 assert.equal(chooseTier("plan", { failures: 1 }, routes), "frontier");
98 const why = route("update", { failures: 2 }, routes).reason;
99 assert.equal(why, "Used the most capable model (Claude Opus 5.5): the last 2 attempts at this work failed.");
100});
101
102test("a change left at low confidence sends the next attempt one tier up", () => {
103 assert.equal(chooseTier("revise", { lowConfidence: true }, routes), "frontier");
104 assert.equal(chooseTier("update", { lowConfidence: true }, routes), "large");
105});
106
107test("a tier the workspace chose wins over Auto", () => {
108 const chosen = route("review", { change: small, failures: 3, chosen: "large" }, routes);
109 assert.equal(chosen.tier, "large");
110 assert.equal(chosen.reason, "Used the standard model (Claude Sonnet 5.5): the workspace chose the standard model for this work.");
111});
112
113const ok = (tier: "small" | "large" | "frontier", n: number) => Array.from({ length: n }, () => ({ tier, ok: true }));
114const bad = (tier: "small" | "large" | "frontier", n: number) => Array.from({ length: n }, () => ({ tier, ok: false }));
115
116test("a repository whose fast runs finish nearly always steps the work down", () => {
117 const history = [...ok("small", 9), ...bad("small", 1)];
118 const routed = route("implement", { history }, routes);
119 assert.equal(routed.tier, "small");
120 assert.equal(routed.reason, "Used a fast model (Claude Haiku 4.5): it finished 9 of its last 10 runs like this here.");
121 // Too few runs to tell, or not quite enough of them finished: no change.
122 assert.equal(chooseTier("implement", { history: ok("small", 4) }, routes), "large");
123 assert.equal(chooseTier("implement", { history: [...ok("small", 8), ...bad("small", 2)] }, routes), "large");
124 // Plans step down from the most capable model the same way.
125 assert.equal(chooseTier("plan", { history: ok("large", 6) }, routes), "large");
126});
127
128test("learning never steps down sensitive or labelled work, nor a retry", () => {
129 const history = ok("small", 10);
130 assert.equal(chooseTier("review", { change: { ...small, files: 20, sensitive: ["secrets"] }, history }, routes), "large");
131 assert.equal(chooseTier("implement", { labels: ["security"], history }, routes), "large");
132 assert.equal(chooseTier("implement", { failures: 1, history }, routes), "frontier");
133});
134
135test("a tier that fails half its runs in a repository hands the work up", () => {
136 const history = [...bad("large", 3), ...ok("large", 2)];
137 const routed = route("implement", { history }, routes);
138 assert.equal(routed.tier, "frontier");
139 assert.equal(routed.reason, "Used the most capable model (Claude Opus 5.5): the standard model failed 3 of its last 5 runs like this here.");
140});
141
142test("past runs are read by the tier they ran on and whether they did the work", () => {
143 const runs = [
144 { status: "running", model: "Claude Haiku 4.5" },
145 { status: "succeeded", model: "Claude Haiku 4.5" },
146 { status: "succeeded", model: "Claude Sonnet 5.5", confidence: { level: "low" } },
147 { status: "failed", model: "Claude Opus 5.5" },
148 { status: "stopped", halted: null, model: "Claude Haiku 4.5" },
149 { status: "stopped", halted: "budget", model: "claude-haiku-4-5-20251001" },
150 { status: "succeeded", model: "Claude Sonnet 5.5, through Acme" },
151 ];
152 assert.deepEqual(outcomesOf(runs, routes), [
153 { tier: "small", ok: true },
154 { tier: "large", ok: false },
155 { tier: "frontier", ok: false },
156 { tier: "small", ok: false },
157 { tier: null, ok: true },
158 ]);
159 assert.equal(tierOfModel("Claude Opus 5.5", routes), "frontier");
160 assert.equal(tierOfModel(null, routes), null);
161});
162
163test("failures are counted in a row, newest first, and confidence is read from the last finished one", () => {
164 assert.equal(failuresInARow([]), 0);
165 assert.equal(failuresInARow([{ status: "failed" }, { status: "failed" }, { status: "succeeded" }, { status: "failed" }]), 2);
166 assert.equal(failuresInARow([{ status: "stopped", halted: "time" }, { status: "failed" }]), 2);
167 assert.equal(failuresInARow([{ status: "stopped", halted: null }, { status: "failed" }]), 0);
168 assert.equal(failuresInARow([{ status: "failed", title: "Add search" }, { status: "failed", title: "Other" }], "Add search"), 1);
169 assert.equal(leftLowConfidence([{ status: "succeeded", confidence: { level: "low" } }]), true);
170 assert.equal(leftLowConfidence([{ status: "succeeded", confidence: { level: "medium" } }]), false);
171 assert.equal(leftLowConfidence([]), false);
172});
173
174test("a retry is the same work again after its latest attempt failed", () => {
175 assert.equal(lastAttemptFailed([]), false);
176 assert.equal(lastAttemptFailed([{ status: "failed" }, { status: "succeeded" }]), true);
177 assert.equal(lastAttemptFailed([{ status: "succeeded" }, { status: "failed" }]), false);
178 assert.equal(lastAttemptFailed([{ status: "stopped", halted: "budget" }]), true);
179 assert.equal(lastAttemptFailed([{ status: "stopped", halted: null }]), false);
180 assert.equal(lastAttemptFailed([{ status: "running" }]), false);
181 assert.equal(lastAttemptFailed([{ status: "failed", title: "Add search" }], "Add search "), true);
182 assert.equal(lastAttemptFailed([{ status: "failed", title: "Add search" }], "Add billing"), false);
183});
184
185test("the configuration decides the catalogue, the rules and the limits", () => {
186 const parsed = parseRouting(
187 JSON.stringify({
188 tiers: { small: { modelName: "Small", model: "small-1" }, frontier: { model: "" } },
189 tasks: { update: "large", plan: "huge", answer: "change" },
190 smallChange: { lines: 50 },
191 frontierLabels: ["hard"],
192 learning: { minRuns: 3 },
193 }),
194 );
195 assert.deepEqual(parsed.tiers.small, { modelName: "Small", model: "small-1" });
196 assert.deepEqual(parsed.tiers.large, DEFAULT_ROUTING.tiers.large);
197 // A tier with no model keeps the default.
198 assert.deepEqual(parsed.tiers.frontier, DEFAULT_ROUTING.tiers.frontier);
199 assert.equal(chooseTier("update", {}, parsed), "large");
200 // A rule that names no tier keeps the default.
201 assert.equal(chooseTier("plan", {}, parsed), "frontier");
202 assert.equal(parsed.tasks.answer, "change");
203 assert.equal(chooseTier("review", { change: { ...small, lines: 51 } }, parsed), "large");
204 assert.equal(chooseTier("review", { change: { ...small, files: 10, lines: 50 } }, parsed), "small");
205 assert.equal(chooseTier("implement", { labels: ["hard"] }, parsed), "frontier");
206 assert.equal(parsed.learning.minRuns, 3);
207 assert.equal(parsed.learning.window, DEFAULT_ROUTING.learning.window);
208 assert.deepEqual(parseRouting(undefined), DEFAULT_ROUTING);
209 assert.deepEqual(parseRouting("not json"), DEFAULT_ROUTING);
210 assert.deepEqual(parseRouting("[1]"), DEFAULT_ROUTING);
211});
212
213test("the routing the runner ships with is the default, written as configuration", async () => {
214 const { readFile } = await import("node:fs/promises");
215 const config = await readFile(new URL("../wrangler.jsonc", import.meta.url), "utf8");
216 const line = config.split("\n").find((l) => l.trim().startsWith('"AGENT_ROUTING"'));
217 assert.ok(line, "wrangler.jsonc sets AGENT_ROUTING");
218 const value = JSON.parse(line.trim().replace(/^"AGENT_ROUTING":\s*/, "").replace(/,$/, ""));
219 assert.deepEqual(parseRouting(value), DEFAULT_ROUTING);
220});
221
222test("a change's size is its files and the lines added and removed", () => {
223 assert.deepEqual(
224 changeSize([{ additions: 10, deletions: 2 }, { additions: 0, deletions: 5 }], ["secrets"]),
225 { files: 2, lines: 17, sensitive: ["secrets"] },
226 );
227});
228
229test("the tier decides the model, and the harness's small tasks use the small tier", () => {
230 const large = modelEnv(direct, routes, "implement", "large", tags);
231 assert.equal(large.ANTHROPIC_MODEL, "claude-sonnet-5-5");
232 assert.equal(large.AGENT_MODEL_NAME, "Claude Sonnet 5.5");
233 assert.equal(large.ANTHROPIC_SMALL_FAST_MODEL, "claude-haiku-4-5-20251001");
234 assert.equal(large.ANTHROPIC_DEFAULT_HAIKU_MODEL, "claude-haiku-4-5-20251001");
235 const review = modelEnv(direct, routes, "review", "small", tags);
236 assert.equal(review.ANTHROPIC_MODEL, "claude-haiku-4-5-20251001");
237 assert.equal(review.AGENT_MODEL_NAME, "Claude Haiku 4.5");
238});
239
240test("without a gateway, requests go to the provider directly", () => {
241 const vars = modelEnv(direct, routes, "implement", "large", tags);
242 assert.equal(vars.ANTHROPIC_BASE_URL, undefined);
243 assert.equal(vars.ANTHROPIC_CUSTOM_HEADERS, undefined);
244 assert.equal(vars.ANTHROPIC_API_KEY, "sk-test");
245});
246
247test("with a gateway, requests go through it and say what they are for", () => {
248 const vars = modelEnv({ ...direct, AI_GATEWAY_ID: "g1t" }, routes, "review", "large", tags);
249 assert.equal(vars.ANTHROPIC_BASE_URL, "https://gateway.ai.cloudflare.com/v1/acct/g1t/anthropic");
250 assert.deepEqual(customHeaders(vars), {
251 "cf-aig-metadata": '{"task":"review","tier":"large","repo":"acme/site","pull":12}',
252 });
253});
254
255test("a run straight to the gateway carries its session, so billing can settle it", () => {
256 const session = gatewaySession();
257 assert.match(session, /^rs_[0-9a-f]{24}$/);
258 assert.notEqual(gatewaySession(), session);
259 const vars = modelEnv({ ...direct, AI_GATEWAY_ID: "g1t" }, routes, "implement", "small", { ...tags, session });
260 const metadata = JSON.parse(customHeaders(vars)["cf-aig-metadata"]);
261 assert.equal(metadata.session, session);
262 // The gateway keeps at most five metadata entries.
263 assert.ok(Object.keys(metadata).length <= 5);
264});
265
266test("an authenticated gateway is sent its token", () => {
267 const vars = modelEnv({ ...direct, AI_GATEWAY_ID: "g1t", AI_GATEWAY_TOKEN: "tok" }, routes, "implement", "large", tags);
268 assert.equal(customHeaders(vars)["cf-aig-authorization"], "Bearer tok");
269 assert.equal(vars.ANTHROPIC_API_KEY, "sk-test");
270});
271
272test("when the gateway holds the provider's key, the sandbox never gets it", () => {
273 const gatewayOnly = { AI_GATEWAY_ID: "g1t", AI_GATEWAY_TOKEN: "tok", CLOUDFLARE_ACCOUNT_ID: "acct" };
274 assert.equal(canReachModel(gatewayOnly), true);
275 assert.equal(modelEnv(gatewayOnly, routes, "implement", "large", tags).ANTHROPIC_API_KEY, "tok");
276 assert.equal(canReachModel({ AI_GATEWAY_ID: "g1t", CLOUDFLARE_ACCOUNT_ID: "acct" }), false);
277 assert.equal(canReachModel(direct), true);
278});
279
280// A stand-in for the gateway: what a sandbox's agent sends, given these
281// variables, arrives at the gateway's path with both credentials.
282test("a request built from these variables reaches a gateway as expected", async () => {
283 const seen: { url?: string; key?: string; gateway?: string; metadata?: string; model?: string } = {};
284 const server = createServer((request, response) => {
285 let body = "";
286 request.on("data", (chunk) => (body += chunk));
287 request.on("end", () => {
288 seen.url = request.url;
289 seen.key = request.headers["x-api-key"] as string;
290 seen.gateway = request.headers["cf-aig-authorization"] as string;
291 seen.metadata = request.headers["cf-aig-metadata"] as string;
292 seen.model = JSON.parse(body).model;
293 response.setHeader("content-type", "application/json");
294 response.end(JSON.stringify({ type: "message", content: [] }));
295 });
296 });
297 await new Promise<void>((resolve) => server.listen(0, resolve));
298 const { port } = server.address() as { port: number };
299
300 const vars = modelEnv({ ...direct, AI_GATEWAY_ID: "g1t", AI_GATEWAY_TOKEN: "tok" }, routes, "implement", "large", tags);
301 // Same path as the real gateway, on the stand-in's address.
302 const base = vars.ANTHROPIC_BASE_URL.replace("https://gateway.ai.cloudflare.com", `http://localhost:${port}`);
303 await fetch(`${base}/v1/messages`, {
304 method: "POST",
305 headers: { "x-api-key": vars.ANTHROPIC_API_KEY, ...customHeaders(vars), "content-type": "application/json" },
306 body: JSON.stringify({ model: vars.ANTHROPIC_MODEL, max_tokens: 1, messages: [] }),
307 });
308 server.close();
309
310 assert.deepEqual(seen, {
311 url: "/v1/acct/g1t/anthropic/v1/messages",
312 key: "sk-test",
313 gateway: "Bearer tok",
314 metadata: '{"task":"implement","tier":"large","repo":"acme/site","pull":12}',
315 model: "claude-sonnet-5-5",
316 });
317});