Skip to content

Commit

Ops: estimate what Auto saves, offline; plans start on the standard model

scripts/ops/routing-savings.mjs replays tasks through the router and prices their tokens at each tier's list price from the runner's own configuration: Auto, routing as it was before, and the most capable model for everything, with failed attempts paid for and retried as each policy would, and cost per merged change. It never calls a model. The sample is a handful of tasks shaped like g1t's own work (tokens from the profile in docs/research/reviewbench.md, failures assumed); --live reads billing's settled runs and their counted tokens, read-only. On the sample, plans on the most capable model cost 2.6 times what they did on the fast one for no measured gain, so they start on the standard model now; an architecture label or a failure still takes them up.

syntaqxcommitted Parentce9ca12Browse files
6 files+539−80/6 viewed
+229−0
1+#!/usr/bin/env node
2+// What Auto (the agent model router, services/runner/src/model-env.ts)
3+// saves against routing as it was before, and against running everything
4+// on the most capable model: an estimate, offline, that never calls a
5+// model.
6+//
7+// node scripts/ops/routing-savings.mjs # the sample in routing-savings.sample.json
8+// node scripts/ops/routing-savings.mjs --json # the same, as JSON
9+// node scripts/ops/routing-savings.mjs --live 30 # g1t's own runs of the last 30 days, from billing
10+// node scripts/ops/routing-savings.mjs --small-turns 1.5 # assume the fast model takes 50% more tokens
11+//
12+// How it estimates. Each task's tokens (input, output, cache reads and
13+// writes) are priced at each tier's list price from the routing's
14+// catalogue (AGENT_ROUTING in services/runner/wrangler.jsonc). Prices are
15+// per token, so the same work on a cheaper tier costs its price ratio,
16+// except that a smaller model may take more turns: its tokens are scaled
17+// by --small-turns (1.25 by default). A task the sample says a tier
18+// cannot do is charged that failed attempt and then the retry Auto makes
19+// one tier up (two failures in a row go to the most capable), so savings
20+// are net of escalation. Cost per merged change is the policy's cost over
21+// the tasks that end merged.
22+//
23+// --live reads, read-only, the runs billing settled and the tokens the
24+// model proxy counted for them (g1t-billing: runs, token_usage), through
25+// Wrangler as you are logged in, or CLOUDFLARE_D1_TOKEN. Live runs carry no
26+// change size, labels or outcome, so a review keeps the tier it ran on and
27+// nothing is escalated: a first look, not a verdict.
28+
29+import { readFileSync } from "node:fs";
30+import { join } from "node:path";
31+
32+import { exec, jsonFrom, wranglerEnv } from "../deploy/cloudflare.mjs";
33+import { ROOT, parseJsonc } from "../deploy/stack.mjs";
34+import { DEFAULT_ROUTING, TIERS, parseRouting, route } from "../../services/runner/src/model-env.ts";
35+
36+const WRANGLER = join(ROOT, "node_modules/wrangler/bin/wrangler.js");
37+const DATABASE = "g1t-billing";
38+const SAMPLE = join(ROOT, "scripts/ops/routing-savings.sample.json");
39+
40+/** The routing the next deploy runs: AGENT_ROUTING in the runner's wrangler.jsonc. */
41+export function configuredRouting(wranglerText) {
42+ try {
43+ return parseRouting(parseJsonc(wranglerText).vars?.AGENT_ROUTING);
44+ } catch {
45+ return DEFAULT_ROUTING;
46+ }
47+}
48+
49+/** What `tokens` cost on `tier`, in dollars, with the fast tier's extra turns. */
50+export function costOn(tier, tokens, routing, smallTurns = 1) {
51+ const price = routing.tiers[tier].price;
52+ if (!price) return null;
53+ const scale = tier === "small" ? smallTurns : 1;
54+ const dollars =
55+ (tokens.input ?? 0) * price.input +
56+ (tokens.output ?? 0) * price.output +
57+ (tokens.cacheRead ?? 0) * price.cacheRead +
58+ (tokens.cacheWrite ?? 0) * price.cacheWrite;
59+ return (dollars * scale) / 1_000_000;
60+}
61+
62+/**
63+ * The tier routing chose before Auto: changes, revisions and answers on
64+ * the large tier, plans and catching up on the small, a review small for
65+ * a small change that touches nothing sensitive, and a retry large.
66+ */
67+export function previousTier(task) {
68+ const kind = task.kind;
69+ if (kind === "plan" || kind === "update") return "small";
70+ if (kind !== "review") return "large";
71+ const change = task.change;
72+ if (!change || !change.files) return "large";
73+ const security = (task.labels ?? []).some((label) => label.toLowerCase() === "security");
74+ const small = !security && (change.sensitive ?? []).length === 0 && change.files <= 10 && change.lines <= 200;
75+ return small ? "small" : "large";
76+}
77+
78+/** Auto's retry: one tier up, the most capable after `frontierAfter` failures in a row. */
79+export function autoNext(routing) {
80+ return (tier, failures) =>
81+ tier === "frontier" ? null : failures >= routing.frontierAfter ? "frontier" : TIERS[TIERS.indexOf(tier) + 1];
82+}
83+
84+/**
85+ * Routing before Auto: a retry ran on the large tier, and after two
86+ * failures there a person was asked.
87+ */
88+export function previousNext(tier, failures) {
89+ return failures >= 2 && tier === "large" ? null : "large";
90+}
91+
92+/**
93+ * What one task costs under a policy that starts it on `first` and, after
94+ * each failure, retries on what `next(tier, failures)` says (null: it
95+ * stops, for a person). `failsOn` lists the tiers the task fails on.
96+ * Returns the attempts, the cost and whether it ended done.
97+ */
98+export function play(task, first, next, routing, smallTurns) {
99+ const fails = new Set(task.failsOn ?? []);
100+ const attempts = [];
101+ let tier = first;
102+ let failures = 0;
103+ for (;;) {
104+ attempts.push(tier);
105+ if (!fails.has(tier)) return { attempts, cost: sum(attempts, task, routing, smallTurns), done: true };
106+ failures += 1;
107+ const then = attempts.length < 5 ? next(tier, failures) : null;
108+ if (!then) return { attempts, cost: sum(attempts, task, routing, smallTurns), done: false };
109+ tier = then;
110+ }
111+}
112+
113+function sum(attempts, task, routing, smallTurns) {
114+ return attempts.reduce((total, tier) => total + (costOn(tier, task.tokens, routing, smallTurns) ?? 0), 0);
115+}
116+
117+/** Auto's first tier for a task, and why, as the runner would route it. */
118+export function autoFirst(task, routing) {
119+ if (task.kind === "review" && task.keepTier) return { tier: task.keepTier, reason: "kept: the change's size is not on record" };
120+ return route(task.kind, { change: task.change ?? null, labels: task.labels ?? [] }, routing);
121+}
122+
123+/** Every policy's cost over the tasks: Auto, routing before it, and the most capable for all. */
124+export function compare(tasks, routing, { smallTurns = 1.25, escalate = true } = {}) {
125+ const stop = () => null;
126+ const policies = {
127+ auto: [(task) => autoFirst(task, routing).tier, escalate ? autoNext(routing) : stop],
128+ before: [(task) => previousTier(task), escalate ? previousNext : stop],
129+ frontier: [() => "frontier", stop],
130+ };
131+ const rows = tasks.map((task) => {
132+ const out = { id: task.id, kind: task.kind, merged: task.merged !== false };
133+ for (const [name, [first, next]] of Object.entries(policies)) {
134+ const played = play(task, first(task), next, routing, smallTurns);
135+ out[name] = { tiers: played.attempts, cost: played.cost, done: played.done };
136+ }
137+ out.reason = autoFirst(task, routing).reason;
138+ return out;
139+ });
140+ const totals = {};
141+ for (const name of Object.keys(policies)) {
142+ const cost = rows.reduce((total, row) => total + row[name].cost, 0);
143+ const merged = rows.filter((row) => row.merged && row[name].done).length;
144+ totals[name] = { cost, merged, perMerged: merged ? cost / merged : null };
145+ }
146+ const saving = (from) => (totals[from].cost > 0 ? 1 - totals.auto.cost / totals[from].cost : 0);
147+ return { rows, totals, savings: { vsBefore: saving("before"), vsFrontier: saving("frontier") } };
148+}
149+
150+/** Billing's runs and their tokens, as tasks. Reviews keep the tier they ran on. */
151+export function tasksFromBilling(rows) {
152+ return rows
153+ .filter((row) => (row.input ?? 0) + (row.output ?? 0) + (row.cache_read ?? 0) + (row.cache_write ?? 0) > 0)
154+ .map((row) => ({
155+ id: row.id,
156+ kind: ["implement", "review", "update", "plan"].includes(row.task) ? row.task : "implement",
157+ keepTier: row.task === "review" && TIERS.includes(row.tier) ? row.tier : null,
158+ tokens: { input: row.input ?? 0, output: row.output ?? 0, cacheRead: row.cache_read ?? 0, cacheWrite: row.cache_write ?? 0 },
159+ merged: true,
160+ }));
161+}
162+
163+async function d1(sql) {
164+ const env = wranglerEnv();
165+ if (process.env.CLOUDFLARE_D1_TOKEN) env.CLOUDFLARE_API_TOKEN = process.env.CLOUDFLARE_D1_TOKEN;
166+ const { code, out } = await exec(process.execPath, [WRANGLER, "d1", "execute", DATABASE, "--remote", "--json", "--command", sql], {
167+ env,
168+ });
169+ if (code !== 0) throw new Error(`wrangler d1 execute failed:\n${out.slice(-800)}`);
170+ return jsonFrom(out)[0]?.results ?? [];
171+}
172+
173+function liveSql(days) {
174+ const since = new Date(Date.now() - days * 86_400_000).toISOString();
175+ return `SELECT r.id, r.task, r.tier, r.model,
176+ SUM(t.input) AS input, SUM(t.output) AS output, SUM(t.cache_read) AS cache_read, SUM(t.cache_write) AS cache_write
177+ FROM runs r JOIN token_usage t ON t.session = r.session_id AND t.workspace = r.workspace
178+ WHERE r.created_at >= '${since}' AND r.session_id IS NOT NULL
179+ GROUP BY r.id LIMIT 5000`;
180+}
181+
182+function dollars(n) {
183+ return n == null ? "—" : `$${n.toFixed(n < 1 ? 4 : 2)}`;
184+}
185+
186+function print(result) {
187+ const { rows, totals, savings } = result;
188+ console.log("task kind auto before most capable");
189+ for (const row of rows) {
190+ const cell = (p) => `${dollars(row[p].cost)} ${row[p].tiers.join(">")}${row[p].done ? "" : " (failed)"}`;
191+ console.log(`${row.id.padEnd(26)} ${row.kind.padEnd(10)} ${cell("auto").padEnd(20)} ${cell("before").padEnd(14)} ${cell("frontier")}`);
192+ }
193+ console.log("");
194+ for (const [name, total] of Object.entries(totals)) {
195+ console.log(`${name.padEnd(9)} ${dollars(total.cost).padStart(10)} merged ${total.merged} per merged change ${dollars(total.perMerged)}`);
196+ }
197+ console.log("");
198+ console.log(`Auto against routing before it: ${(savings.vsBefore * 100).toFixed(1)}% ${savings.vsBefore >= 0 ? "less" : "more"}`);
199+ console.log(`Auto against the most capable model for everything: ${(savings.vsFrontier * 100).toFixed(1)}% less`);
200+}
201+
202+async function main() {
203+ const args = process.argv.slice(2);
204+ const flag = (name) => {
205+ const at = args.indexOf(name);
206+ return at >= 0 ? args[at + 1] : undefined;
207+ };
208+ const routing = configuredRouting(readFileSync(join(ROOT, "services/runner/wrangler.jsonc"), "utf8"));
209+ const smallTurns = Number(flag("--small-turns") ?? 1.25);
210+ let tasks;
211+ let escalate = true;
212+ if (args.includes("--live")) {
213+ const days = Number(flag("--live") ?? 30) || 30;
214+ tasks = tasksFromBilling(await d1(liveSql(days)));
215+ escalate = false;
216+ } else {
217+ tasks = JSON.parse(readFileSync(SAMPLE, "utf8")).tasks;
218+ }
219+ const result = compare(tasks, routing, { smallTurns, escalate });
220+ if (args.includes("--json")) console.log(JSON.stringify(result, null, 2));
221+ else print(result);
222+}
223+
224+if (process.argv[1]?.replaceAll("\\", "/").endsWith("scripts/ops/routing-savings.mjs")) {
225+ main().catch((error) => {
226+ console.error(String(error?.message ?? error));
227+ process.exit(1);
228+ });
229+}
+212−0
1+{
2+ "about": "A handful of tasks shaped like g1t's own agent work, for estimating what routing saves. Illustrative, not production data: tokens follow the profile in docs/research/reviewbench.md (turns from the change's size, 92% of input read from cache, output per turn), and failsOn is an assumption, the tiers a task of that difficulty is expected to fail on. Run with --live for g1t's real runs.",
3+ "tasks": [
4+ {
5+ "id": "fix-typo-in-readme",
6+ "kind": "implement",
7+ "change": {
8+ "files": 1,
9+ "lines": 6,
10+ "sensitive": []
11+ },
12+ "labels": [
13+ "docs"
14+ ],
15+ "merged": true,
16+ "tokens": {
17+ "input": 11455,
18+ "output": 8400,
19+ "cacheRead": 351314,
20+ "cacheWrite": 48072
21+ }
22+ },
23+ {
24+ "id": "add-search-filter",
25+ "kind": "implement",
26+ "change": {
27+ "files": 6,
28+ "lines": 240,
29+ "sensitive": []
30+ },
31+ "merged": true,
32+ "tokens": {
33+ "input": 80004,
34+ "output": 31200,
35+ "cacheRead": 2453474,
36+ "cacheWrite": 118380
37+ }
38+ },
39+ {
40+ "id": "rework-billing-ledger",
41+ "kind": "implement",
42+ "change": {
43+ "files": 14,
44+ "lines": 900,
45+ "sensitive": []
46+ },
47+ "failsOn": [
48+ "small",
49+ "large"
50+ ],
51+ "merged": true,
52+ "tokens": {
53+ "input": 184590,
54+ "output": 48000,
55+ "cacheRead": 5660760,
56+ "cacheWrite": 178800
57+ }
58+ },
59+ {
60+ "id": "split-auth-service",
61+ "kind": "implement",
62+ "change": {
63+ "files": 20,
64+ "lines": 1500,
65+ "sensitive": []
66+ },
67+ "labels": [
68+ "architecture"
69+ ],
70+ "failsOn": [
71+ "small",
72+ "large"
73+ ],
74+ "merged": true,
75+ "tokens": {
76+ "input": 278415,
77+ "output": 60000,
78+ "cacheRead": 8538060,
79+ "cacheWrite": 180000
80+ }
81+ },
82+ {
83+ "id": "fix-failing-check",
84+ "kind": "revise",
85+ "change": {
86+ "files": 2,
87+ "lines": 30,
88+ "sensitive": []
89+ },
90+ "merged": false,
91+ "tokens": {
92+ "input": 19563,
93+ "output": 11900,
94+ "cacheRead": 599950,
95+ "cacheWrite": 60860
96+ }
97+ },
98+ {
99+ "id": "what-does-this-flag-do",
100+ "kind": "answer",
101+ "merged": false,
102+ "tokens": {
103+ "input": 7560,
104+ "output": 3600,
105+ "cacheRead": 231840,
106+ "cacheWrite": 40500
107+ }
108+ },
109+ {
110+ "id": "why-is-this-slow",
111+ "kind": "answer",
112+ "failsOn": [
113+ "small"
114+ ],
115+ "merged": false,
116+ "tokens": {
117+ "input": 19380,
118+ "output": 6800,
119+ "cacheRead": 594320,
120+ "cacheWrite": 60500
121+ }
122+ },
123+ {
124+ "id": "review-small-change",
125+ "kind": "review",
126+ "change": {
127+ "files": 3,
128+ "lines": 80,
129+ "sensitive": []
130+ },
131+ "merged": false,
132+ "tokens": {
133+ "input": 25626,
134+ "output": 9000,
135+ "cacheRead": 785864,
136+ "cacheWrite": 68960
137+ }
138+ },
139+ {
140+ "id": "review-ci-change",
141+ "kind": "review",
142+ "change": {
143+ "files": 2,
144+ "lines": 40,
145+ "sensitive": [
146+ "CI workflows"
147+ ]
148+ },
149+ "merged": false,
150+ "tokens": {
151+ "input": 21454,
152+ "output": 8100,
153+ "cacheRead": 657928,
154+ "cacheWrite": 63480
155+ }
156+ },
157+ {
158+ "id": "review-medium-change",
159+ "kind": "review",
160+ "change": {
161+ "files": 12,
162+ "lines": 450,
163+ "sensitive": []
164+ },
165+ "merged": false,
166+ "tokens": {
167+ "input": 65943,
168+ "output": 15300,
169+ "cacheRead": 2022252,
170+ "cacheWrite": 108400
171+ }
172+ },
173+ {
174+ "id": "review-large-migration",
175+ "kind": "review",
176+ "change": {
177+ "files": 80,
178+ "lines": 4200,
179+ "sensitive": []
180+ },
181+ "merged": false,
182+ "tokens": {
183+ "input": 355590,
184+ "output": 36000,
185+ "cacheRead": 10904760,
186+ "cacheWrite": 180000
187+ }
188+ },
189+ {
190+ "id": "catch-up-with-conflict",
191+ "kind": "update",
192+ "merged": false,
193+ "tokens": {
194+ "input": 21583,
195+ "output": 9000,
196+ "cacheRead": 661903,
197+ "cacheWrite": 63720
198+ }
199+ },
200+ {
201+ "id": "plan-checkout-redesign",
202+ "kind": "plan",
203+ "merged": false,
204+ "tokens": {
205+ "input": 54480,
206+ "output": 28800,
207+ "cacheRead": 1670720,
208+ "cacheWrite": 98000
209+ }
210+ }
211+ ]
212+}
+89−0
1+import assert from "node:assert/strict";
2+import { readFileSync } from "node:fs";
3+import { test } from "node:test";
4+
5+import { DEFAULT_ROUTING } from "../../services/runner/src/model-env.ts";
6+import { autoNext, compare, configuredRouting, costOn, play, previousNext, previousTier, tasksFromBilling } from "./routing-savings.mjs";
7+
8+const routing = DEFAULT_ROUTING;
9+const million = { input: 1_000_000, output: 0, cacheRead: 0, cacheWrite: 0 };
10+
11+test("tokens are priced at each tier's list price, the fast tier's scaled for extra turns", () => {
12+ assert.equal(costOn("small", million, routing), 1);
13+ assert.equal(costOn("large", million, routing), 2);
14+ assert.equal(costOn("frontier", million, routing), 4);
15+ assert.equal(costOn("small", million, routing, 1.5), 1.5);
16+ assert.equal(costOn("large", million, routing, 1.5), 2);
17+ const mixed = { input: 0, output: 1_000_000, cacheRead: 10_000_000, cacheWrite: 1_000_000 };
18+ // $10 of output, $2 of cache reads, $2.50 of cache writes.
19+ assert.equal(costOn("large", mixed, routing), 14.5);
20+ const unpriced = { ...routing, tiers: { ...routing.tiers, large: { modelName: "X", model: "x" } } };
21+ assert.equal(costOn("large", million, unpriced), null);
22+});
23+
24+test("a failed attempt is paid for, and Auto retries one tier up until the most capable", () => {
25+ const task = { id: "t", kind: "implement", failsOn: ["small", "large"], tokens: million };
26+ assert.deepEqual(play(task, "small", autoNext(routing), routing, 1), { attempts: ["small", "large", "frontier"], cost: 7, done: true });
27+ // Routing before Auto stayed on the large tier and asked a person after two failures.
28+ assert.deepEqual(play(task, "large", previousNext, routing, 1), { attempts: ["large", "large"], cost: 4, done: false });
29+ // Nothing past the most capable model.
30+ const hopeless = { ...task, failsOn: ["small", "large", "frontier"] };
31+ assert.equal(play(hopeless, "small", autoNext(routing), routing, 1).done, false);
32+});
33+
34+test("routing before Auto is replayed as it was", () => {
35+ assert.equal(previousTier({ kind: "plan" }), "small");
36+ assert.equal(previousTier({ kind: "update" }), "small");
37+ assert.equal(previousTier({ kind: "answer" }), "large");
38+ assert.equal(previousTier({ kind: "review", change: { files: 3, lines: 80, sensitive: [] } }), "small");
39+ assert.equal(previousTier({ kind: "review", change: { files: 3, lines: 80, sensitive: [] }, labels: ["Security"] }), "large");
40+ assert.equal(previousTier({ kind: "review" }), "large");
41+});
42+
43+test("the comparison totals each policy and its cost per merged change", () => {
44+ const tasks = [
45+ { id: "a", kind: "update", tokens: million, merged: false },
46+ { id: "b", kind: "implement", labels: ["docs"], tokens: million, merged: true },
47+ ];
48+ const { totals, savings, rows } = compare(tasks, routing, { smallTurns: 1 });
49+ // Auto: both fast. Before: catching up fast, the change standard.
50+ assert.equal(totals.auto.cost, 2);
51+ assert.equal(totals.before.cost, 3);
52+ assert.equal(totals.frontier.cost, 8);
53+ assert.equal(totals.auto.perMerged, 2);
54+ assert.ok(Math.abs(savings.vsBefore - 1 / 3) < 1e-9);
55+ assert.equal(savings.vsFrontier, 0.75);
56+ assert.match(rows[1].reason, /labelled docs/);
57+});
58+
59+test("the sample is well formed and every task is priced", () => {
60+ const sample = JSON.parse(readFileSync(new URL("./routing-savings.sample.json", import.meta.url), "utf8"));
61+ assert.ok(sample.tasks.length >= 10);
62+ for (const task of sample.tasks) {
63+ assert.ok(["implement", "revise", "answer", "review", "update", "plan"].includes(task.kind), task.id);
64+ assert.ok(costOn("large", task.tokens, routing) > 0, task.id);
65+ }
66+ const { totals } = compare(sample.tasks, routing);
67+ assert.ok(totals.auto.cost > 0 && totals.before.cost > 0 && totals.frontier.cost > 0);
68+});
69+
70+test("billing's runs become tasks; reviews keep the tier they ran on", () => {
71+ const tasks = tasksFromBilling([
72+ { id: "run_1", task: "review", tier: "small", input: 10, output: 5, cache_read: 0, cache_write: 0 },
73+ { id: "run_2", task: "implement", tier: "large", input: 0, output: 0, cache_read: 0, cache_write: 0 },
74+ { id: "run_3", task: "plan", tier: null, input: 1, output: 1, cache_read: 1, cache_write: 1 },
75+ ]);
76+ assert.deepEqual(
77+ tasks.map((t) => [t.id, t.kind, t.keepTier]),
78+ [
79+ ["run_1", "review", "small"],
80+ ["run_3", "plan", null],
81+ ],
82+ );
83+});
84+
85+test("the routing estimated is the runner's configuration", () => {
86+ const text = readFileSync(new URL("../../services/runner/wrangler.jsonc", import.meta.url), "utf8");
87+ assert.deepEqual(configuredRouting(text), DEFAULT_ROUTING);
88+ assert.deepEqual(configuredRouting("not jsonc {"), DEFAULT_ROUTING);
89+});
+7−6
4444
4545 const small: ChangeSize = { files: 3, lines: 80, sensitive: [] };
4646
47−test("Auto starts each kind of job on its tier: fast for catching up and answering, standard for changes, most capable for plans", () => {
47+test("Auto starts each kind of job on its tier: fast for catching up and answering, standard for changes and plans", () => {
4848 assert.equal(chooseTier("update", {}, routes), "small");
4949 assert.equal(chooseTier("answer", {}, routes), "small");
5050 assert.equal(chooseTier("implement", {}, routes), "large");
5151 assert.equal(chooseTier("revise", {}, routes), "large");
5252 assert.equal(chooseTier("implement", { change: small }, routes), "large");
53− assert.equal(chooseTier("plan", {}, routes), "frontier");
53+ assert.equal(chooseTier("plan", {}, routes), "large");
54+ assert.equal(chooseTier("plan", { labels: ["architecture"] }, routes), "frontier");
5455 });
5556
5657 test("revising and answering go to the workspace's implement route and bill", () => {
8485 assert.equal(chooseTier("implement", { labels: ["docs"] }, routes), "small");
8586 assert.equal(chooseTier("answer", { labels: ["typo"] }, routes), "small");
8687 // A small label never takes a review or a plan down.
87− assert.equal(chooseTier("plan", { labels: ["docs"] }, routes), "frontier");
88+ assert.equal(chooseTier("plan", { labels: ["docs"] }, routes), "large");
8889 // Security outranks documentation.
8990 assert.equal(chooseTier("implement", { labels: ["docs", "security"] }, routes), "large");
9091 });
121122 // Too few runs to tell, or not quite enough of them finished: no change.
122123 assert.equal(chooseTier("implement", { history: ok("small", 4) }, routes), "large");
123124 assert.equal(chooseTier("implement", { history: [...ok("small", 8), ...bad("small", 2)] }, routes), "large");
124− // Plans step down from the most capable model the same way.
125− assert.equal(chooseTier("plan", { history: ok("large", 6) }, routes), "large");
125+ // Plans step down the same way.
126+ assert.equal(chooseTier("plan", { history: ok("small", 6) }, routes), "small");
126127 });
127128
128129 test("learning never steps down sensitive or labelled work, nor a retry", () => {
198199 assert.deepEqual(parsed.tiers.frontier, DEFAULT_ROUTING.tiers.frontier);
199200 assert.equal(chooseTier("update", {}, parsed), "large");
200201 // A rule that names no tier keeps the default.
201− assert.equal(chooseTier("plan", {}, parsed), "frontier");
202+ assert.equal(chooseTier("plan", {}, parsed), "large");
202203 assert.equal(parsed.tasks.answer, "change");
203204 assert.equal(chooseTier("review", { change: { ...small, lines: 51 } }, parsed), "large");
204205 assert.equal(chooseTier("review", { change: { ...small, files: 10, lines: 50 } }, parsed), "small");
+1−1
143143 price: { input: 4, output: 20, cacheRead: 0.2, cacheWrite: 5 },
144144 },
145145 },
146− tasks: { implement: "large", revise: "large", answer: "small", review: "change", update: "small", plan: "frontier" },
146+ tasks: { implement: "large", revise: "large", answer: "small", review: "change", update: "small", plan: "large" },
147147 smallChange: { files: 10, lines: 200 },
148148 largeChange: { files: 60, lines: 3000 },
149149 largeLabels: ["security"],
+1−1
8787 // "smallLabels" move work by its issue's labels. A failed attempt goes one
8888 // tier up and "frontierAfter" failures in a row to frontier; "learning"
8989 // steps work down or up by the repository's own recent runs of the kind.
90− "AGENT_ROUTING": "{\"tiers\":{\"small\":{\"modelName\":\"Claude Haiku 4.5\",\"model\":\"claude-haiku-4-5-20251001\",\"price\":{\"input\":1,\"output\":5,\"cacheRead\":0.1,\"cacheWrite\":1.25}},\"large\":{\"modelName\":\"Claude Sonnet 5.5\",\"model\":\"claude-sonnet-5-5\",\"price\":{\"input\":2,\"output\":10,\"cacheRead\":0.2,\"cacheWrite\":2.5}},\"frontier\":{\"modelName\":\"Claude Opus 5.5\",\"model\":\"claude-opus-5-5\",\"price\":{\"input\":4,\"output\":20,\"cacheRead\":0.2,\"cacheWrite\":5}}},\"tasks\":{\"implement\":\"large\",\"revise\":\"large\",\"answer\":\"small\",\"review\":\"change\",\"update\":\"small\",\"plan\":\"frontier\"},\"smallChange\":{\"files\":10,\"lines\":200},\"largeChange\":{\"files\":60,\"lines\":3000},\"largeLabels\":[\"security\"],\"frontierLabels\":[\"architecture\"],\"smallLabels\":[\"documentation\",\"docs\",\"typo\"],\"frontierAfter\":2,\"learning\":{\"window\":20,\"minRuns\":5,\"stepDownAt\":0.9,\"stepUpAt\":0.5}}",
90+ "AGENT_ROUTING": "{\"tiers\":{\"small\":{\"modelName\":\"Claude Haiku 4.5\",\"model\":\"claude-haiku-4-5-20251001\",\"price\":{\"input\":1,\"output\":5,\"cacheRead\":0.1,\"cacheWrite\":1.25}},\"large\":{\"modelName\":\"Claude Sonnet 5.5\",\"model\":\"claude-sonnet-5-5\",\"price\":{\"input\":2,\"output\":10,\"cacheRead\":0.2,\"cacheWrite\":2.5}},\"frontier\":{\"modelName\":\"Claude Opus 5.5\",\"model\":\"claude-opus-5-5\",\"price\":{\"input\":4,\"output\":20,\"cacheRead\":0.2,\"cacheWrite\":5}}},\"tasks\":{\"implement\":\"large\",\"revise\":\"large\",\"answer\":\"small\",\"review\":\"change\",\"update\":\"small\",\"plan\":\"large\"},\"smallChange\":{\"files\":10,\"lines\":200},\"largeChange\":{\"files\":60,\"lines\":3000},\"largeLabels\":[\"security\"],\"frontierLabels\":[\"architecture\"],\"smallLabels\":[\"documentation\",\"docs\",\"typo\"],\"frontierAfter\":2,\"learning\":{\"window\":20,\"minRuns\":5,\"stepDownAt\":0.9,\"stepUpAt\":0.5}}",
9191 // Where sandboxes send model requests, with a token for their run.
9292 // The proxy holds the keys: g1t's gateway's, or the workspace's own.
9393 "MODELS_URL": "https://models.g1t.sh",