Ops: estimate what Auto saves, offline; plans start on the standard model
scripts/ops/routing-savings.mjs replays tasks through the router and prices their tokens at each tier's list price from the runner's own configuration: Auto, routing as it was before, and the most capable model for everything, with failed attempts paid for and retried as each policy would, and cost per merged change. It never calls a model. The sample is a handful of tasks shaped like g1t's own work (tokens from the profile in docs/research/reviewbench.md, failures assumed); --live reads billing's settled runs and their counted tokens, read-only. On the sample, plans on the most capable model cost 2.6 times what they did on the fast one for no measured gain, so they start on the standard model now; an architecture label or a failure still takes them up.
| 1 | + | #!/usr/bin/env node | |
| 2 | + | // What Auto (the agent model router, services/runner/src/model-env.ts) | |
| 3 | + | // saves against routing as it was before, and against running everything | |
| 4 | + | // on the most capable model: an estimate, offline, that never calls a | |
| 5 | + | // model. | |
| 6 | + | // | |
| 7 | + | // node scripts/ops/routing-savings.mjs # the sample in routing-savings.sample.json | |
| 8 | + | // node scripts/ops/routing-savings.mjs --json # the same, as JSON | |
| 9 | + | // node scripts/ops/routing-savings.mjs --live 30 # g1t's own runs of the last 30 days, from billing | |
| 10 | + | // node scripts/ops/routing-savings.mjs --small-turns 1.5 # assume the fast model takes 50% more tokens | |
| 11 | + | // | |
| 12 | + | // How it estimates. Each task's tokens (input, output, cache reads and | |
| 13 | + | // writes) are priced at each tier's list price from the routing's | |
| 14 | + | // catalogue (AGENT_ROUTING in services/runner/wrangler.jsonc). Prices are | |
| 15 | + | // per token, so the same work on a cheaper tier costs its price ratio, | |
| 16 | + | // except that a smaller model may take more turns: its tokens are scaled | |
| 17 | + | // by --small-turns (1.25 by default). A task the sample says a tier | |
| 18 | + | // cannot do is charged that failed attempt and then the retry Auto makes | |
| 19 | + | // one tier up (two failures in a row go to the most capable), so savings | |
| 20 | + | // are net of escalation. Cost per merged change is the policy's cost over | |
| 21 | + | // the tasks that end merged. | |
| 22 | + | // | |
| 23 | + | // --live reads, read-only, the runs billing settled and the tokens the | |
| 24 | + | // model proxy counted for them (g1t-billing: runs, token_usage), through | |
| 25 | + | // Wrangler as you are logged in, or CLOUDFLARE_D1_TOKEN. Live runs carry no | |
| 26 | + | // change size, labels or outcome, so a review keeps the tier it ran on and | |
| 27 | + | // nothing is escalated: a first look, not a verdict. | |
| 28 | + | ||
| 29 | + | import { readFileSync } from "node:fs"; | |
| 30 | + | import { join } from "node:path"; | |
| 31 | + | ||
| 32 | + | import { exec, jsonFrom, wranglerEnv } from "../deploy/cloudflare.mjs"; | |
| 33 | + | import { ROOT, parseJsonc } from "../deploy/stack.mjs"; | |
| 34 | + | import { DEFAULT_ROUTING, TIERS, parseRouting, route } from "../../services/runner/src/model-env.ts"; | |
| 35 | + | ||
| 36 | + | const WRANGLER = join(ROOT, "node_modules/wrangler/bin/wrangler.js"); | |
| 37 | + | const DATABASE = "g1t-billing"; | |
| 38 | + | const SAMPLE = join(ROOT, "scripts/ops/routing-savings.sample.json"); | |
| 39 | + | ||
| 40 | + | /** The routing the next deploy runs: AGENT_ROUTING in the runner's wrangler.jsonc. */ | |
| 41 | + | export function configuredRouting(wranglerText) { | |
| 42 | + | try { | |
| 43 | + | return parseRouting(parseJsonc(wranglerText).vars?.AGENT_ROUTING); | |
| 44 | + | } catch { | |
| 45 | + | return DEFAULT_ROUTING; | |
| 46 | + | } | |
| 47 | + | } | |
| 48 | + | ||
| 49 | + | /** What `tokens` cost on `tier`, in dollars, with the fast tier's extra turns. */ | |
| 50 | + | export function costOn(tier, tokens, routing, smallTurns = 1) { | |
| 51 | + | const price = routing.tiers[tier].price; | |
| 52 | + | if (!price) return null; | |
| 53 | + | const scale = tier === "small" ? smallTurns : 1; | |
| 54 | + | const dollars = | |
| 55 | + | (tokens.input ?? 0) * price.input + | |
| 56 | + | (tokens.output ?? 0) * price.output + | |
| 57 | + | (tokens.cacheRead ?? 0) * price.cacheRead + | |
| 58 | + | (tokens.cacheWrite ?? 0) * price.cacheWrite; | |
| 59 | + | return (dollars * scale) / 1_000_000; | |
| 60 | + | } | |
| 61 | + | ||
| 62 | + | /** | |
| 63 | + | * The tier routing chose before Auto: changes, revisions and answers on | |
| 64 | + | * the large tier, plans and catching up on the small, a review small for | |
| 65 | + | * a small change that touches nothing sensitive, and a retry large. | |
| 66 | + | */ | |
| 67 | + | export function previousTier(task) { | |
| 68 | + | const kind = task.kind; | |
| 69 | + | if (kind === "plan" || kind === "update") return "small"; | |
| 70 | + | if (kind !== "review") return "large"; | |
| 71 | + | const change = task.change; | |
| 72 | + | if (!change || !change.files) return "large"; | |
| 73 | + | const security = (task.labels ?? []).some((label) => label.toLowerCase() === "security"); | |
| 74 | + | const small = !security && (change.sensitive ?? []).length === 0 && change.files <= 10 && change.lines <= 200; | |
| 75 | + | return small ? "small" : "large"; | |
| 76 | + | } | |
| 77 | + | ||
| 78 | + | /** Auto's retry: one tier up, the most capable after `frontierAfter` failures in a row. */ | |
| 79 | + | export function autoNext(routing) { | |
| 80 | + | return (tier, failures) => | |
| 81 | + | tier === "frontier" ? null : failures >= routing.frontierAfter ? "frontier" : TIERS[TIERS.indexOf(tier) + 1]; | |
| 82 | + | } | |
| 83 | + | ||
| 84 | + | /** | |
| 85 | + | * Routing before Auto: a retry ran on the large tier, and after two | |
| 86 | + | * failures there a person was asked. | |
| 87 | + | */ | |
| 88 | + | export function previousNext(tier, failures) { | |
| 89 | + | return failures >= 2 && tier === "large" ? null : "large"; | |
| 90 | + | } | |
| 91 | + | ||
| 92 | + | /** | |
| 93 | + | * What one task costs under a policy that starts it on `first` and, after | |
| 94 | + | * each failure, retries on what `next(tier, failures)` says (null: it | |
| 95 | + | * stops, for a person). `failsOn` lists the tiers the task fails on. | |
| 96 | + | * Returns the attempts, the cost and whether it ended done. | |
| 97 | + | */ | |
| 98 | + | export function play(task, first, next, routing, smallTurns) { | |
| 99 | + | const fails = new Set(task.failsOn ?? []); | |
| 100 | + | const attempts = []; | |
| 101 | + | let tier = first; | |
| 102 | + | let failures = 0; | |
| 103 | + | for (;;) { | |
| 104 | + | attempts.push(tier); | |
| 105 | + | if (!fails.has(tier)) return { attempts, cost: sum(attempts, task, routing, smallTurns), done: true }; | |
| 106 | + | failures += 1; | |
| 107 | + | const then = attempts.length < 5 ? next(tier, failures) : null; | |
| 108 | + | if (!then) return { attempts, cost: sum(attempts, task, routing, smallTurns), done: false }; | |
| 109 | + | tier = then; | |
| 110 | + | } | |
| 111 | + | } | |
| 112 | + | ||
| 113 | + | function sum(attempts, task, routing, smallTurns) { | |
| 114 | + | return attempts.reduce((total, tier) => total + (costOn(tier, task.tokens, routing, smallTurns) ?? 0), 0); | |
| 115 | + | } | |
| 116 | + | ||
| 117 | + | /** Auto's first tier for a task, and why, as the runner would route it. */ | |
| 118 | + | export function autoFirst(task, routing) { | |
| 119 | + | if (task.kind === "review" && task.keepTier) return { tier: task.keepTier, reason: "kept: the change's size is not on record" }; | |
| 120 | + | return route(task.kind, { change: task.change ?? null, labels: task.labels ?? [] }, routing); | |
| 121 | + | } | |
| 122 | + | ||
| 123 | + | /** Every policy's cost over the tasks: Auto, routing before it, and the most capable for all. */ | |
| 124 | + | export function compare(tasks, routing, { smallTurns = 1.25, escalate = true } = {}) { | |
| 125 | + | const stop = () => null; | |
| 126 | + | const policies = { | |
| 127 | + | auto: [(task) => autoFirst(task, routing).tier, escalate ? autoNext(routing) : stop], | |
| 128 | + | before: [(task) => previousTier(task), escalate ? previousNext : stop], | |
| 129 | + | frontier: [() => "frontier", stop], | |
| 130 | + | }; | |
| 131 | + | const rows = tasks.map((task) => { | |
| 132 | + | const out = { id: task.id, kind: task.kind, merged: task.merged !== false }; | |
| 133 | + | for (const [name, [first, next]] of Object.entries(policies)) { | |
| 134 | + | const played = play(task, first(task), next, routing, smallTurns); | |
| 135 | + | out[name] = { tiers: played.attempts, cost: played.cost, done: played.done }; | |
| 136 | + | } | |
| 137 | + | out.reason = autoFirst(task, routing).reason; | |
| 138 | + | return out; | |
| 139 | + | }); | |
| 140 | + | const totals = {}; | |
| 141 | + | for (const name of Object.keys(policies)) { | |
| 142 | + | const cost = rows.reduce((total, row) => total + row[name].cost, 0); | |
| 143 | + | const merged = rows.filter((row) => row.merged && row[name].done).length; | |
| 144 | + | totals[name] = { cost, merged, perMerged: merged ? cost / merged : null }; | |
| 145 | + | } | |
| 146 | + | const saving = (from) => (totals[from].cost > 0 ? 1 - totals.auto.cost / totals[from].cost : 0); | |
| 147 | + | return { rows, totals, savings: { vsBefore: saving("before"), vsFrontier: saving("frontier") } }; | |
| 148 | + | } | |
| 149 | + | ||
| 150 | + | /** Billing's runs and their tokens, as tasks. Reviews keep the tier they ran on. */ | |
| 151 | + | export function tasksFromBilling(rows) { | |
| 152 | + | return rows | |
| 153 | + | .filter((row) => (row.input ?? 0) + (row.output ?? 0) + (row.cache_read ?? 0) + (row.cache_write ?? 0) > 0) | |
| 154 | + | .map((row) => ({ | |
| 155 | + | id: row.id, | |
| 156 | + | kind: ["implement", "review", "update", "plan"].includes(row.task) ? row.task : "implement", | |
| 157 | + | keepTier: row.task === "review" && TIERS.includes(row.tier) ? row.tier : null, | |
| 158 | + | tokens: { input: row.input ?? 0, output: row.output ?? 0, cacheRead: row.cache_read ?? 0, cacheWrite: row.cache_write ?? 0 }, | |
| 159 | + | merged: true, | |
| 160 | + | })); | |
| 161 | + | } | |
| 162 | + | ||
| 163 | + | async function d1(sql) { | |
| 164 | + | const env = wranglerEnv(); | |
| 165 | + | if (process.env.CLOUDFLARE_D1_TOKEN) env.CLOUDFLARE_API_TOKEN = process.env.CLOUDFLARE_D1_TOKEN; | |
| 166 | + | const { code, out } = await exec(process.execPath, [WRANGLER, "d1", "execute", DATABASE, "--remote", "--json", "--command", sql], { | |
| 167 | + | env, | |
| 168 | + | }); | |
| 169 | + | if (code !== 0) throw new Error(`wrangler d1 execute failed:\n${out.slice(-800)}`); | |
| 170 | + | return jsonFrom(out)[0]?.results ?? []; | |
| 171 | + | } | |
| 172 | + | ||
| 173 | + | function liveSql(days) { | |
| 174 | + | const since = new Date(Date.now() - days * 86_400_000).toISOString(); | |
| 175 | + | return `SELECT r.id, r.task, r.tier, r.model, | |
| 176 | + | SUM(t.input) AS input, SUM(t.output) AS output, SUM(t.cache_read) AS cache_read, SUM(t.cache_write) AS cache_write | |
| 177 | + | FROM runs r JOIN token_usage t ON t.session = r.session_id AND t.workspace = r.workspace | |
| 178 | + | WHERE r.created_at >= '${since}' AND r.session_id IS NOT NULL | |
| 179 | + | GROUP BY r.id LIMIT 5000`; | |
| 180 | + | } | |
| 181 | + | ||
| 182 | + | function dollars(n) { | |
| 183 | + | return n == null ? "—" : `$${n.toFixed(n < 1 ? 4 : 2)}`; | |
| 184 | + | } | |
| 185 | + | ||
| 186 | + | function print(result) { | |
| 187 | + | const { rows, totals, savings } = result; | |
| 188 | + | console.log("task kind auto before most capable"); | |
| 189 | + | for (const row of rows) { | |
| 190 | + | const cell = (p) => `${dollars(row[p].cost)} ${row[p].tiers.join(">")}${row[p].done ? "" : " (failed)"}`; | |
| 191 | + | console.log(`${row.id.padEnd(26)} ${row.kind.padEnd(10)} ${cell("auto").padEnd(20)} ${cell("before").padEnd(14)} ${cell("frontier")}`); | |
| 192 | + | } | |
| 193 | + | console.log(""); | |
| 194 | + | for (const [name, total] of Object.entries(totals)) { | |
| 195 | + | console.log(`${name.padEnd(9)} ${dollars(total.cost).padStart(10)} merged ${total.merged} per merged change ${dollars(total.perMerged)}`); | |
| 196 | + | } | |
| 197 | + | console.log(""); | |
| 198 | + | console.log(`Auto against routing before it: ${(savings.vsBefore * 100).toFixed(1)}% ${savings.vsBefore >= 0 ? "less" : "more"}`); | |
| 199 | + | console.log(`Auto against the most capable model for everything: ${(savings.vsFrontier * 100).toFixed(1)}% less`); | |
| 200 | + | } | |
| 201 | + | ||
| 202 | + | async function main() { | |
| 203 | + | const args = process.argv.slice(2); | |
| 204 | + | const flag = (name) => { | |
| 205 | + | const at = args.indexOf(name); | |
| 206 | + | return at >= 0 ? args[at + 1] : undefined; | |
| 207 | + | }; | |
| 208 | + | const routing = configuredRouting(readFileSync(join(ROOT, "services/runner/wrangler.jsonc"), "utf8")); | |
| 209 | + | const smallTurns = Number(flag("--small-turns") ?? 1.25); | |
| 210 | + | let tasks; | |
| 211 | + | let escalate = true; | |
| 212 | + | if (args.includes("--live")) { | |
| 213 | + | const days = Number(flag("--live") ?? 30) || 30; | |
| 214 | + | tasks = tasksFromBilling(await d1(liveSql(days))); | |
| 215 | + | escalate = false; | |
| 216 | + | } else { | |
| 217 | + | tasks = JSON.parse(readFileSync(SAMPLE, "utf8")).tasks; | |
| 218 | + | } | |
| 219 | + | const result = compare(tasks, routing, { smallTurns, escalate }); | |
| 220 | + | if (args.includes("--json")) console.log(JSON.stringify(result, null, 2)); | |
| 221 | + | else print(result); | |
| 222 | + | } | |
| 223 | + | ||
| 224 | + | if (process.argv[1]?.replaceAll("\\", "/").endsWith("scripts/ops/routing-savings.mjs")) { | |
| 225 | + | main().catch((error) => { | |
| 226 | + | console.error(String(error?.message ?? error)); | |
| 227 | + | process.exit(1); | |
| 228 | + | }); | |
| 229 | + | } |
| 1 | + | { | |
| 2 | + | "about": "A handful of tasks shaped like g1t's own agent work, for estimating what routing saves. Illustrative, not production data: tokens follow the profile in docs/research/reviewbench.md (turns from the change's size, 92% of input read from cache, output per turn), and failsOn is an assumption, the tiers a task of that difficulty is expected to fail on. Run with --live for g1t's real runs.", | |
| 3 | + | "tasks": [ | |
| 4 | + | { | |
| 5 | + | "id": "fix-typo-in-readme", | |
| 6 | + | "kind": "implement", | |
| 7 | + | "change": { | |
| 8 | + | "files": 1, | |
| 9 | + | "lines": 6, | |
| 10 | + | "sensitive": [] | |
| 11 | + | }, | |
| 12 | + | "labels": [ | |
| 13 | + | "docs" | |
| 14 | + | ], | |
| 15 | + | "merged": true, | |
| 16 | + | "tokens": { | |
| 17 | + | "input": 11455, | |
| 18 | + | "output": 8400, | |
| 19 | + | "cacheRead": 351314, | |
| 20 | + | "cacheWrite": 48072 | |
| 21 | + | } | |
| 22 | + | }, | |
| 23 | + | { | |
| 24 | + | "id": "add-search-filter", | |
| 25 | + | "kind": "implement", | |
| 26 | + | "change": { | |
| 27 | + | "files": 6, | |
| 28 | + | "lines": 240, | |
| 29 | + | "sensitive": [] | |
| 30 | + | }, | |
| 31 | + | "merged": true, | |
| 32 | + | "tokens": { | |
| 33 | + | "input": 80004, | |
| 34 | + | "output": 31200, | |
| 35 | + | "cacheRead": 2453474, | |
| 36 | + | "cacheWrite": 118380 | |
| 37 | + | } | |
| 38 | + | }, | |
| 39 | + | { | |
| 40 | + | "id": "rework-billing-ledger", | |
| 41 | + | "kind": "implement", | |
| 42 | + | "change": { | |
| 43 | + | "files": 14, | |
| 44 | + | "lines": 900, | |
| 45 | + | "sensitive": [] | |
| 46 | + | }, | |
| 47 | + | "failsOn": [ | |
| 48 | + | "small", | |
| 49 | + | "large" | |
| 50 | + | ], | |
| 51 | + | "merged": true, | |
| 52 | + | "tokens": { | |
| 53 | + | "input": 184590, | |
| 54 | + | "output": 48000, | |
| 55 | + | "cacheRead": 5660760, | |
| 56 | + | "cacheWrite": 178800 | |
| 57 | + | } | |
| 58 | + | }, | |
| 59 | + | { | |
| 60 | + | "id": "split-auth-service", | |
| 61 | + | "kind": "implement", | |
| 62 | + | "change": { | |
| 63 | + | "files": 20, | |
| 64 | + | "lines": 1500, | |
| 65 | + | "sensitive": [] | |
| 66 | + | }, | |
| 67 | + | "labels": [ | |
| 68 | + | "architecture" | |
| 69 | + | ], | |
| 70 | + | "failsOn": [ | |
| 71 | + | "small", | |
| 72 | + | "large" | |
| 73 | + | ], | |
| 74 | + | "merged": true, | |
| 75 | + | "tokens": { | |
| 76 | + | "input": 278415, | |
| 77 | + | "output": 60000, | |
| 78 | + | "cacheRead": 8538060, | |
| 79 | + | "cacheWrite": 180000 | |
| 80 | + | } | |
| 81 | + | }, | |
| 82 | + | { | |
| 83 | + | "id": "fix-failing-check", | |
| 84 | + | "kind": "revise", | |
| 85 | + | "change": { | |
| 86 | + | "files": 2, | |
| 87 | + | "lines": 30, | |
| 88 | + | "sensitive": [] | |
| 89 | + | }, | |
| 90 | + | "merged": false, | |
| 91 | + | "tokens": { | |
| 92 | + | "input": 19563, | |
| 93 | + | "output": 11900, | |
| 94 | + | "cacheRead": 599950, | |
| 95 | + | "cacheWrite": 60860 | |
| 96 | + | } | |
| 97 | + | }, | |
| 98 | + | { | |
| 99 | + | "id": "what-does-this-flag-do", | |
| 100 | + | "kind": "answer", | |
| 101 | + | "merged": false, | |
| 102 | + | "tokens": { | |
| 103 | + | "input": 7560, | |
| 104 | + | "output": 3600, | |
| 105 | + | "cacheRead": 231840, | |
| 106 | + | "cacheWrite": 40500 | |
| 107 | + | } | |
| 108 | + | }, | |
| 109 | + | { | |
| 110 | + | "id": "why-is-this-slow", | |
| 111 | + | "kind": "answer", | |
| 112 | + | "failsOn": [ | |
| 113 | + | "small" | |
| 114 | + | ], | |
| 115 | + | "merged": false, | |
| 116 | + | "tokens": { | |
| 117 | + | "input": 19380, | |
| 118 | + | "output": 6800, | |
| 119 | + | "cacheRead": 594320, | |
| 120 | + | "cacheWrite": 60500 | |
| 121 | + | } | |
| 122 | + | }, | |
| 123 | + | { | |
| 124 | + | "id": "review-small-change", | |
| 125 | + | "kind": "review", | |
| 126 | + | "change": { | |
| 127 | + | "files": 3, | |
| 128 | + | "lines": 80, | |
| 129 | + | "sensitive": [] | |
| 130 | + | }, | |
| 131 | + | "merged": false, | |
| 132 | + | "tokens": { | |
| 133 | + | "input": 25626, | |
| 134 | + | "output": 9000, | |
| 135 | + | "cacheRead": 785864, | |
| 136 | + | "cacheWrite": 68960 | |
| 137 | + | } | |
| 138 | + | }, | |
| 139 | + | { | |
| 140 | + | "id": "review-ci-change", | |
| 141 | + | "kind": "review", | |
| 142 | + | "change": { | |
| 143 | + | "files": 2, | |
| 144 | + | "lines": 40, | |
| 145 | + | "sensitive": [ | |
| 146 | + | "CI workflows" | |
| 147 | + | ] | |
| 148 | + | }, | |
| 149 | + | "merged": false, | |
| 150 | + | "tokens": { | |
| 151 | + | "input": 21454, | |
| 152 | + | "output": 8100, | |
| 153 | + | "cacheRead": 657928, | |
| 154 | + | "cacheWrite": 63480 | |
| 155 | + | } | |
| 156 | + | }, | |
| 157 | + | { | |
| 158 | + | "id": "review-medium-change", | |
| 159 | + | "kind": "review", | |
| 160 | + | "change": { | |
| 161 | + | "files": 12, | |
| 162 | + | "lines": 450, | |
| 163 | + | "sensitive": [] | |
| 164 | + | }, | |
| 165 | + | "merged": false, | |
| 166 | + | "tokens": { | |
| 167 | + | "input": 65943, | |
| 168 | + | "output": 15300, | |
| 169 | + | "cacheRead": 2022252, | |
| 170 | + | "cacheWrite": 108400 | |
| 171 | + | } | |
| 172 | + | }, | |
| 173 | + | { | |
| 174 | + | "id": "review-large-migration", | |
| 175 | + | "kind": "review", | |
| 176 | + | "change": { | |
| 177 | + | "files": 80, | |
| 178 | + | "lines": 4200, | |
| 179 | + | "sensitive": [] | |
| 180 | + | }, | |
| 181 | + | "merged": false, | |
| 182 | + | "tokens": { | |
| 183 | + | "input": 355590, | |
| 184 | + | "output": 36000, | |
| 185 | + | "cacheRead": 10904760, | |
| 186 | + | "cacheWrite": 180000 | |
| 187 | + | } | |
| 188 | + | }, | |
| 189 | + | { | |
| 190 | + | "id": "catch-up-with-conflict", | |
| 191 | + | "kind": "update", | |
| 192 | + | "merged": false, | |
| 193 | + | "tokens": { | |
| 194 | + | "input": 21583, | |
| 195 | + | "output": 9000, | |
| 196 | + | "cacheRead": 661903, | |
| 197 | + | "cacheWrite": 63720 | |
| 198 | + | } | |
| 199 | + | }, | |
| 200 | + | { | |
| 201 | + | "id": "plan-checkout-redesign", | |
| 202 | + | "kind": "plan", | |
| 203 | + | "merged": false, | |
| 204 | + | "tokens": { | |
| 205 | + | "input": 54480, | |
| 206 | + | "output": 28800, | |
| 207 | + | "cacheRead": 1670720, | |
| 208 | + | "cacheWrite": 98000 | |
| 209 | + | } | |
| 210 | + | } | |
| 211 | + | ] | |
| 212 | + | } |
| 1 | + | import assert from "node:assert/strict"; | |
| 2 | + | import { readFileSync } from "node:fs"; | |
| 3 | + | import { test } from "node:test"; | |
| 4 | + | ||
| 5 | + | import { DEFAULT_ROUTING } from "../../services/runner/src/model-env.ts"; | |
| 6 | + | import { autoNext, compare, configuredRouting, costOn, play, previousNext, previousTier, tasksFromBilling } from "./routing-savings.mjs"; | |
| 7 | + | ||
| 8 | + | const routing = DEFAULT_ROUTING; | |
| 9 | + | const million = { input: 1_000_000, output: 0, cacheRead: 0, cacheWrite: 0 }; | |
| 10 | + | ||
| 11 | + | test("tokens are priced at each tier's list price, the fast tier's scaled for extra turns", () => { | |
| 12 | + | assert.equal(costOn("small", million, routing), 1); | |
| 13 | + | assert.equal(costOn("large", million, routing), 2); | |
| 14 | + | assert.equal(costOn("frontier", million, routing), 4); | |
| 15 | + | assert.equal(costOn("small", million, routing, 1.5), 1.5); | |
| 16 | + | assert.equal(costOn("large", million, routing, 1.5), 2); | |
| 17 | + | const mixed = { input: 0, output: 1_000_000, cacheRead: 10_000_000, cacheWrite: 1_000_000 }; | |
| 18 | + | // $10 of output, $2 of cache reads, $2.50 of cache writes. | |
| 19 | + | assert.equal(costOn("large", mixed, routing), 14.5); | |
| 20 | + | const unpriced = { ...routing, tiers: { ...routing.tiers, large: { modelName: "X", model: "x" } } }; | |
| 21 | + | assert.equal(costOn("large", million, unpriced), null); | |
| 22 | + | }); | |
| 23 | + | ||
| 24 | + | test("a failed attempt is paid for, and Auto retries one tier up until the most capable", () => { | |
| 25 | + | const task = { id: "t", kind: "implement", failsOn: ["small", "large"], tokens: million }; | |
| 26 | + | assert.deepEqual(play(task, "small", autoNext(routing), routing, 1), { attempts: ["small", "large", "frontier"], cost: 7, done: true }); | |
| 27 | + | // Routing before Auto stayed on the large tier and asked a person after two failures. | |
| 28 | + | assert.deepEqual(play(task, "large", previousNext, routing, 1), { attempts: ["large", "large"], cost: 4, done: false }); | |
| 29 | + | // Nothing past the most capable model. | |
| 30 | + | const hopeless = { ...task, failsOn: ["small", "large", "frontier"] }; | |
| 31 | + | assert.equal(play(hopeless, "small", autoNext(routing), routing, 1).done, false); | |
| 32 | + | }); | |
| 33 | + | ||
| 34 | + | test("routing before Auto is replayed as it was", () => { | |
| 35 | + | assert.equal(previousTier({ kind: "plan" }), "small"); | |
| 36 | + | assert.equal(previousTier({ kind: "update" }), "small"); | |
| 37 | + | assert.equal(previousTier({ kind: "answer" }), "large"); | |
| 38 | + | assert.equal(previousTier({ kind: "review", change: { files: 3, lines: 80, sensitive: [] } }), "small"); | |
| 39 | + | assert.equal(previousTier({ kind: "review", change: { files: 3, lines: 80, sensitive: [] }, labels: ["Security"] }), "large"); | |
| 40 | + | assert.equal(previousTier({ kind: "review" }), "large"); | |
| 41 | + | }); | |
| 42 | + | ||
| 43 | + | test("the comparison totals each policy and its cost per merged change", () => { | |
| 44 | + | const tasks = [ | |
| 45 | + | { id: "a", kind: "update", tokens: million, merged: false }, | |
| 46 | + | { id: "b", kind: "implement", labels: ["docs"], tokens: million, merged: true }, | |
| 47 | + | ]; | |
| 48 | + | const { totals, savings, rows } = compare(tasks, routing, { smallTurns: 1 }); | |
| 49 | + | // Auto: both fast. Before: catching up fast, the change standard. | |
| 50 | + | assert.equal(totals.auto.cost, 2); | |
| 51 | + | assert.equal(totals.before.cost, 3); | |
| 52 | + | assert.equal(totals.frontier.cost, 8); | |
| 53 | + | assert.equal(totals.auto.perMerged, 2); | |
| 54 | + | assert.ok(Math.abs(savings.vsBefore - 1 / 3) < 1e-9); | |
| 55 | + | assert.equal(savings.vsFrontier, 0.75); | |
| 56 | + | assert.match(rows[1].reason, /labelled docs/); | |
| 57 | + | }); | |
| 58 | + | ||
| 59 | + | test("the sample is well formed and every task is priced", () => { | |
| 60 | + | const sample = JSON.parse(readFileSync(new URL("./routing-savings.sample.json", import.meta.url), "utf8")); | |
| 61 | + | assert.ok(sample.tasks.length >= 10); | |
| 62 | + | for (const task of sample.tasks) { | |
| 63 | + | assert.ok(["implement", "revise", "answer", "review", "update", "plan"].includes(task.kind), task.id); | |
| 64 | + | assert.ok(costOn("large", task.tokens, routing) > 0, task.id); | |
| 65 | + | } | |
| 66 | + | const { totals } = compare(sample.tasks, routing); | |
| 67 | + | assert.ok(totals.auto.cost > 0 && totals.before.cost > 0 && totals.frontier.cost > 0); | |
| 68 | + | }); | |
| 69 | + | ||
| 70 | + | test("billing's runs become tasks; reviews keep the tier they ran on", () => { | |
| 71 | + | const tasks = tasksFromBilling([ | |
| 72 | + | { id: "run_1", task: "review", tier: "small", input: 10, output: 5, cache_read: 0, cache_write: 0 }, | |
| 73 | + | { id: "run_2", task: "implement", tier: "large", input: 0, output: 0, cache_read: 0, cache_write: 0 }, | |
| 74 | + | { id: "run_3", task: "plan", tier: null, input: 1, output: 1, cache_read: 1, cache_write: 1 }, | |
| 75 | + | ]); | |
| 76 | + | assert.deepEqual( | |
| 77 | + | tasks.map((t) => [t.id, t.kind, t.keepTier]), | |
| 78 | + | [ | |
| 79 | + | ["run_1", "review", "small"], | |
| 80 | + | ["run_3", "plan", null], | |
| 81 | + | ], | |
| 82 | + | ); | |
| 83 | + | }); | |
| 84 | + | ||
| 85 | + | test("the routing estimated is the runner's configuration", () => { | |
| 86 | + | const text = readFileSync(new URL("../../services/runner/wrangler.jsonc", import.meta.url), "utf8"); | |
| 87 | + | assert.deepEqual(configuredRouting(text), DEFAULT_ROUTING); | |
| 88 | + | assert.deepEqual(configuredRouting("not jsonc {"), DEFAULT_ROUTING); | |
| 89 | + | }); |
| 44 | 44 | ||
| 45 | 45 | const small: ChangeSize = { files: 3, lines: 80, sensitive: [] }; | |
| 46 | 46 | ||
| 47 | − | test("Auto starts each kind of job on its tier: fast for catching up and answering, standard for changes, most capable for plans", () => { | |
| 47 | + | test("Auto starts each kind of job on its tier: fast for catching up and answering, standard for changes and plans", () => { | |
| 48 | 48 | assert.equal(chooseTier("update", {}, routes), "small"); | |
| 49 | 49 | assert.equal(chooseTier("answer", {}, routes), "small"); | |
| 50 | 50 | assert.equal(chooseTier("implement", {}, routes), "large"); | |
| 51 | 51 | assert.equal(chooseTier("revise", {}, routes), "large"); | |
| 52 | 52 | assert.equal(chooseTier("implement", { change: small }, routes), "large"); | |
| 53 | − | assert.equal(chooseTier("plan", {}, routes), "frontier"); | |
| 53 | + | assert.equal(chooseTier("plan", {}, routes), "large"); | |
| 54 | + | assert.equal(chooseTier("plan", { labels: ["architecture"] }, routes), "frontier"); | |
| 54 | 55 | }); | |
| 55 | 56 | ||
| 56 | 57 | test("revising and answering go to the workspace's implement route and bill", () => { | |
| ⋯ | |||
| 84 | 85 | assert.equal(chooseTier("implement", { labels: ["docs"] }, routes), "small"); | |
| 85 | 86 | assert.equal(chooseTier("answer", { labels: ["typo"] }, routes), "small"); | |
| 86 | 87 | // A small label never takes a review or a plan down. | |
| 87 | − | assert.equal(chooseTier("plan", { labels: ["docs"] }, routes), "frontier"); | |
| 88 | + | assert.equal(chooseTier("plan", { labels: ["docs"] }, routes), "large"); | |
| 88 | 89 | // Security outranks documentation. | |
| 89 | 90 | assert.equal(chooseTier("implement", { labels: ["docs", "security"] }, routes), "large"); | |
| 90 | 91 | }); | |
| ⋯ | |||
| 121 | 122 | // Too few runs to tell, or not quite enough of them finished: no change. | |
| 122 | 123 | assert.equal(chooseTier("implement", { history: ok("small", 4) }, routes), "large"); | |
| 123 | 124 | assert.equal(chooseTier("implement", { history: [...ok("small", 8), ...bad("small", 2)] }, routes), "large"); | |
| 124 | − | // Plans step down from the most capable model the same way. | |
| 125 | − | assert.equal(chooseTier("plan", { history: ok("large", 6) }, routes), "large"); | |
| 125 | + | // Plans step down the same way. | |
| 126 | + | assert.equal(chooseTier("plan", { history: ok("small", 6) }, routes), "small"); | |
| 126 | 127 | }); | |
| 127 | 128 | ||
| 128 | 129 | test("learning never steps down sensitive or labelled work, nor a retry", () => { | |
| ⋯ | |||
| 198 | 199 | assert.deepEqual(parsed.tiers.frontier, DEFAULT_ROUTING.tiers.frontier); | |
| 199 | 200 | assert.equal(chooseTier("update", {}, parsed), "large"); | |
| 200 | 201 | // A rule that names no tier keeps the default. | |
| 201 | − | assert.equal(chooseTier("plan", {}, parsed), "frontier"); | |
| 202 | + | assert.equal(chooseTier("plan", {}, parsed), "large"); | |
| 202 | 203 | assert.equal(parsed.tasks.answer, "change"); | |
| 203 | 204 | assert.equal(chooseTier("review", { change: { ...small, lines: 51 } }, parsed), "large"); | |
| 204 | 205 | assert.equal(chooseTier("review", { change: { ...small, files: 10, lines: 50 } }, parsed), "small"); | |
| 143 | 143 | price: { input: 4, output: 20, cacheRead: 0.2, cacheWrite: 5 }, | |
| 144 | 144 | }, | |
| 145 | 145 | }, | |
| 146 | − | tasks: { implement: "large", revise: "large", answer: "small", review: "change", update: "small", plan: "frontier" }, | |
| 146 | + | tasks: { implement: "large", revise: "large", answer: "small", review: "change", update: "small", plan: "large" }, | |
| 147 | 147 | smallChange: { files: 10, lines: 200 }, | |
| 148 | 148 | largeChange: { files: 60, lines: 3000 }, | |
| 149 | 149 | largeLabels: ["security"], |
| 87 | 87 | // "smallLabels" move work by its issue's labels. A failed attempt goes one | |
| 88 | 88 | // tier up and "frontierAfter" failures in a row to frontier; "learning" | |
| 89 | 89 | // steps work down or up by the repository's own recent runs of the kind. | |
| 90 | − | "AGENT_ROUTING": "{\"tiers\":{\"small\":{\"modelName\":\"Claude Haiku 4.5\",\"model\":\"claude-haiku-4-5-20251001\",\"price\":{\"input\":1,\"output\":5,\"cacheRead\":0.1,\"cacheWrite\":1.25}},\"large\":{\"modelName\":\"Claude Sonnet 5.5\",\"model\":\"claude-sonnet-5-5\",\"price\":{\"input\":2,\"output\":10,\"cacheRead\":0.2,\"cacheWrite\":2.5}},\"frontier\":{\"modelName\":\"Claude Opus 5.5\",\"model\":\"claude-opus-5-5\",\"price\":{\"input\":4,\"output\":20,\"cacheRead\":0.2,\"cacheWrite\":5}}},\"tasks\":{\"implement\":\"large\",\"revise\":\"large\",\"answer\":\"small\",\"review\":\"change\",\"update\":\"small\",\"plan\":\"frontier\"},\"smallChange\":{\"files\":10,\"lines\":200},\"largeChange\":{\"files\":60,\"lines\":3000},\"largeLabels\":[\"security\"],\"frontierLabels\":[\"architecture\"],\"smallLabels\":[\"documentation\",\"docs\",\"typo\"],\"frontierAfter\":2,\"learning\":{\"window\":20,\"minRuns\":5,\"stepDownAt\":0.9,\"stepUpAt\":0.5}}", | |
| 90 | + | "AGENT_ROUTING": "{\"tiers\":{\"small\":{\"modelName\":\"Claude Haiku 4.5\",\"model\":\"claude-haiku-4-5-20251001\",\"price\":{\"input\":1,\"output\":5,\"cacheRead\":0.1,\"cacheWrite\":1.25}},\"large\":{\"modelName\":\"Claude Sonnet 5.5\",\"model\":\"claude-sonnet-5-5\",\"price\":{\"input\":2,\"output\":10,\"cacheRead\":0.2,\"cacheWrite\":2.5}},\"frontier\":{\"modelName\":\"Claude Opus 5.5\",\"model\":\"claude-opus-5-5\",\"price\":{\"input\":4,\"output\":20,\"cacheRead\":0.2,\"cacheWrite\":5}}},\"tasks\":{\"implement\":\"large\",\"revise\":\"large\",\"answer\":\"small\",\"review\":\"change\",\"update\":\"small\",\"plan\":\"large\"},\"smallChange\":{\"files\":10,\"lines\":200},\"largeChange\":{\"files\":60,\"lines\":3000},\"largeLabels\":[\"security\"],\"frontierLabels\":[\"architecture\"],\"smallLabels\":[\"documentation\",\"docs\",\"typo\"],\"frontierAfter\":2,\"learning\":{\"window\":20,\"minRuns\":5,\"stepDownAt\":0.9,\"stepUpAt\":0.5}}", | |
| 91 | 91 | // Where sandboxes send model requests, with a token for their run. | |
| 92 | 92 | // The proxy holds the keys: g1t's gateway's, or the workspace's own. | |
| 93 | 93 | "MODELS_URL": "https://models.g1t.sh", |