| 1 | #!/usr/bin/env node |
| 2 | // What Auto (the agent model router, services/runner/src/model-env.ts) |
| 3 | // saves against routing as it was before, and against running everything |
| 4 | // on the most capable model: an estimate, offline, that never calls a |
| 5 | // model. |
| 6 | // |
| 7 | // node scripts/ops/routing-savings.mjs # the sample in routing-savings.sample.json |
| 8 | // node scripts/ops/routing-savings.mjs --json # the same, as JSON |
| 9 | // node scripts/ops/routing-savings.mjs --live 30 # g1t's own runs of the last 30 days, from billing |
| 10 | // node scripts/ops/routing-savings.mjs --small-turns 1.5 # assume the fast model takes 50% more tokens |
| 11 | // |
| 12 | // How it estimates. Each task's tokens (input, output, cache reads and |
| 13 | // writes) are priced at each tier's list price from the routing's |
| 14 | // catalogue (AGENT_ROUTING in services/runner/wrangler.jsonc). Prices are |
| 15 | // per token, so the same work on a cheaper tier costs its price ratio, |
| 16 | // except that a smaller model may take more turns: its tokens are scaled |
| 17 | // by --small-turns (1.25 by default). A task the sample says a tier |
| 18 | // cannot do is charged that failed attempt and then the retry Auto makes |
| 19 | // one tier up (two failures in a row go to the most capable), so savings |
| 20 | // are net of escalation. Cost per merged change is the policy's cost over |
| 21 | // the tasks that end merged. |
| 22 | // |
| 23 | // --live reads, read-only, the runs billing settled and the tokens the |
| 24 | // model proxy counted for them (g1t-billing: runs, token_usage), through |
| 25 | // Wrangler as you are logged in, or CLOUDFLARE_D1_TOKEN. Live runs carry no |
| 26 | // change size, labels or outcome, so a review keeps the tier it ran on and |
| 27 | // nothing is escalated: a first look, not a verdict. |
| 28 | |
| 29 | import { readFileSync } from "node:fs"; |
| 30 | import { join } from "node:path"; |
| 31 | |
| 32 | import { exec, jsonFrom, wranglerEnv } from "../deploy/cloudflare.mjs"; |
| 33 | import { ROOT, parseJsonc } from "../deploy/stack.mjs"; |
| 34 | import { DEFAULT_ROUTING, TIERS, parseRouting, route } from "../../services/runner/src/model-env.ts"; |
| 35 | |
| 36 | const WRANGLER = join(ROOT, "node_modules/wrangler/bin/wrangler.js"); |
| 37 | const DATABASE = "g1t-billing"; |
| 38 | const SAMPLE = join(ROOT, "scripts/ops/routing-savings.sample.json"); |
| 39 | |
| 40 | /** The routing the next deploy runs: AGENT_ROUTING in the runner's wrangler.jsonc. */ |
| 41 | export function configuredRouting(wranglerText) { |
| 42 | try { |
| 43 | return parseRouting(parseJsonc(wranglerText).vars?.AGENT_ROUTING); |
| 44 | } catch { |
| 45 | return DEFAULT_ROUTING; |
| 46 | } |
| 47 | } |
| 48 | |
| 49 | /** What `tokens` cost on `tier`, in dollars, with the fast tier's extra turns. */ |
| 50 | export function costOn(tier, tokens, routing, smallTurns = 1) { |
| 51 | const price = routing.tiers[tier].price; |
| 52 | if (!price) return null; |
| 53 | const scale = tier === "small" ? smallTurns : 1; |
| 54 | const dollars = |
| 55 | (tokens.input ?? 0) * price.input + |
| 56 | (tokens.output ?? 0) * price.output + |
| 57 | (tokens.cacheRead ?? 0) * price.cacheRead + |
| 58 | (tokens.cacheWrite ?? 0) * price.cacheWrite; |
| 59 | return (dollars * scale) / 1_000_000; |
| 60 | } |
| 61 | |
| 62 | /** |
| 63 | * The tier routing chose before Auto: changes, revisions and answers on |
| 64 | * the large tier, plans and catching up on the small, a review small for |
| 65 | * a small change that touches nothing sensitive, and a retry large. |
| 66 | */ |
| 67 | export function previousTier(task) { |
| 68 | const kind = task.kind; |
| 69 | if (kind === "plan" || kind === "update") return "small"; |
| 70 | if (kind !== "review") return "large"; |
| 71 | const change = task.change; |
| 72 | if (!change || !change.files) return "large"; |
| 73 | const security = (task.labels ?? []).some((label) => label.toLowerCase() === "security"); |
| 74 | const small = !security && (change.sensitive ?? []).length === 0 && change.files <= 10 && change.lines <= 200; |
| 75 | return small ? "small" : "large"; |
| 76 | } |
| 77 | |
| 78 | /** Auto's retry: one tier up, the most capable after `frontierAfter` failures in a row. */ |
| 79 | export function autoNext(routing) { |
| 80 | return (tier, failures) => |
| 81 | tier === "frontier" ? null : failures >= routing.frontierAfter ? "frontier" : TIERS[TIERS.indexOf(tier) + 1]; |
| 82 | } |
| 83 | |
| 84 | /** |
| 85 | * Routing before Auto: a retry ran on the large tier, and after two |
| 86 | * failures there a person was asked. |
| 87 | */ |
| 88 | export function previousNext(tier, failures) { |
| 89 | return failures >= 2 && tier === "large" ? null : "large"; |
| 90 | } |
| 91 | |
| 92 | /** |
| 93 | * What one task costs under a policy that starts it on `first` and, after |
| 94 | * each failure, retries on what `next(tier, failures)` says (null: it |
| 95 | * stops, for a person). `failsOn` lists the tiers the task fails on. |
| 96 | * Returns the attempts, the cost and whether it ended done. |
| 97 | */ |
| 98 | export function play(task, first, next, routing, smallTurns) { |
| 99 | const fails = new Set(task.failsOn ?? []); |
| 100 | const attempts = []; |
| 101 | let tier = first; |
| 102 | let failures = 0; |
| 103 | for (;;) { |
| 104 | attempts.push(tier); |
| 105 | if (!fails.has(tier)) return { attempts, cost: sum(attempts, task, routing, smallTurns), done: true }; |
| 106 | failures += 1; |
| 107 | const then = attempts.length < 5 ? next(tier, failures) : null; |
| 108 | if (!then) return { attempts, cost: sum(attempts, task, routing, smallTurns), done: false }; |
| 109 | tier = then; |
| 110 | } |
| 111 | } |
| 112 | |
| 113 | function sum(attempts, task, routing, smallTurns) { |
| 114 | return attempts.reduce((total, tier) => total + (costOn(tier, task.tokens, routing, smallTurns) ?? 0), 0); |
| 115 | } |
| 116 | |
| 117 | /** Auto's first tier for a task, and why, as the runner would route it. */ |
| 118 | export function autoFirst(task, routing) { |
| 119 | if (task.kind === "review" && task.keepTier) return { tier: task.keepTier, reason: "kept: the change's size is not on record" }; |
| 120 | return route(task.kind, { change: task.change ?? null, labels: task.labels ?? [] }, routing); |
| 121 | } |
| 122 | |
| 123 | /** Every policy's cost over the tasks: Auto, routing before it, and the most capable for all. */ |
| 124 | export function compare(tasks, routing, { smallTurns = 1.25, escalate = true } = {}) { |
| 125 | const stop = () => null; |
| 126 | const policies = { |
| 127 | auto: [(task) => autoFirst(task, routing).tier, escalate ? autoNext(routing) : stop], |
| 128 | before: [(task) => previousTier(task), escalate ? previousNext : stop], |
| 129 | frontier: [() => "frontier", stop], |
| 130 | }; |
| 131 | const rows = tasks.map((task) => { |
| 132 | const out = { id: task.id, kind: task.kind, merged: task.merged !== false }; |
| 133 | for (const [name, [first, next]] of Object.entries(policies)) { |
| 134 | const played = play(task, first(task), next, routing, smallTurns); |
| 135 | out[name] = { tiers: played.attempts, cost: played.cost, done: played.done }; |
| 136 | } |
| 137 | out.reason = autoFirst(task, routing).reason; |
| 138 | return out; |
| 139 | }); |
| 140 | const totals = {}; |
| 141 | for (const name of Object.keys(policies)) { |
| 142 | const cost = rows.reduce((total, row) => total + row[name].cost, 0); |
| 143 | const merged = rows.filter((row) => row.merged && row[name].done).length; |
| 144 | totals[name] = { cost, merged, perMerged: merged ? cost / merged : null }; |
| 145 | } |
| 146 | const saving = (from) => (totals[from].cost > 0 ? 1 - totals.auto.cost / totals[from].cost : 0); |
| 147 | return { rows, totals, savings: { vsBefore: saving("before"), vsFrontier: saving("frontier") } }; |
| 148 | } |
| 149 | |
| 150 | /** Billing's runs and their tokens, as tasks. Reviews keep the tier they ran on. */ |
| 151 | export function tasksFromBilling(rows) { |
| 152 | return rows |
| 153 | .filter((row) => (row.input ?? 0) + (row.output ?? 0) + (row.cache_read ?? 0) + (row.cache_write ?? 0) > 0) |
| 154 | .map((row) => ({ |
| 155 | id: row.id, |
| 156 | kind: ["implement", "review", "update", "plan"].includes(row.task) ? row.task : "implement", |
| 157 | keepTier: row.task === "review" && TIERS.includes(row.tier) ? row.tier : null, |
| 158 | tokens: { input: row.input ?? 0, output: row.output ?? 0, cacheRead: row.cache_read ?? 0, cacheWrite: row.cache_write ?? 0 }, |
| 159 | merged: true, |
| 160 | })); |
| 161 | } |
| 162 | |
| 163 | async function d1(sql) { |
| 164 | const env = wranglerEnv(); |
| 165 | if (process.env.CLOUDFLARE_D1_TOKEN) env.CLOUDFLARE_API_TOKEN = process.env.CLOUDFLARE_D1_TOKEN; |
| 166 | const { code, out } = await exec(process.execPath, [WRANGLER, "d1", "execute", DATABASE, "--remote", "--json", "--command", sql], { |
| 167 | env, |
| 168 | }); |
| 169 | if (code !== 0) throw new Error(`wrangler d1 execute failed:\n${out.slice(-800)}`); |
| 170 | return jsonFrom(out)[0]?.results ?? []; |
| 171 | } |
| 172 | |
| 173 | function liveSql(days) { |
| 174 | const since = new Date(Date.now() - days * 86_400_000).toISOString(); |
| 175 | return `SELECT r.id, r.task, r.tier, r.model, |
| 176 | SUM(t.input) AS input, SUM(t.output) AS output, SUM(t.cache_read) AS cache_read, SUM(t.cache_write) AS cache_write |
| 177 | FROM runs r JOIN token_usage t ON t.session = r.session_id AND t.workspace = r.workspace |
| 178 | WHERE r.created_at >= '${since}' AND r.session_id IS NOT NULL |
| 179 | GROUP BY r.id LIMIT 5000`; |
| 180 | } |
| 181 | |
| 182 | function dollars(n) { |
| 183 | return n == null ? "—" : `$${n.toFixed(n < 1 ? 4 : 2)}`; |
| 184 | } |
| 185 | |
| 186 | function print(result) { |
| 187 | const { rows, totals, savings } = result; |
| 188 | console.log("task kind auto before most capable"); |
| 189 | for (const row of rows) { |
| 190 | const cell = (p) => `${dollars(row[p].cost)} ${row[p].tiers.join(">")}${row[p].done ? "" : " (failed)"}`; |
| 191 | console.log(`${row.id.padEnd(26)} ${row.kind.padEnd(10)} ${cell("auto").padEnd(20)} ${cell("before").padEnd(14)} ${cell("frontier")}`); |
| 192 | } |
| 193 | console.log(""); |
| 194 | for (const [name, total] of Object.entries(totals)) { |
| 195 | console.log(`${name.padEnd(9)} ${dollars(total.cost).padStart(10)} merged ${total.merged} per merged change ${dollars(total.perMerged)}`); |
| 196 | } |
| 197 | console.log(""); |
| 198 | const say = (share) => `${Math.abs(share * 100).toFixed(1)}% ${share >= 0 ? "less" : "more"}`; |
| 199 | console.log(`Auto against routing before it: ${say(savings.vsBefore)}`); |
| 200 | console.log(`Auto against the most capable model for everything: ${say(savings.vsFrontier)}`); |
| 201 | } |
| 202 | |
| 203 | async function main() { |
| 204 | const args = process.argv.slice(2); |
| 205 | const flag = (name) => { |
| 206 | const at = args.indexOf(name); |
| 207 | return at >= 0 ? args[at + 1] : undefined; |
| 208 | }; |
| 209 | const routing = configuredRouting(readFileSync(join(ROOT, "services/runner/wrangler.jsonc"), "utf8")); |
| 210 | const smallTurns = Number(flag("--small-turns") ?? 1.25); |
| 211 | let tasks; |
| 212 | let escalate = true; |
| 213 | if (args.includes("--live")) { |
| 214 | const days = Number(flag("--live") ?? 30) || 30; |
| 215 | tasks = tasksFromBilling(await d1(liveSql(days))); |
| 216 | escalate = false; |
| 217 | } else { |
| 218 | tasks = JSON.parse(readFileSync(SAMPLE, "utf8")).tasks; |
| 219 | } |
| 220 | const result = compare(tasks, routing, { smallTurns, escalate }); |
| 221 | if (args.includes("--json")) console.log(JSON.stringify(result, null, 2)); |
| 222 | else print(result); |
| 223 | } |
| 224 | |
| 225 | if (process.argv[1]?.replaceAll("\\", "/").endsWith("scripts/ops/routing-savings.mjs")) { |
| 226 | main().catch((error) => { |
| 227 | console.error(String(error?.message ?? error)); |
| 228 | process.exit(1); |
| 229 | }); |
| 230 | } |