Pick any line to see why it is the way it is: the commit, the pull request and issue it came from, and what the agent was thinking.
| Merge branch 'model-routing' | 1 | import assert from "node:assert/strict"; |
| 2 | import { readFileSync } from "node:fs"; | |
| 3 | import { test } from "node:test"; | |
| 4 | ||
| 5 | import { DEFAULT_ROUTING } from "../../services/runner/src/model-env.ts"; | |
| 6 | import { autoNext, compare, configuredRouting, costOn, play, previousNext, previousTier, tasksFromBilling } from "./routing-savings.mjs"; | |
| 7 | ||
| 8 | const routing = DEFAULT_ROUTING; | |
| Merge membership: owners, org roles, GitHub's repo roles, privileges, 2FA | 9 | // The arithmetic below runs on fixed prices, so a new model in a tier |
| 10 | // (Haiku 5.5 replaced Haiku 4.5 on the fast tier) doesn't change it. | |
| 11 | const priced = (input, output, cacheRead, cacheWrite) => ({ input, output, cacheRead, cacheWrite }); | |
| 12 | const fixed = { | |
| 13 | ...DEFAULT_ROUTING, | |
| 14 | tasks: { ...DEFAULT_ROUTING.tasks, plan: "large" }, | |
| 15 | tiers: { | |
| 16 | small: { modelName: "Fast", model: "fast", price: priced(1, 5, 0.1, 1.25) }, | |
| 17 | large: { modelName: "Standard", model: "standard", price: priced(2, 10, 0.2, 2.5) }, | |
| 18 | frontier: { modelName: "Most capable", model: "capable", price: priced(4, 20, 0.2, 5) }, | |
| 19 | }, | |
| 20 | }; | |
| Merge branch 'model-routing' | 21 | const million = { input: 1_000_000, output: 0, cacheRead: 0, cacheWrite: 0 }; |
| 22 | ||
| 23 | test("tokens are priced at each tier's list price, the fast tier's scaled for extra turns", () => { | |
| Merge membership: owners, org roles, GitHub's repo roles, privileges, 2FA | 24 | assert.equal(costOn("small", million, fixed), 1); |
| 25 | assert.equal(costOn("large", million, fixed), 2); | |
| 26 | assert.equal(costOn("frontier", million, fixed), 4); | |
| 27 | assert.equal(costOn("small", million, fixed, 1.5), 1.5); | |
| 28 | assert.equal(costOn("large", million, fixed, 1.5), 2); | |
| Merge branch 'model-routing' | 29 | const mixed = { input: 0, output: 1_000_000, cacheRead: 10_000_000, cacheWrite: 1_000_000 }; |
| 30 | // $10 of output, $2 of cache reads, $2.50 of cache writes. | |
| Merge membership: owners, org roles, GitHub's repo roles, privileges, 2FA | 31 | assert.equal(costOn("large", mixed, fixed), 14.5); |
| 32 | const unpriced = { ...fixed, tiers: { ...fixed.tiers, large: { modelName: "X", model: "x" } } }; | |
| Merge branch 'model-routing' | 33 | assert.equal(costOn("large", million, unpriced), null); |
| 34 | }); | |
| 35 | ||
| 36 | test("a failed attempt is paid for, and Auto retries one tier up until the most capable", () => { | |
| 37 | const task = { id: "t", kind: "implement", failsOn: ["small", "large"], tokens: million }; | |
| Merge membership: owners, org roles, GitHub's repo roles, privileges, 2FA | 38 | assert.deepEqual(play(task, "small", autoNext(fixed), fixed, 1), { attempts: ["small", "large", "frontier"], cost: 7, done: true }); |
| Merge branch 'model-routing' | 39 | // Routing before Auto stayed on the large tier and asked a person after two failures. |
| Merge membership: owners, org roles, GitHub's repo roles, privileges, 2FA | 40 | assert.deepEqual(play(task, "large", previousNext, fixed, 1), { attempts: ["large", "large"], cost: 4, done: false }); |
| Merge branch 'model-routing' | 41 | // Nothing past the most capable model. |
| 42 | const hopeless = { ...task, failsOn: ["small", "large", "frontier"] }; | |
| Merge membership: owners, org roles, GitHub's repo roles, privileges, 2FA | 43 | assert.equal(play(hopeless, "small", autoNext(fixed), fixed, 1).done, false); |
| Merge branch 'model-routing' | 44 | }); |
| 45 | ||
| 46 | test("routing before Auto is replayed as it was", () => { | |
| 47 | assert.equal(previousTier({ kind: "plan" }), "small"); | |
| 48 | assert.equal(previousTier({ kind: "update" }), "small"); | |
| 49 | assert.equal(previousTier({ kind: "answer" }), "large"); | |
| 50 | assert.equal(previousTier({ kind: "review", change: { files: 3, lines: 80, sensitive: [] } }), "small"); | |
| 51 | assert.equal(previousTier({ kind: "review", change: { files: 3, lines: 80, sensitive: [] }, labels: ["Security"] }), "large"); | |
| 52 | assert.equal(previousTier({ kind: "review" }), "large"); | |
| 53 | }); | |
| 54 | ||
| 55 | test("the comparison totals each policy and its cost per merged change", () => { | |
| 56 | const tasks = [ | |
| 57 | { id: "a", kind: "update", tokens: million, merged: false }, | |
| 58 | { id: "b", kind: "implement", labels: ["docs"], tokens: million, merged: true }, | |
| 59 | ]; | |
| Merge membership: owners, org roles, GitHub's repo roles, privileges, 2FA | 60 | const { totals, savings, rows } = compare(tasks, fixed, { smallTurns: 1 }); |
| Merge branch 'model-routing' | 61 | // Auto: both fast. Before: catching up fast, the change standard. |
| 62 | assert.equal(totals.auto.cost, 2); | |
| 63 | assert.equal(totals.before.cost, 3); | |
| 64 | assert.equal(totals.frontier.cost, 8); | |
| 65 | assert.equal(totals.auto.perMerged, 2); | |
| 66 | assert.ok(Math.abs(savings.vsBefore - 1 / 3) < 1e-9); | |
| 67 | assert.equal(savings.vsFrontier, 0.75); | |
| 68 | assert.match(rows[1].reason, /labelled docs/); | |
| 69 | }); | |
| 70 | ||
| 71 | test("the sample is well formed and every task is priced", () => { | |
| 72 | const sample = JSON.parse(readFileSync(new URL("./routing-savings.sample.json", import.meta.url), "utf8")); | |
| 73 | assert.ok(sample.tasks.length >= 10); | |
| 74 | for (const task of sample.tasks) { | |
| 75 | assert.ok(["implement", "revise", "answer", "review", "update", "plan"].includes(task.kind), task.id); | |
| 76 | assert.ok(costOn("large", task.tokens, routing) > 0, task.id); | |
| 77 | } | |
| 78 | const { totals } = compare(sample.tasks, routing); | |
| 79 | assert.ok(totals.auto.cost > 0 && totals.before.cost > 0 && totals.frontier.cost > 0); | |
| 80 | }); | |
| 81 | ||
| 82 | test("billing's runs become tasks; reviews keep the tier they ran on", () => { | |
| 83 | const tasks = tasksFromBilling([ | |
| 84 | { id: "run_1", task: "review", tier: "small", input: 10, output: 5, cache_read: 0, cache_write: 0 }, | |
| 85 | { id: "run_2", task: "implement", tier: "large", input: 0, output: 0, cache_read: 0, cache_write: 0 }, | |
| 86 | { id: "run_3", task: "plan", tier: null, input: 1, output: 1, cache_read: 1, cache_write: 1 }, | |
| 87 | ]); | |
| 88 | assert.deepEqual( | |
| 89 | tasks.map((t) => [t.id, t.kind, t.keepTier]), | |
| 90 | [ | |
| 91 | ["run_1", "review", "small"], | |
| 92 | ["run_3", "plan", null], | |
| 93 | ], | |
| 94 | ); | |
| 95 | }); | |
| 96 | ||
| 97 | test("the routing estimated is the runner's configuration", () => { | |
| 98 | const text = readFileSync(new URL("../../services/runner/wrangler.jsonc", import.meta.url), "utf8"); | |
| 99 | assert.deepEqual(configuredRouting(text), DEFAULT_ROUTING); | |
| 100 | assert.deepEqual(configuredRouting("not jsonc {"), DEFAULT_ROUTING); | |
| 101 | }); |
This file's history is long; its oldest lines are credited to the oldest commit read.