Pick any line to see why it is the way it is: the commit, the pull request and issue it came from, and what the agent was thinking.
| Merge branch 'model-routing' | 1 | import assert from "node:assert/strict"; |
| 2 | import { readFileSync } from "node:fs"; | |
| 3 | import { test } from "node:test"; | |
| 4 | ||
| 5 | import { DEFAULT_ROUTING } from "../../services/runner/src/model-env.ts"; | |
| 6 | import { autoNext, compare, configuredRouting, costOn, play, previousNext, previousTier, tasksFromBilling } from "./routing-savings.mjs"; | |
| 7 | ||
| 8 | const routing = DEFAULT_ROUTING; | |
| 9 | const million = { input: 1_000_000, output: 0, cacheRead: 0, cacheWrite: 0 }; | |
| 10 | ||
| 11 | test("tokens are priced at each tier's list price, the fast tier's scaled for extra turns", () => { | |
| 12 | assert.equal(costOn("small", million, routing), 1); | |
| 13 | assert.equal(costOn("large", million, routing), 2); | |
| 14 | assert.equal(costOn("frontier", million, routing), 4); | |
| 15 | assert.equal(costOn("small", million, routing, 1.5), 1.5); | |
| 16 | assert.equal(costOn("large", million, routing, 1.5), 2); | |
| 17 | const mixed = { input: 0, output: 1_000_000, cacheRead: 10_000_000, cacheWrite: 1_000_000 }; | |
| 18 | // $10 of output, $2 of cache reads, $2.50 of cache writes. | |
| 19 | assert.equal(costOn("large", mixed, routing), 14.5); | |
| 20 | const unpriced = { ...routing, tiers: { ...routing.tiers, large: { modelName: "X", model: "x" } } }; | |
| 21 | assert.equal(costOn("large", million, unpriced), null); | |
| 22 | }); | |
| 23 | ||
| 24 | test("a failed attempt is paid for, and Auto retries one tier up until the most capable", () => { | |
| 25 | const task = { id: "t", kind: "implement", failsOn: ["small", "large"], tokens: million }; | |
| 26 | assert.deepEqual(play(task, "small", autoNext(routing), routing, 1), { attempts: ["small", "large", "frontier"], cost: 7, done: true }); | |
| 27 | // Routing before Auto stayed on the large tier and asked a person after two failures. | |
| 28 | assert.deepEqual(play(task, "large", previousNext, routing, 1), { attempts: ["large", "large"], cost: 4, done: false }); | |
| 29 | // Nothing past the most capable model. | |
| 30 | const hopeless = { ...task, failsOn: ["small", "large", "frontier"] }; | |
| 31 | assert.equal(play(hopeless, "small", autoNext(routing), routing, 1).done, false); | |
| 32 | }); | |
| 33 | ||
| 34 | test("routing before Auto is replayed as it was", () => { | |
| 35 | assert.equal(previousTier({ kind: "plan" }), "small"); | |
| 36 | assert.equal(previousTier({ kind: "update" }), "small"); | |
| 37 | assert.equal(previousTier({ kind: "answer" }), "large"); | |
| 38 | assert.equal(previousTier({ kind: "review", change: { files: 3, lines: 80, sensitive: [] } }), "small"); | |
| 39 | assert.equal(previousTier({ kind: "review", change: { files: 3, lines: 80, sensitive: [] }, labels: ["Security"] }), "large"); | |
| 40 | assert.equal(previousTier({ kind: "review" }), "large"); | |
| 41 | }); | |
| 42 | ||
| 43 | test("the comparison totals each policy and its cost per merged change", () => { | |
| 44 | const tasks = [ | |
| 45 | { id: "a", kind: "update", tokens: million, merged: false }, | |
| 46 | { id: "b", kind: "implement", labels: ["docs"], tokens: million, merged: true }, | |
| 47 | ]; | |
| 48 | const { totals, savings, rows } = compare(tasks, routing, { smallTurns: 1 }); | |
| 49 | // Auto: both fast. Before: catching up fast, the change standard. | |
| 50 | assert.equal(totals.auto.cost, 2); | |
| 51 | assert.equal(totals.before.cost, 3); | |
| 52 | assert.equal(totals.frontier.cost, 8); | |
| 53 | assert.equal(totals.auto.perMerged, 2); | |
| 54 | assert.ok(Math.abs(savings.vsBefore - 1 / 3) < 1e-9); | |
| 55 | assert.equal(savings.vsFrontier, 0.75); | |
| 56 | assert.match(rows[1].reason, /labelled docs/); | |
| 57 | }); | |
| 58 | ||
| 59 | test("the sample is well formed and every task is priced", () => { | |
| 60 | const sample = JSON.parse(readFileSync(new URL("./routing-savings.sample.json", import.meta.url), "utf8")); | |
| 61 | assert.ok(sample.tasks.length >= 10); | |
| 62 | for (const task of sample.tasks) { | |
| 63 | assert.ok(["implement", "revise", "answer", "review", "update", "plan"].includes(task.kind), task.id); | |
| 64 | assert.ok(costOn("large", task.tokens, routing) > 0, task.id); | |
| 65 | } | |
| 66 | const { totals } = compare(sample.tasks, routing); | |
| 67 | assert.ok(totals.auto.cost > 0 && totals.before.cost > 0 && totals.frontier.cost > 0); | |
| 68 | }); | |
| 69 | ||
| 70 | test("billing's runs become tasks; reviews keep the tier they ran on", () => { | |
| 71 | const tasks = tasksFromBilling([ | |
| 72 | { id: "run_1", task: "review", tier: "small", input: 10, output: 5, cache_read: 0, cache_write: 0 }, | |
| 73 | { id: "run_2", task: "implement", tier: "large", input: 0, output: 0, cache_read: 0, cache_write: 0 }, | |
| 74 | { id: "run_3", task: "plan", tier: null, input: 1, output: 1, cache_read: 1, cache_write: 1 }, | |
| 75 | ]); | |
| 76 | assert.deepEqual( | |
| 77 | tasks.map((t) => [t.id, t.kind, t.keepTier]), | |
| 78 | [ | |
| 79 | ["run_1", "review", "small"], | |
| 80 | ["run_3", "plan", null], | |
| 81 | ], | |
| 82 | ); | |
| 83 | }); | |
| 84 | ||
| 85 | test("the routing estimated is the runner's configuration", () => { | |
| 86 | const text = readFileSync(new URL("../../services/runner/wrangler.jsonc", import.meta.url), "utf8"); | |
| 87 | assert.deepEqual(configuredRouting(text), DEFAULT_ROUTING); | |
| 88 | assert.deepEqual(configuredRouting("not jsonc {"), DEFAULT_ROUTING); | |
| 89 | }); |
This file's history is long; its oldest lines are credited to the oldest commit read.