Pick any line to see why it is the way it is: the commit, the pull request and issue it came from, and what the agent was thinking.
| Models: g1t keeps up with new models, and staff choose each default in sudo | 1 | import assert from "node:assert/strict"; |
| 2 | import { test } from "node:test"; | |
| 3 | ||
| 4 | import { DEFAULT_ROUTING, type StoredDefaults, effortFor, route, routingReader, tierVars, withDefaults } from "./model-env.ts"; | |
| 5 | ||
| 6 | const stored = (): StoredDefaults => ({ | |
| 7 | models: [ | |
| 8 | { | |
| 9 | purpose: "tier_small", | |
| 10 | model: { model: "claude-haiku-6", name: "Claude Haiku 6", inputMicros: 80_000, outputMicros: 400_000, cacheReadMicros: 8_000, cacheWriteMicros: 100_000 }, | |
| 11 | capabilities: ["effort", "thinking", "tools"], | |
| 12 | note: null, | |
| 13 | }, | |
| 14 | { | |
| 15 | purpose: "tier_frontier", | |
| 16 | model: { model: "claude-opus-5-5", name: "Claude Opus 5.5", inputMicros: 4_000_000, outputMicros: 20_000_000, cacheReadMicros: 200_000, cacheWriteMicros: 5_000_000 }, | |
| 17 | capabilities: ["effort"], | |
| 18 | note: "Claude Fable 6 is retired; using Claude Opus 5.5 instead.", | |
| 19 | }, | |
| 20 | { | |
| 21 | purpose: "background", | |
| 22 | model: { model: "claude-haiku-4-5", name: "Claude Haiku 4.5", inputMicros: 1_000_000, outputMicros: 5_000_000, cacheReadMicros: 100_000, cacheWriteMicros: 1_250_000 }, | |
| 23 | capabilities: ["thinking", "tools"], | |
| 24 | note: null, | |
| 25 | }, | |
| 26 | // Nothing suited: the configured tier stays. | |
| 27 | { purpose: "tier_large", model: null, capabilities: [], note: "Claude Sonnet 6 is retired, and no other model suits it." }, | |
| 28 | ], | |
| 29 | jobs: [ | |
| 30 | { kind: "implement", tier: "frontier", effort: "xhigh" }, | |
| 31 | { kind: "plan", tier: "large", effort: null }, | |
| 32 | { kind: "update", tier: "change", effort: "low" }, | |
| 33 | { kind: "nonsense", tier: "small", effort: "low" }, | |
| 34 | ], | |
| 35 | }); | |
| 36 | ||
| 37 | test("staff's defaults from the catalogue replace each tier's model, named as the catalogue names it", () => { | |
| 38 | const routing = withDefaults(DEFAULT_ROUTING, stored()); | |
| 39 | assert.equal(routing.tiers.small.model, "claude-haiku-6"); | |
| 40 | assert.deepEqual(routing.tiers.small.price, { input: 0.08, output: 0.4, cacheRead: 0.008, cacheWrite: 0.1 }); | |
| 41 | assert.equal(routing.tiers.large.model, DEFAULT_ROUTING.tiers.large.model); | |
| 42 | // The run's line uses the catalogue's name. | |
| 43 | assert.equal(route("update", {}, routing).reason, "Used a fast model (Claude Haiku 6): catching up with the base branch."); | |
| 44 | // A tier that fell back says so. | |
| 45 | assert.equal( | |
| 46 | route("plan", { labels: ["architecture"] }, routing).reason, | |
| 47 | "Used the most capable model (Claude Opus 5.5): the issue is labelled architecture. Claude Fable 6 is retired; using Claude Opus 5.5 instead.", | |
| 48 | ); | |
| 49 | }); | |
| 50 | ||
| 51 | test("staff's defaults set each job's tier and effort, and the background model", () => { | |
| 52 | const routing = withDefaults(DEFAULT_ROUTING, stored()); | |
| 53 | assert.equal(routing.tasks.implement, "frontier"); | |
| 54 | assert.equal(routing.effort.implement, "xhigh"); | |
| 55 | // No effort: the harness's own. | |
| 56 | assert.equal(routing.tasks.plan, "large"); | |
| 57 | assert.equal(routing.effort.plan, undefined); | |
| 58 | // Only a review is sized by its change. | |
| 59 | assert.equal(routing.tasks.update, DEFAULT_ROUTING.tasks.update); | |
| 60 | const vars = tierVars(routing, "large"); | |
| 61 | assert.equal(vars.ANTHROPIC_SMALL_FAST_MODEL, "claude-haiku-4-5"); | |
| 62 | assert.equal(vars.ANTHROPIC_DEFAULT_HAIKU_MODEL, "claude-haiku-4-5"); | |
| 63 | // Without a background model, the fast tier's. | |
| 64 | assert.equal(tierVars(DEFAULT_ROUTING, "large").ANTHROPIC_SMALL_FAST_MODEL, DEFAULT_ROUTING.tiers.small.model); | |
| 65 | }); | |
| 66 | ||
| 67 | test("effort is left to the harness on a model that takes none", () => { | |
| 68 | const routing = withDefaults(DEFAULT_ROUTING, stored()); | |
| 69 | assert.equal(effortFor(routing, "update", "small"), "low"); | |
| 70 | const noEffort = { ...routing, tiers: { ...routing.tiers, small: { ...routing.tiers.small, capabilities: ["tools"] } } }; | |
| 71 | assert.equal(effortFor(noEffort, "update", "small"), undefined); | |
| 72 | // A route from configuration says nothing about capabilities: effort is sent. | |
| 73 | assert.equal(effortFor(DEFAULT_ROUTING, "plan", "small"), "high"); | |
| 74 | }); | |
| 75 | ||
| 76 | test("routing reads billing's defaults once a minute, and AGENT_ROUTING alone when billing cannot be read", async () => { | |
| 77 | const now = routingReader(60_000, 10_000); | |
| 78 | let reads = 0; | |
| 79 | const read = async () => { | |
| 80 | reads += 1; | |
| 81 | return stored(); | |
| 82 | }; | |
| 83 | assert.equal((await now(undefined, read, 0)).tiers.small.model, "claude-haiku-6"); | |
| 84 | assert.equal((await now(undefined, read, 59_000)).tiers.small.model, "claude-haiku-6"); | |
| 85 | assert.equal(reads, 1); | |
| 86 | await now(undefined, read, 60_000); | |
| 87 | assert.equal(reads, 2); | |
| 88 | ||
| 89 | const failing = routingReader(60_000, 10_000); | |
| 90 | const broken = async (): Promise<StoredDefaults> => { | |
| 91 | reads += 1; | |
| 92 | throw new Error("billing is down"); | |
| 93 | }; | |
| 94 | const configured = JSON.stringify({ tiers: { small: { modelName: "Claude Haiku 4.5", model: "claude-haiku-4-5" } } }); | |
| 95 | const fallback = await failing(configured, broken, 0); | |
| 96 | assert.equal(fallback.tiers.small.model, "claude-haiku-4-5"); | |
| 97 | assert.equal(fallback.tiers.large.model, DEFAULT_ROUTING.tiers.large.model); | |
| 98 | // Asked again after ten seconds, not on every run. | |
| 99 | const before = reads; | |
| 100 | await failing(configured, broken, 5_000); | |
| 101 | assert.equal(reads, before); | |
| 102 | assert.equal((await failing(configured, read, 10_000)).tiers.small.model, "claude-haiku-6"); | |
| 103 | }); |