Pick any line to see why it is the way it is: the commit, the pull request and issue it came from, and what the agent was thinking.
| Merge platform pause and the hourly usage watcher: staff can pause compute, schedules, indexing or renders for everyone, the watcher emails on a breach and is never blind quietly, and the models proxy holds each run to its cap (billing 0051, integrations 0006) | 1 | import assert from "node:assert/strict"; |
| 2 | import { test } from "node:test"; | |
| 3 | ||
| 4 | import type { GatewayModel } from "@g1t/contracts"; | |
| 5 | ||
| 6 | import { | |
| 7 | BACKSTOP_CAP_MICROS, | |
| 8 | FRONTIER_PRICES, | |
| 9 | HOLD_MS, | |
| 10 | MAX_IN_FLIGHT, | |
| 11 | STANDARD_PRICES, | |
| 12 | SpendTally, | |
| 13 | capOf, | |
| 14 | capReached, | |
| 15 | ceilingMicros, | |
| 16 | chargeFor, | |
| 17 | costMicros, | |
| 18 | pricesFor, | |
| 19 | tooBusy, | |
| 20 | } from "./spend.ts"; | |
| 21 | import { NO_TOKENS } from "./usage.ts"; | |
| 22 | ||
| 23 | function model(fields: Partial<GatewayModel> & { model: string }): GatewayModel { | |
| 24 | return { name: fields.model, provider: "anthropic", inputMicros: 0, outputMicros: 0, cacheReadMicros: 0, cacheWriteMicros: 0, ...fields }; | |
| 25 | } | |
| 26 | ||
| 27 | const sonnet = model({ model: "claude-sonnet-5-5", inputMicros: 3_000_000, outputMicros: 15_000_000, cacheReadMicros: 300_000, cacheWriteMicros: 3_750_000, cacheWrite1hMicros: 6_000_000 }); | |
| 28 | const haiku = model({ model: "claude-haiku-5", inputMicros: 1_000_000, outputMicros: 5_000_000, cacheReadMicros: 100_000, cacheWriteMicros: 1_250_000 }); | |
| 29 | const embed = model({ model: "@cf/baai/bge-m3", provider: "workers-ai", kind: "embeddings", inputMicros: 12_000, outputMicros: 900_000_000 }); | |
| 30 | const offered = [haiku, sonnet, embed]; | |
| 31 | ||
| 32 | test("a run's cap is the one its sandbox set, else the most any run may cost", () => { | |
| 33 | assert.equal(capOf({ capMicros: 2_000_000 }), 2_000_000); | |
| 34 | assert.equal(capOf({ capMicros: null }), BACKSTOP_CAP_MICROS); | |
| 35 | assert.equal(capOf({}), BACKSTOP_CAP_MICROS); | |
| 36 | assert.equal(capOf({ capMicros: 0 }), BACKSTOP_CAP_MICROS); | |
| 37 | }); | |
| 38 | ||
| 39 | test("an answer is priced by its model in the catalogue, dated ids and provider prefixes included", () => { | |
| 40 | assert.equal(pricesFor("claude-sonnet-5-5", "g1t", offered), sonnet); | |
| 41 | assert.equal(pricesFor("claude-sonnet-5-5-20260901", "g1t", offered), sonnet); | |
| 42 | assert.equal(pricesFor("anthropic/claude-haiku-5", "anthropic", offered), haiku); | |
| 43 | }); | |
| 44 | ||
| 45 | test("a model missing from the catalogue is priced high on g1t's models, at standard prices on the workspace's own", () => { | |
| 46 | assert.equal(pricesFor("mystery", "g1t", offered), FRONTIER_PRICES); | |
| 47 | assert.equal(pricesFor(null, "g1t", []), FRONTIER_PRICES); | |
| 48 | const dearer = model({ model: "claude-opus-9", inputMicros: 20_000_000, outputMicros: 100_000_000 }); | |
| 49 | assert.equal(pricesFor("mystery", "g1t", [...offered, dearer]), dearer); | |
| 50 | assert.equal(pricesFor("llama-local", "endpoint", offered), STANDARD_PRICES); | |
| 51 | }); | |
| 52 | ||
| 53 | test("cost is billing's sum: each kind of token at its price, rounded up", () => { | |
| 54 | // 1,000 in at $3, 500 out at $15, 10,000 cache reads at $0.30, 2,000 writes (500 of them hour-long). | |
| 55 | const tokens = { input: 1_000, output: 500, cacheRead: 10_000, cacheWrite: 2_000, cacheWrite1h: 500 }; | |
| 56 | assert.equal(costMicros(sonnet, tokens), 3_000 + 7_500 + 3_000 + 5_625 + 3_000); | |
| 57 | assert.equal(costMicros(sonnet, { ...NO_TOKENS, input: 1 }), 3); | |
| 58 | assert.equal(costMicros(haiku, { ...NO_TOKENS, input: 1 }), 1); | |
| 59 | // No hour-long price: those writes cost what five-minute ones do. | |
| 60 | assert.equal(costMicros(haiku, { ...NO_TOKENS, cacheWrite: 1_000, cacheWrite1h: 1_000 }), 1_250); | |
| 61 | }); | |
| 62 | ||
| 63 | test("a prompt past the model's threshold puts the whole request at the higher prices", () => { | |
| 64 | const long = model({ model: "long", inputMicros: 1_000_000, outputMicros: 2_000_000, threshold: 100, overInputMicros: 2_000_000, overOutputMicros: 4_000_000 }); | |
| 65 | assert.equal(costMicros(long, { ...NO_TOKENS, input: 100, output: 1_000 }), 100 + 2_000); | |
| 66 | assert.equal(costMicros(long, { ...NO_TOKENS, input: 101, output: 1_000 }), 202 + 4_000); | |
| 67 | }); | |
| 68 | ||
| 69 | test("an answer is charged its cost, nothing when refused, and its ceiling when its usage could not be read", () => { | |
| 70 | const ceiling = ceilingMicros(sonnet, 4_000, 1_000); | |
| 71 | assert.equal(ceiling, 3_000 + 15_000); | |
| 72 | assert.equal(ceilingMicros(sonnet, 0, undefined), 32_000 * 15); | |
| 73 | assert.equal(chargeFor(sonnet, { ...NO_TOKENS, output: 100 }, true, ceiling), 1_500); | |
| 74 | assert.equal(chargeFor(sonnet, NO_TOKENS, true, ceiling), ceiling); | |
| 75 | assert.equal(chargeFor(sonnet, NO_TOKENS, false, ceiling), 0); | |
| 76 | }); | |
| 77 | ||
| 78 | test("a run under its cap is admitted, and refused once it has spent it", () => { | |
| 79 | const tally = new SpendTally(); | |
| 80 | const first = tally.admit(10_000, 0); | |
| 81 | assert.ok(first.ok); | |
| 82 | assert.equal(tally.settle(first.ticket, 9_999), 9_999); | |
| 83 | const second = tally.admit(10_000, 1); | |
| 84 | assert.ok(second.ok); | |
| 85 | // The answer in flight takes it past; the next is refused. | |
| 86 | tally.settle(second.ticket, 5_000); | |
| 87 | assert.deepEqual(tally.admit(10_000, 2), { ok: false, reason: "cap", spent: 14_999 }); | |
| 88 | }); | |
| 89 | ||
| 90 | test("many requests at once are bounded, and an answer never settled gives its place back in time", () => { | |
| 91 | const tally = new SpendTally(); | |
| 92 | const tickets: string[] = []; | |
| 93 | for (let i = 0; i < MAX_IN_FLIGHT; i++) { | |
| 94 | const admitted = tally.admit(1_000_000, 0); | |
| 95 | assert.ok(admitted.ok); | |
| 96 | tickets.push(admitted.ticket); | |
| 97 | } | |
| 98 | assert.deepEqual(tally.admit(1_000_000, 0), { ok: false, reason: "busy", spent: 0 }); | |
| 99 | tally.settle(tickets[0], 0); | |
| 100 | assert.ok(tally.admit(1_000_000, 1).ok); | |
| 101 | assert.equal(tally.admit(1_000_000, 1).ok, false); | |
| 102 | assert.ok(tally.admit(1_000_000, HOLD_MS + 1).ok); | |
| 103 | // Those from the start have lapsed; the one from a moment later has not. | |
| 104 | assert.equal(tally.inFlight, 2); | |
| 105 | }); | |
| 106 | ||
| 107 | test("a tally resumes from what was stored, and settling twice or with nonsense adds nothing", () => { | |
| 108 | const tally = new SpendTally(500); | |
| 109 | const admitted = tally.admit(1_000, 0); | |
| 110 | assert.ok(admitted.ok); | |
| 111 | tally.settle(admitted.ticket, Number.NaN); | |
| 112 | tally.settle(admitted.ticket, -5); | |
| 113 | assert.equal(tally.spent, 500); | |
| 114 | assert.equal(new SpendTally(-1).spent, 0); | |
| 115 | }); | |
| 116 | ||
| 117 | test("a run past its cap is told so in Anthropic's error shape, with a code", async () => { | |
| 118 | const response = capReached(2_000_000, 2_031_000); | |
| 119 | assert.equal(response.status, 402); | |
| 120 | const body = (await response.json()) as { type: string; error: { type: string; code: string; message: string } }; | |
| 121 | assert.equal(body.type, "error"); | |
| 122 | assert.equal(body.error.type, "billing_error"); | |
| 123 | assert.equal(body.error.code, "run_cap_reached"); | |
| 124 | assert.match(body.error.message, /cost cap of \$2\.00 \(it has spent \$2\.03/); | |
| 125 | const busy = tooBusy(); | |
| 126 | assert.equal(busy.status, 429); | |
| 127 | assert.equal(busy.headers.get("retry-after"), "2"); | |
| 128 | }); |
This file's history is long; its oldest lines are credited to the oldest commit read.