| 1 | /** |
| 2 | * A run's model spend, held to its cost cap by the proxy itself. |
| 3 | * |
| 4 | * The harness stops the agent at the run's cap (`--max-budget-usd`), but |
| 5 | * that is the sandbox's own word: an agent talked into calling the proxy |
| 6 | * directly would never be stopped by it. So the proxy counts what each of |
| 7 | * the run's answers cost and refuses its requests once the run has spent |
| 8 | * its cap, with a 402 the harness shows as the error it is. |
| 9 | * |
| 10 | * The count lives in one Durable Object per run (`run-spend.ts`): a run's |
| 11 | * requests land on many isolates, and an agent can send many at once, so a |
| 12 | * per-isolate count would let each isolate spend the cap again. This file |
| 13 | * is the counting and pricing alone, without the worker, for tests. |
| 14 | */ |
| 15 | import type { GatewayModel, ModelUpstream } from "@g1t/contracts"; |
| 16 | |
| 17 | import { anthropicError } from "./gateway.ts"; |
| 18 | import { type Tokens, total } from "./usage.ts"; |
| 19 | |
| 20 | /** A model's prices per million tokens, in millionths of a dollar. */ |
| 21 | export type Prices = Omit<GatewayModel, "model" | "name" | "provider" | "kind">; |
| 22 | |
| 23 | /** |
| 24 | * The cap of a run whose sandbox never set one (its guardrails and its plan |
| 25 | * both said none, or setting it failed): the most any run may be allowed |
| 26 | * to cost (`MAX_BUDGET_USD` in crates/contracts/src/guardrails.rs). |
| 27 | */ |
| 28 | export const BACKSTOP_CAP_MICROS = 100_000_000; |
| 29 | |
| 30 | /** |
| 31 | * The most of a run's answers in flight at once. The cap is checked as a |
| 32 | * request starts and an answer's cost is known only when it ends, so the |
| 33 | * answers already in flight when the run reaches its cap can take it past |
| 34 | * by their cost; this bounds how many there can be. The harness runs a |
| 35 | * handful at a time, subagents included. |
| 36 | */ |
| 37 | export const MAX_IN_FLIGHT = 16; |
| 38 | |
| 39 | /** |
| 40 | * How long an answer in flight is waited for before it no longer counts |
| 41 | * against `MAX_IN_FLIGHT`: one whose cost was never settled (the proxy's |
| 42 | * isolate went away mid-answer) must not hold a place for good. |
| 43 | */ |
| 44 | export const HOLD_MS = 15 * 60_000; |
| 45 | |
| 46 | /** The output an answer is assumed to reach when its request names no `max_tokens`. */ |
| 47 | const DEFAULT_MAX_TOKENS = 32_000; |
| 48 | |
| 49 | /** |
| 50 | * Prices for a model g1t has none for. On g1t's models, a model missing |
| 51 | * from the catalogue is charged at a frontier model's list prices, so the |
| 52 | * cap errs on the side of stopping early. On the workspace's own provider |
| 53 | * (an endpoint whose model g1t does not list), at a standard model's, so |
| 54 | * an inexpensive model is not stopped far short of its cap; the harness, |
| 55 | * which knows no better, counts such a model much the same way. |
| 56 | */ |
| 57 | export const FRONTIER_PRICES: Prices = { |
| 58 | inputMicros: 15_000_000, |
| 59 | outputMicros: 75_000_000, |
| 60 | cacheReadMicros: 1_500_000, |
| 61 | cacheWriteMicros: 18_750_000, |
| 62 | cacheWrite1hMicros: 30_000_000, |
| 63 | }; |
| 64 | export const STANDARD_PRICES: Prices = { |
| 65 | inputMicros: 3_000_000, |
| 66 | outputMicros: 15_000_000, |
| 67 | cacheReadMicros: 300_000, |
| 68 | cacheWriteMicros: 3_750_000, |
| 69 | cacheWrite1hMicros: 6_000_000, |
| 70 | }; |
| 71 | |
| 72 | /** The run's cap, in millionths of a dollar. */ |
| 73 | export function capOf(upstream: Pick<ModelUpstream, "capMicros">): number { |
| 74 | const cap = upstream.capMicros; |
| 75 | return typeof cap === "number" && Number.isFinite(cap) && cap > 0 ? cap : BACKSTOP_CAP_MICROS; |
| 76 | } |
| 77 | |
| 78 | /** |
| 79 | * The prices an answer is charged at: its model's in g1t's catalogue (by |
| 80 | * its id, or the catalogue id a dated id starts with), else the most |
| 81 | * expensive chat model's on g1t's models, else the fallbacks above. |
| 82 | */ |
| 83 | export function pricesFor(model: unknown, route: ModelUpstream["route"], offered: GatewayModel[]): Prices { |
| 84 | const id = (typeof model === "string" ? model : "").trim().toLowerCase().replace(/^[a-z0-9-]+\//, ""); |
| 85 | if (id) { |
| 86 | const exact = offered.find((entry) => entry.model.toLowerCase() === id); |
| 87 | if (exact) return exact; |
| 88 | const prefixed = offered |
| 89 | .filter((entry) => id.startsWith(`${entry.model.toLowerCase()}-`)) |
| 90 | .sort((a, b) => b.model.length - a.model.length)[0]; |
| 91 | if (prefixed) return prefixed; |
| 92 | } |
| 93 | if (route !== "g1t") return STANDARD_PRICES; |
| 94 | const chat = offered.filter((entry) => (entry.kind ?? "chat") === "chat"); |
| 95 | const dearest = chat.sort((a, b) => b.outputMicros - a.outputMicros)[0]; |
| 96 | return dearest && dearest.outputMicros >= FRONTIER_PRICES.outputMicros ? dearest : FRONTIER_PRICES; |
| 97 | } |
| 98 | |
| 99 | /** |
| 100 | * What tokens cost at a model's prices per million, rounded up to a whole |
| 101 | * millionth of a dollar: billing's own sum (`cost_micros` in |
| 102 | * services/billing/src/gateway.rs). A prompt longer than the model's |
| 103 | * threshold puts the whole request at the over-threshold prices. |
| 104 | */ |
| 105 | export function costMicros(prices: Prices, tokens: Tokens): number { |
| 106 | const prompt = tokens.input + tokens.cacheRead + tokens.cacheWrite; |
| 107 | const over = (prices.threshold ?? 0) > 0 && prompt > (prices.threshold ?? 0); |
| 108 | const pick = (base: number | undefined, above: number | undefined) => Math.max(0, (over ? above : base) ?? 0); |
| 109 | const hour = Math.min(tokens.cacheWrite1h ?? 0, tokens.cacheWrite); |
| 110 | const fiveMinutes = pick(prices.cacheWriteMicros, prices.overCacheWriteMicros); |
| 111 | const hourPrice = pick(prices.cacheWrite1hMicros, prices.overCacheWrite1hMicros) || fiveMinutes; |
| 112 | const millionths = |
| 113 | tokens.input * pick(prices.inputMicros, prices.overInputMicros) + |
| 114 | tokens.output * pick(prices.outputMicros, prices.overOutputMicros) + |
| 115 | tokens.cacheRead * pick(prices.cacheReadMicros, prices.overCacheReadMicros) + |
| 116 | (tokens.cacheWrite - hour) * fiveMinutes + |
| 117 | hour * hourPrice; |
| 118 | return Math.ceil(millionths / 1_000_000); |
| 119 | } |
| 120 | |
| 121 | /** |
| 122 | * The most a request could cost: its whole body as input (about four |
| 123 | * characters a token) and its `max_tokens` as output. What an answer that |
| 124 | * could not be read is charged. |
| 125 | */ |
| 126 | export function ceilingMicros(prices: Prices, bodyLength: number, maxTokens: unknown): number { |
| 127 | const output = typeof maxTokens === "number" && Number.isFinite(maxTokens) && maxTokens > 0 ? maxTokens : DEFAULT_MAX_TOKENS; |
| 128 | return costMicros(prices, { input: Math.ceil(bodyLength / 4), output, cacheRead: 0, cacheWrite: 0 }); |
| 129 | } |
| 130 | |
| 131 | /** |
| 132 | * What one answer is charged against the run's cap: nothing for a refused |
| 133 | * request; its tokens' cost; or, for an answer whose usage could not be |
| 134 | * read (cut off, or unreadable), the most it could have cost, so a reading |
| 135 | * that fails never makes an answer free. |
| 136 | */ |
| 137 | export function chargeFor(prices: Prices, tokens: Tokens, ok: boolean, ceiling: number): number { |
| 138 | if (!ok) return 0; |
| 139 | return total(tokens) === 0 ? ceiling : costMicros(prices, tokens); |
| 140 | } |
| 141 | |
| 142 | export type Admission = { ok: true; ticket: string; spent: number } | { ok: false; reason: "cap" | "busy"; spent: number }; |
| 143 | |
| 144 | /** |
| 145 | * One run's count: what it has spent, and its answers in flight. Every |
| 146 | * method runs to its end without waiting, so a Durable Object's requests |
| 147 | * see each other's changes in order. |
| 148 | */ |
| 149 | export class SpendTally { |
| 150 | spent: number; |
| 151 | private readonly open = new Map<string, number>(); |
| 152 | private issued = 0; |
| 153 | |
| 154 | constructor(spent = 0) { |
| 155 | this.spent = Number.isFinite(spent) && spent > 0 ? spent : 0; |
| 156 | } |
| 157 | |
| 158 | /** Whether another answer may start: the run is under its cap and not too busy. */ |
| 159 | admit(cap: number, now: number): Admission { |
| 160 | for (const [ticket, since] of this.open) if (now - since > HOLD_MS) this.open.delete(ticket); |
| 161 | if (this.spent >= cap) return { ok: false, reason: "cap", spent: this.spent }; |
| 162 | if (this.open.size >= MAX_IN_FLIGHT) return { ok: false, reason: "busy", spent: this.spent }; |
| 163 | const ticket = `t${++this.issued}`; |
| 164 | this.open.set(ticket, now); |
| 165 | return { ok: true, ticket, spent: this.spent }; |
| 166 | } |
| 167 | |
| 168 | /** An answer has ended, costing `micros`. Returns what the run has spent. */ |
| 169 | settle(ticket: string, micros: number): number { |
| 170 | this.open.delete(ticket); |
| 171 | if (Number.isFinite(micros) && micros > 0) this.spent += Math.ceil(micros); |
| 172 | return this.spent; |
| 173 | } |
| 174 | |
| 175 | get inFlight(): number { |
| 176 | return this.open.size; |
| 177 | } |
| 178 | } |
| 179 | |
| 180 | const dollars = (micros: number) => `$${(micros / 1_000_000).toFixed(2)}`; |
| 181 | |
| 182 | /** |
| 183 | * The refusal of a run past its cap: Anthropic's error shape, which the |
| 184 | * harness shows as it is, with `code` for anything that reads it. |
| 185 | */ |
| 186 | export function capReached(cap: number, spent: number): Response { |
| 187 | const message = `This run reached its cost cap of ${dollars(cap)} (it has spent ${dollars(spent)} on models), so g1t refuses its model requests from here. Raise the cap in the project's guardrails or the workspace's billing, then start the work again.`; |
| 188 | return Response.json({ type: "error", error: { type: "billing_error", code: "run_cap_reached", message } }, { status: 402 }); |
| 189 | } |
| 190 | |
| 191 | /** The refusal of a request while the run has too many answers in flight; the harness retries it. */ |
| 192 | export function tooBusy(): Response { |
| 193 | const response = anthropicError(429, "rate_limit_error", `This run has ${MAX_IN_FLIGHT} model requests in flight. Wait for one to finish.`); |
| 194 | response.headers.set("retry-after", "2"); |
| 195 | return response; |
| 196 | } |