Skip to content
196 linesCodeBlameRaw

Pick any line to see why it is the way it is: the commit, the pull request and issue it came from, and what the agent was thinking.

Merge platform pause and the hourly usage watcher: staff can pause compute, schedules, indexing or renders for everyone, the watcher emails on a breach and is never blind quietly, and the models proxy holds each run to its cap (billing 0051, integrations 0006)1/**
2 * A run's model spend, held to its cost cap by the proxy itself.
3 *
4 * The harness stops the agent at the run's cap (`--max-budget-usd`), but
5 * that is the sandbox's own word: an agent talked into calling the proxy
6 * directly would never be stopped by it. So the proxy counts what each of
7 * the run's answers cost and refuses its requests once the run has spent
8 * its cap, with a 402 the harness shows as the error it is.
9 *
10 * The count lives in one Durable Object per run (`run-spend.ts`): a run's
11 * requests land on many isolates, and an agent can send many at once, so a
12 * per-isolate count would let each isolate spend the cap again. This file
13 * is the counting and pricing alone, without the worker, for tests.
14 */
15import type { GatewayModel, ModelUpstream } from "@g1t/contracts";
16
17import { anthropicError } from "./gateway.ts";
18import { type Tokens, total } from "./usage.ts";
19
20/** A model's prices per million tokens, in millionths of a dollar. */
21export type Prices = Omit<GatewayModel, "model" | "name" | "provider" | "kind">;
22
23/**
24 * The cap of a run whose sandbox never set one (its guardrails and its plan
25 * both said none, or setting it failed): the most any run may be allowed
26 * to cost (`MAX_BUDGET_USD` in crates/contracts/src/guardrails.rs).
27 */
28export const BACKSTOP_CAP_MICROS = 100_000_000;
29
30/**
31 * The most of a run's answers in flight at once. The cap is checked as a
32 * request starts and an answer's cost is known only when it ends, so the
33 * answers already in flight when the run reaches its cap can take it past
34 * by their cost; this bounds how many there can be. The harness runs a
35 * handful at a time, subagents included.
36 */
37export const MAX_IN_FLIGHT = 16;
38
39/**
40 * How long an answer in flight is waited for before it no longer counts
41 * against `MAX_IN_FLIGHT`: one whose cost was never settled (the proxy's
42 * isolate went away mid-answer) must not hold a place for good.
43 */
44export const HOLD_MS = 15 * 60_000;
45
46/** The output an answer is assumed to reach when its request names no `max_tokens`. */
47const DEFAULT_MAX_TOKENS = 32_000;
48
49/**
50 * Prices for a model g1t has none for. On g1t's models, a model missing
51 * from the catalogue is charged at a frontier model's list prices, so the
52 * cap errs on the side of stopping early. On the workspace's own provider
53 * (an endpoint whose model g1t does not list), at a standard model's, so
54 * an inexpensive model is not stopped far short of its cap; the harness,
55 * which knows no better, counts such a model much the same way.
56 */
57export const FRONTIER_PRICES: Prices = {
58 inputMicros: 15_000_000,
59 outputMicros: 75_000_000,
60 cacheReadMicros: 1_500_000,
61 cacheWriteMicros: 18_750_000,
62 cacheWrite1hMicros: 30_000_000,
63};
64export const STANDARD_PRICES: Prices = {
65 inputMicros: 3_000_000,
66 outputMicros: 15_000_000,
67 cacheReadMicros: 300_000,
68 cacheWriteMicros: 3_750_000,
69 cacheWrite1hMicros: 6_000_000,
70};
71
72/** The run's cap, in millionths of a dollar. */
73export function capOf(upstream: Pick<ModelUpstream, "capMicros">): number {
74 const cap = upstream.capMicros;
75 return typeof cap === "number" && Number.isFinite(cap) && cap > 0 ? cap : BACKSTOP_CAP_MICROS;
76}
77
78/**
79 * The prices an answer is charged at: its model's in g1t's catalogue (by
80 * its id, or the catalogue id a dated id starts with), else the most
81 * expensive chat model's on g1t's models, else the fallbacks above.
82 */
83export function pricesFor(model: unknown, route: ModelUpstream["route"], offered: GatewayModel[]): Prices {
84 const id = (typeof model === "string" ? model : "").trim().toLowerCase().replace(/^[a-z0-9-]+\//, "");
85 if (id) {
86 const exact = offered.find((entry) => entry.model.toLowerCase() === id);
87 if (exact) return exact;
88 const prefixed = offered
89 .filter((entry) => id.startsWith(`${entry.model.toLowerCase()}-`))
90 .sort((a, b) => b.model.length - a.model.length)[0];
91 if (prefixed) return prefixed;
92 }
93 if (route !== "g1t") return STANDARD_PRICES;
94 const chat = offered.filter((entry) => (entry.kind ?? "chat") === "chat");
95 const dearest = chat.sort((a, b) => b.outputMicros - a.outputMicros)[0];
96 return dearest && dearest.outputMicros >= FRONTIER_PRICES.outputMicros ? dearest : FRONTIER_PRICES;
97}
98
99/**
100 * What tokens cost at a model's prices per million, rounded up to a whole
101 * millionth of a dollar: billing's own sum (`cost_micros` in
102 * services/billing/src/gateway.rs). A prompt longer than the model's
103 * threshold puts the whole request at the over-threshold prices.
104 */
105export function costMicros(prices: Prices, tokens: Tokens): number {
106 const prompt = tokens.input + tokens.cacheRead + tokens.cacheWrite;
107 const over = (prices.threshold ?? 0) > 0 && prompt > (prices.threshold ?? 0);
108 const pick = (base: number | undefined, above: number | undefined) => Math.max(0, (over ? above : base) ?? 0);
109 const hour = Math.min(tokens.cacheWrite1h ?? 0, tokens.cacheWrite);
110 const fiveMinutes = pick(prices.cacheWriteMicros, prices.overCacheWriteMicros);
111 const hourPrice = pick(prices.cacheWrite1hMicros, prices.overCacheWrite1hMicros) || fiveMinutes;
112 const millionths =
113 tokens.input * pick(prices.inputMicros, prices.overInputMicros) +
114 tokens.output * pick(prices.outputMicros, prices.overOutputMicros) +
115 tokens.cacheRead * pick(prices.cacheReadMicros, prices.overCacheReadMicros) +
116 (tokens.cacheWrite - hour) * fiveMinutes +
117 hour * hourPrice;
118 return Math.ceil(millionths / 1_000_000);
119}
120
121/**
122 * The most a request could cost: its whole body as input (about four
123 * characters a token) and its `max_tokens` as output. What an answer that
124 * could not be read is charged.
125 */
126export function ceilingMicros(prices: Prices, bodyLength: number, maxTokens: unknown): number {
127 const output = typeof maxTokens === "number" && Number.isFinite(maxTokens) && maxTokens > 0 ? maxTokens : DEFAULT_MAX_TOKENS;
128 return costMicros(prices, { input: Math.ceil(bodyLength / 4), output, cacheRead: 0, cacheWrite: 0 });
129}
130
131/**
132 * What one answer is charged against the run's cap: nothing for a refused
133 * request; its tokens' cost; or, for an answer whose usage could not be
134 * read (cut off, or unreadable), the most it could have cost, so a reading
135 * that fails never makes an answer free.
136 */
137export function chargeFor(prices: Prices, tokens: Tokens, ok: boolean, ceiling: number): number {
138 if (!ok) return 0;
139 return total(tokens) === 0 ? ceiling : costMicros(prices, tokens);
140}
141
142export type Admission = { ok: true; ticket: string; spent: number } | { ok: false; reason: "cap" | "busy"; spent: number };
143
144/**
145 * One run's count: what it has spent, and its answers in flight. Every
146 * method runs to its end without waiting, so a Durable Object's requests
147 * see each other's changes in order.
148 */
149export class SpendTally {
150 spent: number;
151 private readonly open = new Map<string, number>();
152 private issued = 0;
153
154 constructor(spent = 0) {
155 this.spent = Number.isFinite(spent) && spent > 0 ? spent : 0;
156 }
157
158 /** Whether another answer may start: the run is under its cap and not too busy. */
159 admit(cap: number, now: number): Admission {
160 for (const [ticket, since] of this.open) if (now - since > HOLD_MS) this.open.delete(ticket);
161 if (this.spent >= cap) return { ok: false, reason: "cap", spent: this.spent };
162 if (this.open.size >= MAX_IN_FLIGHT) return { ok: false, reason: "busy", spent: this.spent };
163 const ticket = `t${++this.issued}`;
164 this.open.set(ticket, now);
165 return { ok: true, ticket, spent: this.spent };
166 }
167
168 /** An answer has ended, costing `micros`. Returns what the run has spent. */
169 settle(ticket: string, micros: number): number {
170 this.open.delete(ticket);
171 if (Number.isFinite(micros) && micros > 0) this.spent += Math.ceil(micros);
172 return this.spent;
173 }
174
175 get inFlight(): number {
176 return this.open.size;
177 }
178}
179
180const dollars = (micros: number) => `$${(micros / 1_000_000).toFixed(2)}`;
181
182/**
183 * The refusal of a run past its cap: Anthropic's error shape, which the
184 * harness shows as it is, with `code` for anything that reads it.
185 */
186export function capReached(cap: number, spent: number): Response {
187 const message = `This run reached its cost cap of ${dollars(cap)} (it has spent ${dollars(spent)} on models), so g1t refuses its model requests from here. Raise the cap in the project's guardrails or the workspace's billing, then start the work again.`;
188 return Response.json({ type: "error", error: { type: "billing_error", code: "run_cap_reached", message } }, { status: 402 });
189}
190
191/** The refusal of a request while the run has too many answers in flight; the harness retries it. */
192export function tooBusy(): Response {
193 const response = anthropicError(429, "rate_limit_error", `This run has ${MAX_IN_FLIGHT} model requests in flight. Wait for one to finish.`);
194 response.headers.set("retry-after", "2");
195 return response;
196}

This file's history is long; its oldest lines are credited to the oldest commit read.