| 1 | /** |
| 2 | * What a model answer used, read as it passes through: Anthropic's |
| 3 | * `usage`, from a whole JSON answer or from a stream's `message_start` |
| 4 | * (input and cache) and `message_delta` (output) events. The proxy adds |
| 5 | * these up per run for usage views; billing still prices runs from AI |
| 6 | * Gateway, not from these. |
| 7 | */ |
| 8 | |
| 9 | export type Tokens = { |
| 10 | input: number; |
| 11 | output: number; |
| 12 | /** Prompt tokens read from the provider's cache. */ |
| 13 | cacheRead: number; |
| 14 | /** Prompt tokens written to the provider's cache. */ |
| 15 | cacheWrite: number; |
| 16 | }; |
| 17 | |
| 18 | export const NO_TOKENS: Tokens = { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }; |
| 19 | |
| 20 | type Usage = { |
| 21 | input_tokens?: number; |
| 22 | output_tokens?: number; |
| 23 | cache_read_input_tokens?: number; |
| 24 | cache_creation_input_tokens?: number; |
| 25 | }; |
| 26 | |
| 27 | const count = (value: unknown): number => (typeof value === "number" && Number.isFinite(value) && value > 0 ? value : 0); |
| 28 | |
| 29 | /** The tokens of one `usage` object. */ |
| 30 | export function fromUsage(usage: Usage | undefined | null): Tokens { |
| 31 | return { |
| 32 | input: count(usage?.input_tokens), |
| 33 | output: count(usage?.output_tokens), |
| 34 | cacheRead: count(usage?.cache_read_input_tokens), |
| 35 | cacheWrite: count(usage?.cache_creation_input_tokens), |
| 36 | }; |
| 37 | } |
| 38 | |
| 39 | export function total(tokens: Tokens): number { |
| 40 | return tokens.input + tokens.output + tokens.cacheRead + tokens.cacheWrite; |
| 41 | } |
| 42 | |
| 43 | /** |
| 44 | * Reads a server-sent event stream a chunk at a time. `message_start` |
| 45 | * carries the prompt's tokens; each `message_delta` carries the output so |
| 46 | * far (the last one is the total). |
| 47 | */ |
| 48 | export class StreamUsage { |
| 49 | tokens: Tokens = { ...NO_TOKENS }; |
| 50 | /** The model that answered, as `message_start` names it. */ |
| 51 | model: string | null = null; |
| 52 | private pending = ""; |
| 53 | |
| 54 | push(text: string): void { |
| 55 | this.pending += text; |
| 56 | let end = this.pending.indexOf("\n"); |
| 57 | while (end >= 0) { |
| 58 | this.line(this.pending.slice(0, end).trim()); |
| 59 | this.pending = this.pending.slice(end + 1); |
| 60 | end = this.pending.indexOf("\n"); |
| 61 | } |
| 62 | } |
| 63 | |
| 64 | finish(): Tokens { |
| 65 | this.line(this.pending.trim()); |
| 66 | this.pending = ""; |
| 67 | return this.tokens; |
| 68 | } |
| 69 | |
| 70 | private line(line: string): void { |
| 71 | if (!line.startsWith("data:")) return; |
| 72 | let event: { type?: string; message?: { usage?: Usage; model?: unknown }; usage?: Usage }; |
| 73 | try { |
| 74 | event = JSON.parse(line.slice(5).trim()); |
| 75 | } catch { |
| 76 | return; |
| 77 | } |
| 78 | if (event.type === "message_start") { |
| 79 | if (typeof event.message?.model === "string") this.model = event.message.model; |
| 80 | const start = fromUsage(event.message?.usage); |
| 81 | this.tokens = { ...start, output: Math.max(this.tokens.output, start.output) }; |
| 82 | } else if (event.type === "message_delta" && event.usage) { |
| 83 | const delta = fromUsage(event.usage); |
| 84 | this.tokens.output = Math.max(this.tokens.output, delta.output); |
| 85 | // Some providers repeat the prompt's tokens at the end. |
| 86 | if (delta.input) this.tokens.input = delta.input; |
| 87 | if (delta.cacheRead) this.tokens.cacheRead = delta.cacheRead; |
| 88 | if (delta.cacheWrite) this.tokens.cacheWrite = delta.cacheWrite; |
| 89 | } |
| 90 | } |
| 91 | } |
| 92 | |
| 93 | /** |
| 94 | * The answer as it was, and a promise of what it used, read from a copy of |
| 95 | * its body, with the model that answered when it says. A failed answer, or |
| 96 | * one that is not a message, used nothing. |
| 97 | */ |
| 98 | export function measure(answer: Response): { response: Response; tokens: Promise<Tokens>; model: Promise<string | null> } { |
| 99 | if (!answer.ok || !answer.body) { |
| 100 | return { response: answer, tokens: Promise.resolve({ ...NO_TOKENS }), model: Promise.resolve(null) }; |
| 101 | } |
| 102 | const [passed, copy] = answer.body.tee(); |
| 103 | const response = new Response(passed, answer); |
| 104 | const streaming = (answer.headers.get("content-type") ?? "").includes("text/event-stream"); |
| 105 | let model: string | null = null; |
| 106 | const tokens = (async () => { |
| 107 | try { |
| 108 | if (!streaming) { |
| 109 | const whole = (await new Response(copy).json()) as { usage?: Usage; model?: unknown }; |
| 110 | if (typeof whole.model === "string") model = whole.model; |
| 111 | return fromUsage(whole.usage); |
| 112 | } |
| 113 | const reader = copy.pipeThrough(new TextDecoderStream()).getReader(); |
| 114 | const usage = new StreamUsage(); |
| 115 | for (;;) { |
| 116 | const { done, value } = await reader.read(); |
| 117 | if (done) break; |
| 118 | usage.push(value); |
| 119 | } |
| 120 | const used = usage.finish(); |
| 121 | model = usage.model; |
| 122 | return used; |
| 123 | } catch { |
| 124 | return { ...NO_TOKENS }; |
| 125 | } |
| 126 | })(); |
| 127 | return { response, tokens, model: tokens.then(() => model) }; |
| 128 | } |