Pick any line to see why it is the way it is: the commit, the pull request and issue it came from, and what the agent was thinking.
| AI Gateway: OpenAI's format, open models, and your own providers | 1 | import assert from "node:assert/strict"; |
| 2 | import { test } from "node:test"; | |
| 3 | ||
| 4 | import type { GatewayModel, GatewayProvider, GatewayRecord, User } from "@g1t/contracts"; | |
| 5 | ||
| 6 | import { findOffered, listModels, matchPattern, routeModel } from "./catalogue.ts"; | |
| 7 | import { type GatewayDeps, serveGateway } from "./serve.ts"; | |
| 8 | ||
| 9 | // --- The catalogue and the workspace's own providers --------------------------- | |
| 10 | ||
| 11 | const model = (row: Partial<GatewayModel> & Pick<GatewayModel, "model" | "name" | "provider">): GatewayModel => ({ | |
| 12 | kind: "chat", | |
| 13 | inputMicros: 0, | |
| 14 | outputMicros: 0, | |
| 15 | cacheReadMicros: 0, | |
| 16 | cacheWriteMicros: 0, | |
| 17 | ...row, | |
| 18 | }); | |
| 19 | ||
| 20 | const OFFERED: GatewayModel[] = [ | |
| 21 | model({ | |
| 22 | model: "claude-haiku-5-5", | |
| 23 | name: "Claude Haiku 5.5", | |
| 24 | provider: "anthropic", | |
| 25 | inputMicros: 100_000, | |
| 26 | outputMicros: 500_000, | |
| 27 | cacheReadMicros: 10_000, | |
| 28 | cacheWriteMicros: 125_000, | |
| 29 | cacheWrite1hMicros: 200_000, | |
| 30 | threshold: 100_000, | |
| 31 | overInputMicros: 500_000, | |
| 32 | overOutputMicros: 2_500_000, | |
| 33 | overCacheReadMicros: 50_000, | |
| 34 | overCacheWriteMicros: 625_000, | |
| 35 | overCacheWrite1hMicros: 1_000_000, | |
| 36 | }), | |
| 37 | model({ model: "claude-sonnet-5-5", name: "Claude Sonnet 5.5", provider: "anthropic", inputMicros: 2_000_000, outputMicros: 10_000_000 }), | |
| 38 | model({ model: "claude-haiku-4-5", name: "Claude Haiku 4.5", provider: "anthropic" }), | |
| 39 | model({ model: "claude-haiku-4-5-20251001", name: "Claude Haiku 4.5", provider: "anthropic" }), | |
| 40 | model({ model: "@cf/openai/gpt-oss-120b", name: "gpt-oss-120b", provider: "workers-ai", inputMicros: 350_000, outputMicros: 750_000 }), | |
| 41 | model({ model: "@cf/baai/bge-m3", name: "BGE M3", provider: "workers-ai", kind: "embeddings", inputMicros: 12_000 }), | |
| 42 | ]; | |
| 43 | ||
| 44 | const KEY = "sk-own-0123456789abcdefghij"; | |
| 45 | ||
| 46 | const provider = (row: Partial<GatewayProvider> & Pick<GatewayProvider, "provider" | "api" | "patterns">): GatewayProvider => ({ | |
| 47 | id: "con_1", | |
| 48 | name: "Ours", | |
| 49 | official: false, | |
| 50 | baseUrl: "https://llm.acme.dev/v1", | |
| 51 | apiKey: KEY, | |
| 52 | authHeader: "authorization", | |
| 53 | gatewayToken: null, | |
| 54 | models: [], | |
| 55 | ...row, | |
| 56 | }); | |
| 57 | ||
| 58 | const OLLAMA = provider({ id: "con_ollama", name: "Ollama", provider: "openai_endpoint", api: "openai", patterns: ["ollama/*"], models: ["llama3.3", "qwen3"] }); | |
| 59 | const OPENAI = provider({ id: "con_openai", name: "OpenAI", provider: "openai", api: "openai", official: true, baseUrl: "https://api.openai.com/v1", patterns: ["gpt-*"], models: ["gpt-5.5", "gpt-5.5-mini"] }); | |
| 60 | const ANTHROPIC = provider({ id: "con_anthropic", name: "Our Anthropic key", provider: "anthropic", api: "anthropic", baseUrl: "https://api.anthropic.com", authHeader: "x-api-key", patterns: ["claude-*"] }); | |
| 61 | ||
| 62 | test("patterns take models by id, prefix or namespace", () => { | |
| 63 | assert.equal(matchPattern("gpt-5.5", "gpt-5.5"), "gpt-5.5"); | |
| 64 | assert.equal(matchPattern("gpt-5.5", "gpt-5.5-mini"), null); | |
| 65 | assert.equal(matchPattern("gpt-*", "gpt-5.5-mini"), "gpt-5.5-mini"); | |
| 66 | assert.equal(matchPattern("ollama/*", "ollama/llama3.3"), "llama3.3"); | |
| 67 | assert.equal(matchPattern("ollama/*", "ollama/"), null); | |
| 68 | assert.equal(matchPattern("ollama/*", "llama3.3"), null); | |
| 69 | assert.equal(matchPattern("*", "anything"), "anything"); | |
| 70 | }); | |
| 71 | ||
| 72 | test("a catalogue model is found by its own id or with its provider in front", () => { | |
| 73 | assert.equal(findOffered("claude-sonnet-5-5", OFFERED)?.model, "claude-sonnet-5-5"); | |
| 74 | assert.equal(findOffered("anthropic/claude-sonnet-5-5", OFFERED)?.model, "claude-sonnet-5-5"); | |
| 75 | assert.equal(findOffered("workers-ai/@cf/openai/gpt-oss-120b", OFFERED)?.provider, "workers-ai"); | |
| 76 | assert.equal(findOffered("@cf/openai/gpt-oss-120b", OFFERED)?.provider, "workers-ai"); | |
| 77 | // A provider that does not offer it. | |
| 78 | assert.equal(findOffered("workers-ai/claude-sonnet-5-5", OFFERED), null); | |
| 79 | }); | |
| 80 | ||
| 81 | test("the workspace's own providers come first, then g1t's catalogue", () => { | |
| 82 | const providers = [OLLAMA, OPENAI, ANTHROPIC]; | |
| 83 | const own = routeModel("ollama/qwen3", "chat", providers, OFFERED); | |
| 84 | assert.ok(own.to === "own"); | |
| 85 | assert.equal(own.provider.id, "con_ollama"); | |
| 86 | assert.equal(own.model, "qwen3"); | |
| 87 | // Its own Anthropic key takes Claude, by its bare id or the catalogue's. | |
| 88 | const claude = routeModel("anthropic/claude-sonnet-5-5", "chat", providers, OFFERED); | |
| 89 | assert.ok(claude.to === "own" && claude.provider.id === "con_anthropic" && claude.model === "claude-sonnet-5-5"); | |
| 90 | // Without one, Claude is g1t's. | |
| 91 | const hosted = routeModel("claude-sonnet-5-5", "chat", [OLLAMA], OFFERED); | |
| 92 | assert.ok(hosted.to === "g1t" && hosted.api === "anthropic" && hosted.model === "claude-sonnet-5-5"); | |
| 93 | const open = routeModel("workers-ai/@cf/openai/gpt-oss-120b", "chat", [], OFFERED); | |
| 94 | assert.ok(open.to === "g1t" && open.api === "openai" && open.model === "@cf/openai/gpt-oss-120b"); | |
| 95 | }); | |
| 96 | ||
| 97 | test("a model nobody offers, or of the wrong kind, goes nowhere and says why", () => { | |
| 98 | const none = routeModel("gpt-5.5", "chat", [], OFFERED); | |
| 99 | assert.ok(none.to === "none"); | |
| 100 | assert.equal(none.status, 404); | |
| 101 | assert.match(none.message, /gpt-5\.5 is not offered/); | |
| 102 | assert.match(none.message, /anthropic\/claude-haiku-5-5, anthropic\/claude-sonnet-5-5/); | |
| 103 | assert.doesNotMatch(none.message, /bge-m3/); | |
| 104 | assert.match(none.message, /connect the workspace's own provider/); | |
| 105 | const embedChat = routeModel("@cf/baai/bge-m3", "chat", [], OFFERED); | |
| 106 | assert.ok(embedChat.to === "none" && embedChat.status === 400 && /embeddings model/.test(embedChat.message)); | |
| 107 | const chatEmbed = routeModel("claude-haiku-5-5", "embeddings", [], OFFERED); | |
| 108 | assert.ok(chatEmbed.to === "none" && /chat model/.test(chatEmbed.message)); | |
| 109 | const missing = routeModel(undefined, "chat", [], OFFERED); | |
| 110 | assert.ok(missing.to === "none" && missing.status === 400); | |
| 111 | const anthropicEmbed = routeModel("claude-x", "embeddings", [ANTHROPIC], OFFERED); | |
| 112 | assert.ok(anthropicEmbed.to === "none" && /no embeddings/.test(anthropicEmbed.message)); | |
| 113 | }); | |
| 114 | ||
| 115 | test("the models list: the workspace's own, then the catalogue with g1t's prices, each once", () => { | |
| 116 | const listed = listModels([OLLAMA, OPENAI], OFFERED); | |
| 117 | const ids = listed.map((m) => m.id); | |
| 118 | assert.deepEqual(ids, [ | |
| 119 | "ollama/llama3.3", | |
| 120 | "ollama/qwen3", | |
| 121 | "gpt-5.5", | |
| 122 | "gpt-5.5-mini", | |
| 123 | "anthropic/claude-haiku-5-5", | |
| 124 | "anthropic/claude-sonnet-5-5", | |
| 125 | "anthropic/claude-haiku-4-5", | |
| 126 | "workers-ai/@cf/openai/gpt-oss-120b", | |
| 127 | "workers-ai/@cf/baai/bge-m3", | |
| 128 | ]); | |
| 129 | const ollama = listed[0]!; | |
| 130 | assert.equal(ollama.billed_to, "workspace"); | |
| 131 | assert.equal(ollama.connection, "Ollama"); | |
| 132 | assert.equal(ollama.pricing, null); | |
| 133 | const haiku = listed.find((m) => m.id === "anthropic/claude-haiku-5-5")!; | |
| 134 | assert.equal(haiku.billed_to, "g1t"); | |
| 135 | assert.deepEqual(haiku.pricing, { | |
| 136 | currency: "usd", | |
| 137 | input: 0.1, | |
| 138 | output: 0.5, | |
| 139 | cache_read: 0.01, | |
| 140 | cache_write: 0.125, | |
| 141 | cache_write_1h: 0.2, | |
| 142 | long_prompt: { above_tokens: 100_000, input: 0.5, output: 2.5, cache_read: 0.05, cache_write: 0.625, cache_write_1h: 1 }, | |
| 143 | }); | |
| 144 | assert.equal(listed.find((m) => m.id.endsWith("bge-m3"))!.kind, "embeddings"); | |
| 145 | // With its own Anthropic key, the catalogue's Claude is the workspace's. | |
| 146 | const mine = listModels([ANTHROPIC], OFFERED).find((m) => m.id === "anthropic/claude-sonnet-5-5")!; | |
| 147 | assert.equal(mine.billed_to, "workspace"); | |
| 148 | assert.equal(mine.pricing, null); | |
| 149 | }); | |
| 150 | ||
| 151 | // --- Serving, end to end, with no network --------------------------------------- | |
| 152 | ||
| 153 | const WORKSPACE_TOKEN: User = { | |
| 154 | id: "wsp_1", | |
| 155 | username: "acme", | |
| 156 | kind: "workspace", | |
| 157 | workspaces: [{ slug: "acme", role: "member" }], | |
| 158 | token: { token_id: "tok_1", scopes: ["models:write"], name: "ci" }, | |
| 159 | }; | |
| 160 | ||
| 161 | type Sent = { url: string; headers: Headers; body: Record<string, unknown> }; | |
| 162 | ||
| 163 | function harness(options: { | |
| 164 | providers?: GatewayProvider[]; | |
| 165 | admit?: string | null; | |
| 166 | user?: User | null; | |
| 167 | answer: (sent: Sent) => Response; | |
| 168 | }) { | |
| 169 | const sent: Sent[] = []; | |
| 170 | const records: GatewayRecord[] = []; | |
| 171 | const pending: Promise<unknown>[] = []; | |
| 172 | let admitted = 0; | |
| 173 | const deps: GatewayDeps = { | |
| 174 | hosted: { AI_GATEWAY_ID: "g1t", CLOUDFLARE_ACCOUNT_ID: "acct", AI_GATEWAY_TOKEN: "aig-secret-token", WORKERS_AI_TOKEN: "wai-secret-token" }, | |
| 175 | caller: async () => (options.user === undefined ? WORKSPACE_TOKEN : options.user), | |
| 176 | providers: async () => options.providers ?? [], | |
| 177 | offered: async () => OFFERED, | |
| 178 | admit: async () => { | |
| 179 | admitted += 1; | |
| 180 | return options.admit ?? null; | |
| 181 | }, | |
| 182 | record: async (record) => { | |
| 183 | records.push(record); | |
| 184 | }, | |
| 185 | fetch: async (url, init) => { | |
| 186 | const one = { url, headers: new Headers(init.headers), body: JSON.parse(String(init.body)) as Record<string, unknown> }; | |
| 187 | sent.push(one); | |
| 188 | return options.answer(one); | |
| 189 | }, | |
| 190 | waitUntil: (promise) => { | |
| 191 | pending.push(promise); | |
| 192 | }, | |
| 193 | }; | |
| 194 | return { | |
| 195 | deps, | |
| 196 | sent, | |
| 197 | records, | |
| 198 | admitted: () => admitted, | |
| 199 | settled: () => Promise.all(pending), | |
| 200 | }; | |
| 201 | } | |
| 202 | ||
| 203 | const post = (path: string, body: unknown, headers: Record<string, string> = {}) => | |
| 204 | new Request(`https://models.g1t.sh${path}`, { | |
| 205 | method: "POST", | |
| 206 | headers: { authorization: "Bearer g1t_workspace_token", "content-type": "application/json", ...headers }, | |
| 207 | body: JSON.stringify(body), | |
| 208 | }); | |
| 209 | ||
| 210 | const json = (body: unknown, status = 200) => Response.json(body, { status }); | |
| 211 | const stream = (text: string) => new Response(text, { headers: { "content-type": "text/event-stream" } }); | |
| 212 | const sse = (events: object[]) => events.map((e) => `event: ${(e as { type: string }).type}\ndata: ${JSON.stringify(e)}\n\n`).join(""); | |
| 213 | ||
| 214 | const MESSAGE = { | |
| 215 | id: "msg_1", | |
| 216 | type: "message", | |
| 217 | role: "assistant", | |
| 218 | model: "claude-haiku-5-5", | |
| 219 | content: [{ type: "text", text: "Hello." }], | |
| 220 | stop_reason: "end_turn", | |
| 221 | usage: { input_tokens: 12, output_tokens: 4, cache_read_input_tokens: 100, cache_creation_input_tokens: 20, cache_creation: { ephemeral_5m_input_tokens: 5, ephemeral_1h_input_tokens: 15 } }, | |
| 222 | }; | |
| 223 | ||
| 224 | test("OpenAI's format reaches g1t's Claude, translated, and is charged by Claude's own usage", async () => { | |
| 225 | const h = harness({ answer: () => json(MESSAGE) }); | |
| 226 | const response = await serveGateway( | |
| 227 | post("/openai/v1/chat/completions", { model: "anthropic/claude-haiku-5-5", messages: [{ role: "user", content: "Hi" }], reasoning_effort: "low" }), | |
| 228 | "g1t_workspace_token", | |
| 229 | h.deps, | |
| 230 | ); | |
| 231 | assert.equal(response.status, 200); | |
| 232 | const completion = (await response.json()) as { object: string; model: string; choices: { message: { content: string } }[]; usage: unknown }; | |
| 233 | assert.equal(completion.object, "chat.completion"); | |
| 234 | assert.equal(completion.model, "anthropic/claude-haiku-5-5"); | |
| 235 | assert.equal(completion.choices[0]!.message.content, "Hello."); | |
| 236 | assert.match(response.headers.get("x-g1t-request-id") ?? "", /^gw_/); | |
| 237 | await h.settled(); | |
| 238 | ||
| 239 | const [sent] = h.sent; | |
| 240 | assert.equal(sent!.url, "https://gateway.ai.cloudflare.com/v1/acct/g1t/anthropic/v1/messages"); | |
| 241 | assert.equal(sent!.body.model, "claude-haiku-5-5"); | |
| 242 | assert.deepEqual(sent!.body.output_config, { effort: "low" }); | |
| 243 | assert.equal(sent!.headers.get("authorization"), null); | |
| 244 | assert.equal(sent!.headers.get("anthropic-version"), "2023-06-01"); | |
| 245 | assert.equal(sent!.headers.get("cf-aig-authorization"), "Bearer aig-secret-token"); | |
| 246 | assert.equal(h.admitted(), 1); | |
| 247 | ||
| 248 | const [record] = h.records; | |
| 249 | assert.equal(record!.format, "openai"); | |
| 250 | assert.equal(record!.provider, "anthropic"); | |
| 251 | assert.equal(record!.model, "claude-haiku-5-5"); | |
| 252 | assert.equal(record!.ownKey, false); | |
| 253 | assert.equal(record!.connection, null); | |
| 254 | assert.deepEqual([record!.input, record!.output, record!.cacheRead, record!.cacheWrite, record!.cacheWriteHour], [12, 4, 100, 20, 15]); | |
| 255 | }); | |
| 256 | ||
| 257 | test("a streamed OpenAI-format request to Claude streams chat chunks and counts the tokens", async () => { | |
| 258 | const events = sse([ | |
| 259 | { type: "message_start", message: { id: "msg_2", usage: { input_tokens: 7, cache_read_input_tokens: 50, output_tokens: 1 } } }, | |
| 260 | { type: "content_block_start", index: 0, content_block: { type: "text", text: "" } }, | |
| 261 | { type: "content_block_delta", index: 0, delta: { type: "text_delta", text: "Hi there" } }, | |
| 262 | { type: "content_block_stop", index: 0 }, | |
| 263 | { type: "message_delta", delta: { stop_reason: "end_turn" }, usage: { output_tokens: 9 } }, | |
| 264 | { type: "message_stop" }, | |
| 265 | ]); | |
| 266 | const h = harness({ answer: () => stream(events) }); | |
| 267 | const response = await serveGateway( | |
| 268 | post("/openai/v1/chat/completions", { model: "claude-sonnet-5-5", stream: true, stream_options: { include_usage: true }, messages: [{ role: "user", content: "Hi" }] }), | |
| 269 | "g1t_workspace_token", | |
| 270 | h.deps, | |
| 271 | ); | |
| 272 | assert.equal(response.headers.get("content-type"), "text/event-stream"); | |
| 273 | const text = await response.text(); | |
| 274 | assert.match(text, /"content":"Hi there"/); | |
| 275 | assert.match(text, /"finish_reason":"stop"/); | |
| 276 | assert.match(text, /"prompt_tokens":57/); | |
| 277 | assert.ok(text.trimEnd().endsWith("data: [DONE]")); | |
| 278 | await h.settled(); | |
| 279 | assert.equal(h.sent[0]!.body.stream, true); | |
| 280 | const [record] = h.records; | |
| 281 | assert.deepEqual([record!.input, record!.output, record!.cacheRead], [7, 9, 50]); | |
| 282 | assert.equal(record!.streamed, true); | |
| 283 | }); | |
| 284 | ||
| 285 | test("Anthropic's format reaches an open model on Workers AI, translated both ways", async () => { | |
| 286 | const h = harness({ | |
| 287 | answer: () => | |
| 288 | json({ | |
| 289 | id: "chatcmpl-1", | |
| 290 | model: "@cf/openai/gpt-oss-120b", | |
| 291 | choices: [{ message: { role: "assistant", content: null, tool_calls: [{ id: "call_1", type: "function", function: { name: "f", arguments: '{"a":1}' } }] }, finish_reason: "tool_calls" }], | |
| 292 | usage: { prompt_tokens: 200, completion_tokens: 30, prompt_tokens_details: { cached_tokens: 50 } }, | |
| 293 | }), | |
| 294 | }); | |
| 295 | const response = await serveGateway( | |
| 296 | post("/anthropic/v1/messages", { | |
| 297 | model: "workers-ai/@cf/openai/gpt-oss-120b", | |
| 298 | max_tokens: 100, | |
| 299 | tools: [{ name: "f", input_schema: { type: "object" } }], | |
| 300 | messages: [{ role: "user", content: "call f" }], | |
| 301 | }), | |
| 302 | "g1t_workspace_token", | |
| 303 | h.deps, | |
| 304 | ); | |
| 305 | const message = (await response.json()) as { type: string; stop_reason: string; content: { type: string; input?: unknown }[]; usage: unknown }; | |
| 306 | assert.equal(message.type, "message"); | |
| 307 | assert.equal(message.stop_reason, "tool_use"); | |
| 308 | assert.deepEqual(message.content[0], { type: "tool_use", id: "call_1", name: "f", input: { a: 1 } }); | |
| 309 | assert.deepEqual(message.usage, { input_tokens: 150, output_tokens: 30, cache_read_input_tokens: 50 }); | |
| 310 | await h.settled(); | |
| 311 | const [sent] = h.sent; | |
| 312 | assert.equal(sent!.url, "https://gateway.ai.cloudflare.com/v1/acct/g1t/workers-ai/v1/chat/completions"); | |
| 313 | assert.equal(sent!.headers.get("authorization"), "Bearer wai-secret-token"); | |
| 314 | assert.equal(sent!.body.model, "@cf/openai/gpt-oss-120b"); | |
| 315 | assert.equal((sent!.body.tools as unknown[]).length, 1); | |
| 316 | const [record] = h.records; | |
| 317 | assert.equal(record!.format, "anthropic"); | |
| 318 | assert.equal(record!.provider, "workers-ai"); | |
| 319 | assert.equal(record!.model, "@cf/openai/gpt-oss-120b"); | |
| 320 | assert.deepEqual([record!.input, record!.output, record!.cacheRead], [150, 30, 50]); | |
| 321 | }); | |
| 322 | ||
| 323 | test("the workspace's own endpoint gets the request with its key, never admitted or charged", async () => { | |
| 324 | const h = harness({ | |
| 325 | providers: [OLLAMA], | |
| 326 | answer: () => json({ id: "c", model: "qwen3", choices: [{ message: { role: "assistant", content: "ok" }, finish_reason: "stop" }], usage: { prompt_tokens: 5, completion_tokens: 2 } }), | |
| 327 | }); | |
| 328 | const response = await serveGateway(post("/openai/v1/chat/completions", { model: "ollama/qwen3", messages: [{ role: "user", content: "hi" }] }), "g1t_workspace_token", h.deps); | |
| 329 | assert.equal(response.status, 200); | |
| 330 | const body = await response.text(); | |
| 331 | assert.doesNotMatch(body, new RegExp(KEY)); | |
| 332 | await h.settled(); | |
| 333 | const [sent] = h.sent; | |
| 334 | assert.equal(sent!.url, "https://llm.acme.dev/v1/chat/completions"); | |
| 335 | assert.equal(sent!.body.model, "qwen3"); | |
| 336 | assert.equal(sent!.headers.get("authorization"), `Bearer ${KEY}`); | |
| 337 | assert.equal(sent!.headers.get("cf-aig-metadata"), null); | |
| 338 | assert.equal(h.admitted(), 0); | |
| 339 | const [record] = h.records; | |
| 340 | assert.equal(record!.ownKey, true); | |
| 341 | assert.equal(record!.provider, "openai_endpoint"); | |
| 342 | assert.equal(record!.connection, "Ollama"); | |
| 343 | assert.equal(record!.model, "qwen3"); | |
| 344 | assert.deepEqual([record!.input, record!.output], [5, 2]); | |
| 345 | }); | |
| 346 | ||
| 347 | test("an Anthropic-format request on the workspace's own Anthropic key passes through", async () => { | |
| 348 | const h = harness({ providers: [ANTHROPIC], answer: () => json(MESSAGE) }); | |
| 349 | const response = await serveGateway( | |
| 350 | post("/anthropic/v1/messages", { model: "claude-haiku-5-5", max_tokens: 10, tools: [{ type: "web_search_20260209", name: "web_search" }], messages: [] }, { "x-api-key": "g1t_workspace_token", "anthropic-beta": "x" }), | |
| 351 | "g1t_workspace_token", | |
| 352 | h.deps, | |
| 353 | ); | |
| 354 | // Server tools are the provider's to bill on the workspace's own key. | |
| 355 | assert.equal(response.status, 200); | |
| 356 | assert.deepEqual(await response.json(), MESSAGE); | |
| 357 | await h.settled(); | |
| 358 | const [sent] = h.sent; | |
| 359 | assert.equal(sent!.url, "https://api.anthropic.com/v1/messages"); | |
| 360 | assert.equal(sent!.headers.get("x-api-key"), KEY); | |
| 361 | assert.equal(sent!.headers.get("anthropic-beta"), "x"); | |
| 362 | assert.equal(h.admitted(), 0); | |
| 363 | assert.equal(h.records[0]!.ownKey, true); | |
| 364 | }); | |
| 365 | ||
| 366 | test("an own provider's error never shows its key, to the caller or the log", async () => { | |
| 367 | const leak = { error: { message: `Incorrect API key provided: ${KEY}. Also ${KEY.slice(0, 20)}...`, type: "invalid_request_error", code: "invalid_api_key" } }; | |
| 368 | const h = harness({ providers: [OPENAI], answer: () => json(leak, 401) }); | |
| 369 | const response = await serveGateway(post("/openai/v1/chat/completions", { model: "gpt-5.5", messages: [{ role: "user", content: "x" }] }), "g1t_workspace_token", h.deps); | |
| 370 | assert.equal(response.status, 401); | |
| 371 | const text = await response.text(); | |
| 372 | assert.doesNotMatch(text, new RegExp(KEY.slice(0, 12))); | |
| 373 | assert.match(text, /OpenAI refused the workspace's key/); | |
| 374 | assert.match(text, /\[redacted\]/); | |
| 375 | await h.settled(); | |
| 376 | assert.doesNotMatch(h.records[0]!.error ?? "", new RegExp(KEY.slice(0, 12))); | |
| 377 | assert.equal(h.records[0]!.status, 401); | |
| 378 | // And the Anthropic-format view of the same. | |
| 379 | const a = harness({ providers: [ANTHROPIC], answer: () => json({ type: "error", error: { type: "authentication_error", message: `invalid x-api-key ${KEY}` } }, 401) }); | |
| 380 | const refused = await serveGateway(post("/anthropic/v1/messages", { model: "claude-sonnet-5-5", max_tokens: 5, messages: [] }), "g1t_workspace_token", a.deps); | |
| 381 | const body = (await refused.json()) as { type: string; error: { type: string; message: string } }; | |
| 382 | assert.equal(body.error.type, "authentication_error"); | |
| 383 | assert.doesNotMatch(body.error.message, new RegExp(KEY.slice(0, 12))); | |
| 384 | }); | |
| 385 | ||
| 386 | test("refusals: unknown model, unpriced features, no credit, a token without the scope", async () => { | |
| 387 | const never = () => { | |
| 388 | throw new Error("nothing should be sent"); | |
| 389 | }; | |
| 390 | const unknown = harness({ answer: never }); | |
| 391 | const missing = await serveGateway(post("/openai/v1/chat/completions", { model: "gpt-5.5", messages: [] }), "g1t_workspace_token", unknown.deps); | |
| 392 | assert.equal(missing.status, 404); | |
| 393 | const error = ((await missing.json()) as { error: { code: string; message: string } }).error; | |
| 394 | assert.equal(error.code, "model_not_found"); | |
| 395 | assert.match(error.message, /not offered/); | |
| 396 | await unknown.settled(); | |
| 397 | assert.equal(unknown.records[0]!.status, 404); | |
| 398 | assert.equal(unknown.records[0]!.format, "openai"); | |
| 399 | ||
| 400 | const search = harness({ answer: never }); | |
| 401 | const searched = await serveGateway(post("/openai/v1/chat/completions", { model: "claude-sonnet-5-5", web_search_options: {}, messages: [] }), "g1t_workspace_token", search.deps); | |
| 402 | assert.equal(searched.status, 400); | |
| 403 | assert.equal(search.admitted(), 0); | |
| 404 | ||
| 405 | const fast = harness({ answer: never }); | |
| 406 | const fastAnswer = await serveGateway(post("/anthropic/v1/messages", { model: "claude-sonnet-5-5", speed: "fast", max_tokens: 1, messages: [] }), "g1t_workspace_token", fast.deps); | |
| 407 | assert.equal(fastAnswer.status, 400); | |
| 408 | assert.equal(((await fastAnswer.json()) as { error: { type: string } }).error.type, "invalid_request_error"); | |
| 409 | ||
| 410 | const broke = harness({ admit: "The acme workspace is out of AI credit.", answer: never }); | |
| 411 | const poor = await serveGateway(post("/openai/v1/chat/completions", { model: "claude-sonnet-5-5", messages: [] }), "g1t_workspace_token", broke.deps); | |
| 412 | assert.equal(poor.status, 402); | |
| 413 | assert.equal(((await poor.json()) as { error: { type: string } }).error.type, "insufficient_quota"); | |
| 414 | const poorAnthropic = await serveGateway(post("/anthropic/v1/messages", { model: "claude-sonnet-5-5", max_tokens: 1, messages: [] }), "g1t_workspace_token", broke.deps); | |
| 415 | assert.equal(((await poorAnthropic.json()) as { error: { type: string } }).error.type, "billing_error"); | |
| 416 | ||
| 417 | const reader = harness({ user: { ...WORKSPACE_TOKEN, token: { token_id: "tok_2", scopes: ["models:read"] } }, answer: never }); | |
| 418 | const forbidden = await serveGateway(post("/openai/v1/chat/completions", { model: "claude-sonnet-5-5", messages: [] }), "g1t_x", reader.deps); | |
| 419 | assert.equal(forbidden.status, 403); | |
| 420 | await reader.settled(); | |
| 421 | assert.equal(reader.records.length, 0); | |
| 422 | ||
| 423 | const translation = harness({ answer: never }); | |
| 424 | const many = await serveGateway(post("/openai/v1/chat/completions", { model: "claude-sonnet-5-5", n: 3, messages: [] }), "g1t_workspace_token", translation.deps); | |
| 425 | assert.equal(many.status, 400); | |
| 426 | assert.match(((await many.json()) as { error: { message: string } }).error.message, /answer once/); | |
| 427 | }); | |
| 428 | ||
| 429 | test("the models route lists what the workspace can use, in OpenAI's shape", async () => { | |
| 430 | const h = harness({ providers: [OPENAI], answer: () => json({}) }); | |
| 431 | const response = await serveGateway( | |
| 432 | new Request("https://models.g1t.sh/openai/v1/models", { headers: { authorization: "Bearer g1t_workspace_token" } }), | |
| 433 | "g1t_workspace_token", | |
| 434 | h.deps, | |
| 435 | ); | |
| 436 | const listed = (await response.json()) as { object: string; data: { id: string; billed_to: string }[] }; | |
| 437 | assert.equal(listed.object, "list"); | |
| 438 | assert.equal(listed.data[0]!.id, "gpt-5.5"); | |
| 439 | assert.ok(listed.data.some((m) => m.id === "anthropic/claude-haiku-5-5" && m.billed_to === "g1t")); | |
| 440 | assert.equal(h.sent.length, 0); | |
| 441 | }); | |
| 442 | ||
| 443 | test("embeddings go to Workers AI and are charged by their input", async () => { | |
| 444 | const h = harness({ answer: () => json({ object: "list", data: [{ object: "embedding", index: 0, embedding: [0.1, 0.2] }], model: "@cf/baai/bge-m3", usage: { prompt_tokens: 8, total_tokens: 8 } }) }); | |
| 445 | const response = await serveGateway(post("/openai/v1/embeddings", { model: "workers-ai/@cf/baai/bge-m3", input: "hello" }), "g1t_workspace_token", h.deps); | |
| 446 | assert.equal(response.status, 200); | |
| 447 | assert.equal(((await response.json()) as { data: unknown[] }).data.length, 1); | |
| 448 | await h.settled(); | |
| 449 | assert.equal(h.sent[0]!.url, "https://gateway.ai.cloudflare.com/v1/acct/g1t/workers-ai/v1/embeddings"); | |
| 450 | assert.equal(h.records[0]!.model, "@cf/baai/bge-m3"); | |
| 451 | assert.equal(h.records[0]!.input, 8); | |
| 452 | }); | |
| 453 | ||
| 454 | test("counting tokens for an open model is estimated, and never logged", async () => { | |
| 455 | const h = harness({ answer: () => json({}) }); | |
| 456 | const response = await serveGateway( | |
| 457 | post("/anthropic/v1/messages/count_tokens", { model: "@cf/openai/gpt-oss-120b", messages: [{ role: "user", content: "x".repeat(400) }] }), | |
| 458 | "g1t_workspace_token", | |
| 459 | h.deps, | |
| 460 | ); | |
| 461 | const counted = (await response.json()) as { input_tokens: number }; | |
| 462 | assert.ok(counted.input_tokens > 100); | |
| 463 | await h.settled(); | |
| 464 | assert.equal(h.sent.length, 0); | |
| 465 | assert.equal(h.records.length, 0); | |
| 466 | }); | |
| 467 | ||
| 468 | test("a route the gateway does not have is answered in the caller's format", async () => { | |
| 469 | const h = harness({ answer: () => json({}) }); | |
| 470 | const openai = await serveGateway(post("/openai/v1/responses", {}), "g1t_workspace_token", h.deps); | |
| 471 | assert.equal(openai.status, 404); | |
| 472 | assert.match(((await openai.json()) as { error: { message: string } }).error.message, /chat\/completions/); | |
| 473 | const anthropic = await serveGateway(post("/anthropic/v1/messages/batches", {}), "g1t_workspace_token", h.deps); | |
| 474 | assert.equal(((await anthropic.json()) as { type: string }).type, "error"); | |
| 475 | }); |