| 1 | import assert from "node:assert/strict"; |
| 2 | import { test } from "node:test"; |
| 3 | |
| 4 | import type { GatewayModel, GatewayProvider, GatewayRecord, User } from "@g1t/contracts"; |
| 5 | |
| 6 | import { findOffered, listModels, matchPattern, routeModel } from "./catalogue.ts"; |
| 7 | import { type GatewayDeps, serveGateway } from "./serve.ts"; |
| 8 | |
| 9 | // --- The catalogue and the workspace's own providers --------------------------- |
| 10 | |
| 11 | const model = (row: Partial<GatewayModel> & Pick<GatewayModel, "model" | "name" | "provider">): GatewayModel => ({ |
| 12 | kind: "chat", |
| 13 | inputMicros: 0, |
| 14 | outputMicros: 0, |
| 15 | cacheReadMicros: 0, |
| 16 | cacheWriteMicros: 0, |
| 17 | ...row, |
| 18 | }); |
| 19 | |
| 20 | const OFFERED: GatewayModel[] = [ |
| 21 | model({ |
| 22 | model: "claude-haiku-5-5", |
| 23 | name: "Claude Haiku 5.5", |
| 24 | provider: "anthropic", |
| 25 | inputMicros: 100_000, |
| 26 | outputMicros: 500_000, |
| 27 | cacheReadMicros: 10_000, |
| 28 | cacheWriteMicros: 125_000, |
| 29 | cacheWrite1hMicros: 200_000, |
| 30 | threshold: 100_000, |
| 31 | overInputMicros: 500_000, |
| 32 | overOutputMicros: 2_500_000, |
| 33 | overCacheReadMicros: 50_000, |
| 34 | overCacheWriteMicros: 625_000, |
| 35 | overCacheWrite1hMicros: 1_000_000, |
| 36 | }), |
| 37 | model({ model: "claude-sonnet-5-5", name: "Claude Sonnet 5.5", provider: "anthropic", inputMicros: 2_000_000, outputMicros: 10_000_000 }), |
| 38 | model({ model: "claude-haiku-4-5", name: "Claude Haiku 4.5", provider: "anthropic" }), |
| 39 | model({ model: "claude-haiku-4-5-20251001", name: "Claude Haiku 4.5", provider: "anthropic" }), |
| 40 | model({ model: "@cf/openai/gpt-oss-120b", name: "gpt-oss-120b", provider: "workers-ai", inputMicros: 350_000, outputMicros: 750_000 }), |
| 41 | model({ model: "@cf/baai/bge-m3", name: "BGE M3", provider: "workers-ai", kind: "embeddings", inputMicros: 12_000 }), |
| 42 | ]; |
| 43 | |
| 44 | const KEY = "sk-own-0123456789abcdefghij"; |
| 45 | |
| 46 | const provider = (row: Partial<GatewayProvider> & Pick<GatewayProvider, "provider" | "api" | "patterns">): GatewayProvider => ({ |
| 47 | id: "con_1", |
| 48 | name: "Ours", |
| 49 | official: false, |
| 50 | baseUrl: "https://llm.acme.dev/v1", |
| 51 | apiKey: KEY, |
| 52 | authHeader: "authorization", |
| 53 | gatewayToken: null, |
| 54 | models: [], |
| 55 | ...row, |
| 56 | }); |
| 57 | |
| 58 | const OLLAMA = provider({ id: "con_ollama", name: "Ollama", provider: "openai_endpoint", api: "openai", patterns: ["ollama/*"], models: ["llama3.3", "qwen3"] }); |
| 59 | const OPENAI = provider({ id: "con_openai", name: "OpenAI", provider: "openai", api: "openai", official: true, baseUrl: "https://api.openai.com/v1", patterns: ["gpt-*"], models: ["gpt-5.5", "gpt-5.5-mini"] }); |
| 60 | const ANTHROPIC = provider({ id: "con_anthropic", name: "Our Anthropic key", provider: "anthropic", api: "anthropic", baseUrl: "https://api.anthropic.com", authHeader: "x-api-key", patterns: ["claude-*"] }); |
| 61 | |
| 62 | test("patterns take models by id, prefix or namespace", () => { |
| 63 | assert.equal(matchPattern("gpt-5.5", "gpt-5.5"), "gpt-5.5"); |
| 64 | assert.equal(matchPattern("gpt-5.5", "gpt-5.5-mini"), null); |
| 65 | assert.equal(matchPattern("gpt-*", "gpt-5.5-mini"), "gpt-5.5-mini"); |
| 66 | assert.equal(matchPattern("ollama/*", "ollama/llama3.3"), "llama3.3"); |
| 67 | assert.equal(matchPattern("ollama/*", "ollama/"), null); |
| 68 | assert.equal(matchPattern("ollama/*", "llama3.3"), null); |
| 69 | assert.equal(matchPattern("*", "anything"), "anything"); |
| 70 | }); |
| 71 | |
| 72 | test("a catalogue model is found by its own id or with its provider in front", () => { |
| 73 | assert.equal(findOffered("claude-sonnet-5-5", OFFERED)?.model, "claude-sonnet-5-5"); |
| 74 | assert.equal(findOffered("anthropic/claude-sonnet-5-5", OFFERED)?.model, "claude-sonnet-5-5"); |
| 75 | assert.equal(findOffered("workers-ai/@cf/openai/gpt-oss-120b", OFFERED)?.provider, "workers-ai"); |
| 76 | assert.equal(findOffered("@cf/openai/gpt-oss-120b", OFFERED)?.provider, "workers-ai"); |
| 77 | // A provider that does not offer it. |
| 78 | assert.equal(findOffered("workers-ai/claude-sonnet-5-5", OFFERED), null); |
| 79 | }); |
| 80 | |
| 81 | test("the workspace's own providers come first, then g1t's catalogue", () => { |
| 82 | const providers = [OLLAMA, OPENAI, ANTHROPIC]; |
| 83 | const own = routeModel("ollama/qwen3", "chat", providers, OFFERED); |
| 84 | assert.ok(own.to === "own"); |
| 85 | assert.equal(own.provider.id, "con_ollama"); |
| 86 | assert.equal(own.model, "qwen3"); |
| 87 | // Its own Anthropic key takes Claude, by its bare id or the catalogue's. |
| 88 | const claude = routeModel("anthropic/claude-sonnet-5-5", "chat", providers, OFFERED); |
| 89 | assert.ok(claude.to === "own" && claude.provider.id === "con_anthropic" && claude.model === "claude-sonnet-5-5"); |
| 90 | // Without one, Claude is g1t's. |
| 91 | const hosted = routeModel("claude-sonnet-5-5", "chat", [OLLAMA], OFFERED); |
| 92 | assert.ok(hosted.to === "g1t" && hosted.api === "anthropic" && hosted.model === "claude-sonnet-5-5"); |
| 93 | const open = routeModel("workers-ai/@cf/openai/gpt-oss-120b", "chat", [], OFFERED); |
| 94 | assert.ok(open.to === "g1t" && open.api === "openai" && open.model === "@cf/openai/gpt-oss-120b"); |
| 95 | }); |
| 96 | |
| 97 | test("a model nobody offers, or of the wrong kind, goes nowhere and says why", () => { |
| 98 | const none = routeModel("gpt-5.5", "chat", [], OFFERED); |
| 99 | assert.ok(none.to === "none"); |
| 100 | assert.equal(none.status, 404); |
| 101 | assert.match(none.message, /gpt-5\.5 is not offered/); |
| 102 | assert.match(none.message, /anthropic\/claude-haiku-5-5, anthropic\/claude-sonnet-5-5/); |
| 103 | assert.doesNotMatch(none.message, /bge-m3/); |
| 104 | assert.match(none.message, /connect the workspace's own provider/); |
| 105 | const embedChat = routeModel("@cf/baai/bge-m3", "chat", [], OFFERED); |
| 106 | assert.ok(embedChat.to === "none" && embedChat.status === 400 && /embeddings model/.test(embedChat.message)); |
| 107 | const chatEmbed = routeModel("claude-haiku-5-5", "embeddings", [], OFFERED); |
| 108 | assert.ok(chatEmbed.to === "none" && /chat model/.test(chatEmbed.message)); |
| 109 | const missing = routeModel(undefined, "chat", [], OFFERED); |
| 110 | assert.ok(missing.to === "none" && missing.status === 400); |
| 111 | const anthropicEmbed = routeModel("claude-x", "embeddings", [ANTHROPIC], OFFERED); |
| 112 | assert.ok(anthropicEmbed.to === "none" && /no embeddings/.test(anthropicEmbed.message)); |
| 113 | }); |
| 114 | |
| 115 | test("the models list: the workspace's own, then the catalogue with g1t's prices, each once", () => { |
| 116 | const listed = listModels([OLLAMA, OPENAI], OFFERED); |
| 117 | const ids = listed.map((m) => m.id); |
| 118 | assert.deepEqual(ids, [ |
| 119 | "ollama/llama3.3", |
| 120 | "ollama/qwen3", |
| 121 | "gpt-5.5", |
| 122 | "gpt-5.5-mini", |
| 123 | "anthropic/claude-haiku-5-5", |
| 124 | "anthropic/claude-sonnet-5-5", |
| 125 | "anthropic/claude-haiku-4-5", |
| 126 | "workers-ai/@cf/openai/gpt-oss-120b", |
| 127 | "workers-ai/@cf/baai/bge-m3", |
| 128 | ]); |
| 129 | const ollama = listed[0]!; |
| 130 | assert.equal(ollama.billed_to, "workspace"); |
| 131 | assert.equal(ollama.connection, "Ollama"); |
| 132 | assert.equal(ollama.pricing, null); |
| 133 | const haiku = listed.find((m) => m.id === "anthropic/claude-haiku-5-5")!; |
| 134 | assert.equal(haiku.billed_to, "g1t"); |
| 135 | assert.deepEqual(haiku.pricing, { |
| 136 | currency: "usd", |
| 137 | input: 0.1, |
| 138 | output: 0.5, |
| 139 | cache_read: 0.01, |
| 140 | cache_write: 0.125, |
| 141 | cache_write_1h: 0.2, |
| 142 | long_prompt: { above_tokens: 100_000, input: 0.5, output: 2.5, cache_read: 0.05, cache_write: 0.625, cache_write_1h: 1 }, |
| 143 | }); |
| 144 | assert.equal(listed.find((m) => m.id.endsWith("bge-m3"))!.kind, "embeddings"); |
| 145 | // With its own Anthropic key, the catalogue's Claude is the workspace's. |
| 146 | const mine = listModels([ANTHROPIC], OFFERED).find((m) => m.id === "anthropic/claude-sonnet-5-5")!; |
| 147 | assert.equal(mine.billed_to, "workspace"); |
| 148 | assert.equal(mine.pricing, null); |
| 149 | }); |
| 150 | |
| 151 | // --- Serving, end to end, with no network --------------------------------------- |
| 152 | |
| 153 | const WORKSPACE_TOKEN: User = { |
| 154 | id: "wsp_1", |
| 155 | username: "acme", |
| 156 | kind: "workspace", |
| 157 | workspaces: [{ slug: "acme", role: "member" }], |
| 158 | token: { token_id: "tok_1", scopes: ["models:write"], name: "ci" }, |
| 159 | }; |
| 160 | |
| 161 | type Sent = { url: string; headers: Headers; body: Record<string, unknown> }; |
| 162 | |
| 163 | function harness(options: { |
| 164 | providers?: GatewayProvider[]; |
| 165 | admit?: string | null; |
| 166 | user?: User | null; |
| 167 | answer: (sent: Sent) => Response; |
| 168 | }) { |
| 169 | const sent: Sent[] = []; |
| 170 | const records: GatewayRecord[] = []; |
| 171 | const pending: Promise<unknown>[] = []; |
| 172 | let admitted = 0; |
| 173 | const deps: GatewayDeps = { |
| 174 | hosted: { AI_GATEWAY_ID: "g1t", CLOUDFLARE_ACCOUNT_ID: "acct", AI_GATEWAY_TOKEN: "aig-secret-token", WORKERS_AI_TOKEN: "wai-secret-token" }, |
| 175 | caller: async () => (options.user === undefined ? WORKSPACE_TOKEN : options.user), |
| 176 | providers: async () => options.providers ?? [], |
| 177 | offered: async () => OFFERED, |
| 178 | admit: async () => { |
| 179 | admitted += 1; |
| 180 | return options.admit ?? null; |
| 181 | }, |
| 182 | record: async (record) => { |
| 183 | records.push(record); |
| 184 | }, |
| 185 | fetch: async (url, init) => { |
| 186 | const one = { url, headers: new Headers(init.headers), body: JSON.parse(String(init.body)) as Record<string, unknown> }; |
| 187 | sent.push(one); |
| 188 | return options.answer(one); |
| 189 | }, |
| 190 | waitUntil: (promise) => { |
| 191 | pending.push(promise); |
| 192 | }, |
| 193 | }; |
| 194 | return { |
| 195 | deps, |
| 196 | sent, |
| 197 | records, |
| 198 | admitted: () => admitted, |
| 199 | settled: () => Promise.all(pending), |
| 200 | }; |
| 201 | } |
| 202 | |
| 203 | const post = (path: string, body: unknown, headers: Record<string, string> = {}) => |
| 204 | new Request(`https://models.g1t.sh${path}`, { |
| 205 | method: "POST", |
| 206 | headers: { authorization: "Bearer g1t_workspace_token", "content-type": "application/json", ...headers }, |
| 207 | body: JSON.stringify(body), |
| 208 | }); |
| 209 | |
| 210 | const json = (body: unknown, status = 200) => Response.json(body, { status }); |
| 211 | const stream = (text: string) => new Response(text, { headers: { "content-type": "text/event-stream" } }); |
| 212 | const sse = (events: object[]) => events.map((e) => `event: ${(e as { type: string }).type}\ndata: ${JSON.stringify(e)}\n\n`).join(""); |
| 213 | |
| 214 | const MESSAGE = { |
| 215 | id: "msg_1", |
| 216 | type: "message", |
| 217 | role: "assistant", |
| 218 | model: "claude-haiku-5-5", |
| 219 | content: [{ type: "text", text: "Hello." }], |
| 220 | stop_reason: "end_turn", |
| 221 | usage: { input_tokens: 12, output_tokens: 4, cache_read_input_tokens: 100, cache_creation_input_tokens: 20, cache_creation: { ephemeral_5m_input_tokens: 5, ephemeral_1h_input_tokens: 15 } }, |
| 222 | }; |
| 223 | |
| 224 | test("OpenAI's format reaches g1t's Claude, translated, and is charged by Claude's own usage", async () => { |
| 225 | const h = harness({ answer: () => json(MESSAGE) }); |
| 226 | const response = await serveGateway( |
| 227 | post("/openai/v1/chat/completions", { model: "anthropic/claude-haiku-5-5", messages: [{ role: "user", content: "Hi" }], reasoning_effort: "low" }), |
| 228 | "g1t_workspace_token", |
| 229 | h.deps, |
| 230 | ); |
| 231 | assert.equal(response.status, 200); |
| 232 | const completion = (await response.json()) as { object: string; model: string; choices: { message: { content: string } }[]; usage: unknown }; |
| 233 | assert.equal(completion.object, "chat.completion"); |
| 234 | assert.equal(completion.model, "anthropic/claude-haiku-5-5"); |
| 235 | assert.equal(completion.choices[0]!.message.content, "Hello."); |
| 236 | assert.match(response.headers.get("x-g1t-request-id") ?? "", /^gw_/); |
| 237 | await h.settled(); |
| 238 | |
| 239 | const [sent] = h.sent; |
| 240 | assert.equal(sent!.url, "https://gateway.ai.cloudflare.com/v1/acct/g1t/anthropic/v1/messages"); |
| 241 | assert.equal(sent!.body.model, "claude-haiku-5-5"); |
| 242 | assert.deepEqual(sent!.body.output_config, { effort: "low" }); |
| 243 | assert.equal(sent!.headers.get("authorization"), null); |
| 244 | assert.equal(sent!.headers.get("anthropic-version"), "2023-06-01"); |
| 245 | assert.equal(sent!.headers.get("cf-aig-authorization"), "Bearer aig-secret-token"); |
| 246 | assert.equal(h.admitted(), 1); |
| 247 | |
| 248 | const [record] = h.records; |
| 249 | assert.equal(record!.format, "openai"); |
| 250 | assert.equal(record!.provider, "anthropic"); |
| 251 | assert.equal(record!.model, "claude-haiku-5-5"); |
| 252 | assert.equal(record!.ownKey, false); |
| 253 | assert.equal(record!.connection, null); |
| 254 | assert.deepEqual([record!.input, record!.output, record!.cacheRead, record!.cacheWrite, record!.cacheWriteHour], [12, 4, 100, 20, 15]); |
| 255 | }); |
| 256 | |
| 257 | test("a streamed OpenAI-format request to Claude streams chat chunks and counts the tokens", async () => { |
| 258 | const events = sse([ |
| 259 | { type: "message_start", message: { id: "msg_2", usage: { input_tokens: 7, cache_read_input_tokens: 50, output_tokens: 1 } } }, |
| 260 | { type: "content_block_start", index: 0, content_block: { type: "text", text: "" } }, |
| 261 | { type: "content_block_delta", index: 0, delta: { type: "text_delta", text: "Hi there" } }, |
| 262 | { type: "content_block_stop", index: 0 }, |
| 263 | { type: "message_delta", delta: { stop_reason: "end_turn" }, usage: { output_tokens: 9 } }, |
| 264 | { type: "message_stop" }, |
| 265 | ]); |
| 266 | const h = harness({ answer: () => stream(events) }); |
| 267 | const response = await serveGateway( |
| 268 | post("/openai/v1/chat/completions", { model: "claude-sonnet-5-5", stream: true, stream_options: { include_usage: true }, messages: [{ role: "user", content: "Hi" }] }), |
| 269 | "g1t_workspace_token", |
| 270 | h.deps, |
| 271 | ); |
| 272 | assert.equal(response.headers.get("content-type"), "text/event-stream"); |
| 273 | const text = await response.text(); |
| 274 | assert.match(text, /"content":"Hi there"/); |
| 275 | assert.match(text, /"finish_reason":"stop"/); |
| 276 | assert.match(text, /"prompt_tokens":57/); |
| 277 | assert.ok(text.trimEnd().endsWith("data: [DONE]")); |
| 278 | await h.settled(); |
| 279 | assert.equal(h.sent[0]!.body.stream, true); |
| 280 | const [record] = h.records; |
| 281 | assert.deepEqual([record!.input, record!.output, record!.cacheRead], [7, 9, 50]); |
| 282 | assert.equal(record!.streamed, true); |
| 283 | }); |
| 284 | |
| 285 | test("Anthropic's format reaches an open model on Workers AI, translated both ways", async () => { |
| 286 | const h = harness({ |
| 287 | answer: () => |
| 288 | json({ |
| 289 | id: "chatcmpl-1", |
| 290 | model: "@cf/openai/gpt-oss-120b", |
| 291 | choices: [{ message: { role: "assistant", content: null, tool_calls: [{ id: "call_1", type: "function", function: { name: "f", arguments: '{"a":1}' } }] }, finish_reason: "tool_calls" }], |
| 292 | usage: { prompt_tokens: 200, completion_tokens: 30, prompt_tokens_details: { cached_tokens: 50 } }, |
| 293 | }), |
| 294 | }); |
| 295 | const response = await serveGateway( |
| 296 | post("/anthropic/v1/messages", { |
| 297 | model: "workers-ai/@cf/openai/gpt-oss-120b", |
| 298 | max_tokens: 100, |
| 299 | tools: [{ name: "f", input_schema: { type: "object" } }], |
| 300 | messages: [{ role: "user", content: "call f" }], |
| 301 | }), |
| 302 | "g1t_workspace_token", |
| 303 | h.deps, |
| 304 | ); |
| 305 | const message = (await response.json()) as { type: string; stop_reason: string; content: { type: string; input?: unknown }[]; usage: unknown }; |
| 306 | assert.equal(message.type, "message"); |
| 307 | assert.equal(message.stop_reason, "tool_use"); |
| 308 | assert.deepEqual(message.content[0], { type: "tool_use", id: "call_1", name: "f", input: { a: 1 } }); |
| 309 | assert.deepEqual(message.usage, { input_tokens: 150, output_tokens: 30, cache_read_input_tokens: 50 }); |
| 310 | await h.settled(); |
| 311 | const [sent] = h.sent; |
| 312 | assert.equal(sent!.url, "https://gateway.ai.cloudflare.com/v1/acct/g1t/workers-ai/v1/chat/completions"); |
| 313 | assert.equal(sent!.headers.get("authorization"), "Bearer wai-secret-token"); |
| 314 | assert.equal(sent!.body.model, "@cf/openai/gpt-oss-120b"); |
| 315 | assert.equal((sent!.body.tools as unknown[]).length, 1); |
| 316 | const [record] = h.records; |
| 317 | assert.equal(record!.format, "anthropic"); |
| 318 | assert.equal(record!.provider, "workers-ai"); |
| 319 | assert.equal(record!.model, "@cf/openai/gpt-oss-120b"); |
| 320 | assert.deepEqual([record!.input, record!.output, record!.cacheRead], [150, 30, 50]); |
| 321 | }); |
| 322 | |
| 323 | test("the workspace's own endpoint gets the request with its key, never admitted or charged", async () => { |
| 324 | const h = harness({ |
| 325 | providers: [OLLAMA], |
| 326 | answer: () => json({ id: "c", model: "qwen3", choices: [{ message: { role: "assistant", content: "ok" }, finish_reason: "stop" }], usage: { prompt_tokens: 5, completion_tokens: 2 } }), |
| 327 | }); |
| 328 | const response = await serveGateway(post("/openai/v1/chat/completions", { model: "ollama/qwen3", messages: [{ role: "user", content: "hi" }] }), "g1t_workspace_token", h.deps); |
| 329 | assert.equal(response.status, 200); |
| 330 | const body = await response.text(); |
| 331 | assert.doesNotMatch(body, new RegExp(KEY)); |
| 332 | await h.settled(); |
| 333 | const [sent] = h.sent; |
| 334 | assert.equal(sent!.url, "https://llm.acme.dev/v1/chat/completions"); |
| 335 | assert.equal(sent!.body.model, "qwen3"); |
| 336 | assert.equal(sent!.headers.get("authorization"), `Bearer ${KEY}`); |
| 337 | assert.equal(sent!.headers.get("cf-aig-metadata"), null); |
| 338 | assert.equal(h.admitted(), 0); |
| 339 | const [record] = h.records; |
| 340 | assert.equal(record!.ownKey, true); |
| 341 | assert.equal(record!.provider, "openai_endpoint"); |
| 342 | assert.equal(record!.connection, "Ollama"); |
| 343 | assert.equal(record!.model, "qwen3"); |
| 344 | assert.deepEqual([record!.input, record!.output], [5, 2]); |
| 345 | }); |
| 346 | |
| 347 | test("an Anthropic-format request on the workspace's own Anthropic key passes through", async () => { |
| 348 | const h = harness({ providers: [ANTHROPIC], answer: () => json(MESSAGE) }); |
| 349 | const response = await serveGateway( |
| 350 | post("/anthropic/v1/messages", { model: "claude-haiku-5-5", max_tokens: 10, tools: [{ type: "web_search_20260209", name: "web_search" }], messages: [] }, { "x-api-key": "g1t_workspace_token", "anthropic-beta": "x" }), |
| 351 | "g1t_workspace_token", |
| 352 | h.deps, |
| 353 | ); |
| 354 | // Server tools are the provider's to bill on the workspace's own key. |
| 355 | assert.equal(response.status, 200); |
| 356 | assert.deepEqual(await response.json(), MESSAGE); |
| 357 | await h.settled(); |
| 358 | const [sent] = h.sent; |
| 359 | assert.equal(sent!.url, "https://api.anthropic.com/v1/messages"); |
| 360 | assert.equal(sent!.headers.get("x-api-key"), KEY); |
| 361 | assert.equal(sent!.headers.get("anthropic-beta"), "x"); |
| 362 | assert.equal(h.admitted(), 0); |
| 363 | assert.equal(h.records[0]!.ownKey, true); |
| 364 | }); |
| 365 | |
| 366 | test("an own provider's error never shows its key, to the caller or the log", async () => { |
| 367 | const leak = { error: { message: `Incorrect API key provided: ${KEY}. Also ${KEY.slice(0, 20)}...`, type: "invalid_request_error", code: "invalid_api_key" } }; |
| 368 | const h = harness({ providers: [OPENAI], answer: () => json(leak, 401) }); |
| 369 | const response = await serveGateway(post("/openai/v1/chat/completions", { model: "gpt-5.5", messages: [{ role: "user", content: "x" }] }), "g1t_workspace_token", h.deps); |
| 370 | assert.equal(response.status, 401); |
| 371 | const text = await response.text(); |
| 372 | assert.doesNotMatch(text, new RegExp(KEY.slice(0, 12))); |
| 373 | assert.match(text, /OpenAI refused the workspace's key/); |
| 374 | assert.match(text, /\[redacted\]/); |
| 375 | await h.settled(); |
| 376 | assert.doesNotMatch(h.records[0]!.error ?? "", new RegExp(KEY.slice(0, 12))); |
| 377 | assert.equal(h.records[0]!.status, 401); |
| 378 | // And the Anthropic-format view of the same. |
| 379 | const a = harness({ providers: [ANTHROPIC], answer: () => json({ type: "error", error: { type: "authentication_error", message: `invalid x-api-key ${KEY}` } }, 401) }); |
| 380 | const refused = await serveGateway(post("/anthropic/v1/messages", { model: "claude-sonnet-5-5", max_tokens: 5, messages: [] }), "g1t_workspace_token", a.deps); |
| 381 | const body = (await refused.json()) as { type: string; error: { type: string; message: string } }; |
| 382 | assert.equal(body.error.type, "authentication_error"); |
| 383 | assert.doesNotMatch(body.error.message, new RegExp(KEY.slice(0, 12))); |
| 384 | }); |
| 385 | |
| 386 | test("refusals: unknown model, unpriced features, no credit, a token without the scope", async () => { |
| 387 | const never = () => { |
| 388 | throw new Error("nothing should be sent"); |
| 389 | }; |
| 390 | const unknown = harness({ answer: never }); |
| 391 | const missing = await serveGateway(post("/openai/v1/chat/completions", { model: "gpt-5.5", messages: [] }), "g1t_workspace_token", unknown.deps); |
| 392 | assert.equal(missing.status, 404); |
| 393 | const error = ((await missing.json()) as { error: { code: string; message: string } }).error; |
| 394 | assert.equal(error.code, "model_not_found"); |
| 395 | assert.match(error.message, /not offered/); |
| 396 | await unknown.settled(); |
| 397 | assert.equal(unknown.records[0]!.status, 404); |
| 398 | assert.equal(unknown.records[0]!.format, "openai"); |
| 399 | |
| 400 | const search = harness({ answer: never }); |
| 401 | const searched = await serveGateway(post("/openai/v1/chat/completions", { model: "claude-sonnet-5-5", web_search_options: {}, messages: [] }), "g1t_workspace_token", search.deps); |
| 402 | assert.equal(searched.status, 400); |
| 403 | assert.equal(search.admitted(), 0); |
| 404 | |
| 405 | const fast = harness({ answer: never }); |
| 406 | const fastAnswer = await serveGateway(post("/anthropic/v1/messages", { model: "claude-sonnet-5-5", speed: "fast", max_tokens: 1, messages: [] }), "g1t_workspace_token", fast.deps); |
| 407 | assert.equal(fastAnswer.status, 400); |
| 408 | assert.equal(((await fastAnswer.json()) as { error: { type: string } }).error.type, "invalid_request_error"); |
| 409 | |
| 410 | const broke = harness({ admit: "The acme workspace is out of AI credit.", answer: never }); |
| 411 | const poor = await serveGateway(post("/openai/v1/chat/completions", { model: "claude-sonnet-5-5", messages: [] }), "g1t_workspace_token", broke.deps); |
| 412 | assert.equal(poor.status, 402); |
| 413 | assert.equal(((await poor.json()) as { error: { type: string } }).error.type, "insufficient_quota"); |
| 414 | const poorAnthropic = await serveGateway(post("/anthropic/v1/messages", { model: "claude-sonnet-5-5", max_tokens: 1, messages: [] }), "g1t_workspace_token", broke.deps); |
| 415 | assert.equal(((await poorAnthropic.json()) as { error: { type: string } }).error.type, "billing_error"); |
| 416 | |
| 417 | const reader = harness({ user: { ...WORKSPACE_TOKEN, token: { token_id: "tok_2", scopes: ["models:read"] } }, answer: never }); |
| 418 | const forbidden = await serveGateway(post("/openai/v1/chat/completions", { model: "claude-sonnet-5-5", messages: [] }), "g1t_x", reader.deps); |
| 419 | assert.equal(forbidden.status, 403); |
| 420 | await reader.settled(); |
| 421 | assert.equal(reader.records.length, 0); |
| 422 | |
| 423 | const translation = harness({ answer: never }); |
| 424 | const many = await serveGateway(post("/openai/v1/chat/completions", { model: "claude-sonnet-5-5", n: 3, messages: [] }), "g1t_workspace_token", translation.deps); |
| 425 | assert.equal(many.status, 400); |
| 426 | assert.match(((await many.json()) as { error: { message: string } }).error.message, /answer once/); |
| 427 | }); |
| 428 | |
| 429 | test("the models route lists what the workspace can use, in OpenAI's shape", async () => { |
| 430 | const h = harness({ providers: [OPENAI], answer: () => json({}) }); |
| 431 | const response = await serveGateway( |
| 432 | new Request("https://models.g1t.sh/openai/v1/models", { headers: { authorization: "Bearer g1t_workspace_token" } }), |
| 433 | "g1t_workspace_token", |
| 434 | h.deps, |
| 435 | ); |
| 436 | const listed = (await response.json()) as { object: string; data: { id: string; billed_to: string }[] }; |
| 437 | assert.equal(listed.object, "list"); |
| 438 | assert.equal(listed.data[0]!.id, "gpt-5.5"); |
| 439 | assert.ok(listed.data.some((m) => m.id === "anthropic/claude-haiku-5-5" && m.billed_to === "g1t")); |
| 440 | assert.equal(h.sent.length, 0); |
| 441 | }); |
| 442 | |
| 443 | test("embeddings go to Workers AI and are charged by their input", async () => { |
| 444 | const h = harness({ answer: () => json({ object: "list", data: [{ object: "embedding", index: 0, embedding: [0.1, 0.2] }], model: "@cf/baai/bge-m3", usage: { prompt_tokens: 8, total_tokens: 8 } }) }); |
| 445 | const response = await serveGateway(post("/openai/v1/embeddings", { model: "workers-ai/@cf/baai/bge-m3", input: "hello" }), "g1t_workspace_token", h.deps); |
| 446 | assert.equal(response.status, 200); |
| 447 | assert.equal(((await response.json()) as { data: unknown[] }).data.length, 1); |
| 448 | await h.settled(); |
| 449 | assert.equal(h.sent[0]!.url, "https://gateway.ai.cloudflare.com/v1/acct/g1t/workers-ai/v1/embeddings"); |
| 450 | assert.equal(h.records[0]!.model, "@cf/baai/bge-m3"); |
| 451 | assert.equal(h.records[0]!.input, 8); |
| 452 | }); |
| 453 | |
| 454 | test("counting tokens for an open model is estimated, and never logged", async () => { |
| 455 | const h = harness({ answer: () => json({}) }); |
| 456 | const response = await serveGateway( |
| 457 | post("/anthropic/v1/messages/count_tokens", { model: "@cf/openai/gpt-oss-120b", messages: [{ role: "user", content: "x".repeat(400) }] }), |
| 458 | "g1t_workspace_token", |
| 459 | h.deps, |
| 460 | ); |
| 461 | const counted = (await response.json()) as { input_tokens: number }; |
| 462 | assert.ok(counted.input_tokens > 100); |
| 463 | await h.settled(); |
| 464 | assert.equal(h.sent.length, 0); |
| 465 | assert.equal(h.records.length, 0); |
| 466 | }); |
| 467 | |
| 468 | test("a route the gateway does not have is answered in the caller's format", async () => { |
| 469 | const h = harness({ answer: () => json({}) }); |
| 470 | const openai = await serveGateway(post("/openai/v1/responses", {}), "g1t_workspace_token", h.deps); |
| 471 | assert.equal(openai.status, 404); |
| 472 | assert.match(((await openai.json()) as { error: { message: string } }).error.message, /chat\/completions/); |
| 473 | const anthropic = await serveGateway(post("/anthropic/v1/messages/batches", {}), "g1t_workspace_token", h.deps); |
| 474 | assert.equal(((await anthropic.json()) as { type: string }).type, "error"); |
| 475 | }); |