Skip to content

g1t/services/models/src/serve.test.ts

475 lines24,126 bytesCodeBlame
1import assert from "node:assert/strict";
2import { test } from "node:test";
3
4import type { GatewayModel, GatewayProvider, GatewayRecord, User } from "@g1t/contracts";
5
6import { findOffered, listModels, matchPattern, routeModel } from "./catalogue.ts";
7import { type GatewayDeps, serveGateway } from "./serve.ts";
8
9// --- The catalogue and the workspace's own providers ---------------------------
10
11const model = (row: Partial<GatewayModel> & Pick<GatewayModel, "model" | "name" | "provider">): GatewayModel => ({
12 kind: "chat",
13 inputMicros: 0,
14 outputMicros: 0,
15 cacheReadMicros: 0,
16 cacheWriteMicros: 0,
17 ...row,
18});
19
20const OFFERED: GatewayModel[] = [
21 model({
22 model: "claude-haiku-5-5",
23 name: "Claude Haiku 5.5",
24 provider: "anthropic",
25 inputMicros: 100_000,
26 outputMicros: 500_000,
27 cacheReadMicros: 10_000,
28 cacheWriteMicros: 125_000,
29 cacheWrite1hMicros: 200_000,
30 threshold: 100_000,
31 overInputMicros: 500_000,
32 overOutputMicros: 2_500_000,
33 overCacheReadMicros: 50_000,
34 overCacheWriteMicros: 625_000,
35 overCacheWrite1hMicros: 1_000_000,
36 }),
37 model({ model: "claude-sonnet-5-5", name: "Claude Sonnet 5.5", provider: "anthropic", inputMicros: 2_000_000, outputMicros: 10_000_000 }),
38 model({ model: "claude-haiku-4-5", name: "Claude Haiku 4.5", provider: "anthropic" }),
39 model({ model: "claude-haiku-4-5-20251001", name: "Claude Haiku 4.5", provider: "anthropic" }),
40 model({ model: "@cf/openai/gpt-oss-120b", name: "gpt-oss-120b", provider: "workers-ai", inputMicros: 350_000, outputMicros: 750_000 }),
41 model({ model: "@cf/baai/bge-m3", name: "BGE M3", provider: "workers-ai", kind: "embeddings", inputMicros: 12_000 }),
42];
43
44const KEY = "sk-own-0123456789abcdefghij";
45
46const provider = (row: Partial<GatewayProvider> & Pick<GatewayProvider, "provider" | "api" | "patterns">): GatewayProvider => ({
47 id: "con_1",
48 name: "Ours",
49 official: false,
50 baseUrl: "https://llm.acme.dev/v1",
51 apiKey: KEY,
52 authHeader: "authorization",
53 gatewayToken: null,
54 models: [],
55 ...row,
56});
57
58const OLLAMA = provider({ id: "con_ollama", name: "Ollama", provider: "openai_endpoint", api: "openai", patterns: ["ollama/*"], models: ["llama3.3", "qwen3"] });
59const OPENAI = provider({ id: "con_openai", name: "OpenAI", provider: "openai", api: "openai", official: true, baseUrl: "https://api.openai.com/v1", patterns: ["gpt-*"], models: ["gpt-5.5", "gpt-5.5-mini"] });
60const ANTHROPIC = provider({ id: "con_anthropic", name: "Our Anthropic key", provider: "anthropic", api: "anthropic", baseUrl: "https://api.anthropic.com", authHeader: "x-api-key", patterns: ["claude-*"] });
61
62test("patterns take models by id, prefix or namespace", () => {
63 assert.equal(matchPattern("gpt-5.5", "gpt-5.5"), "gpt-5.5");
64 assert.equal(matchPattern("gpt-5.5", "gpt-5.5-mini"), null);
65 assert.equal(matchPattern("gpt-*", "gpt-5.5-mini"), "gpt-5.5-mini");
66 assert.equal(matchPattern("ollama/*", "ollama/llama3.3"), "llama3.3");
67 assert.equal(matchPattern("ollama/*", "ollama/"), null);
68 assert.equal(matchPattern("ollama/*", "llama3.3"), null);
69 assert.equal(matchPattern("*", "anything"), "anything");
70});
71
72test("a catalogue model is found by its own id or with its provider in front", () => {
73 assert.equal(findOffered("claude-sonnet-5-5", OFFERED)?.model, "claude-sonnet-5-5");
74 assert.equal(findOffered("anthropic/claude-sonnet-5-5", OFFERED)?.model, "claude-sonnet-5-5");
75 assert.equal(findOffered("workers-ai/@cf/openai/gpt-oss-120b", OFFERED)?.provider, "workers-ai");
76 assert.equal(findOffered("@cf/openai/gpt-oss-120b", OFFERED)?.provider, "workers-ai");
77 // A provider that does not offer it.
78 assert.equal(findOffered("workers-ai/claude-sonnet-5-5", OFFERED), null);
79});
80
81test("the workspace's own providers come first, then g1t's catalogue", () => {
82 const providers = [OLLAMA, OPENAI, ANTHROPIC];
83 const own = routeModel("ollama/qwen3", "chat", providers, OFFERED);
84 assert.ok(own.to === "own");
85 assert.equal(own.provider.id, "con_ollama");
86 assert.equal(own.model, "qwen3");
87 // Its own Anthropic key takes Claude, by its bare id or the catalogue's.
88 const claude = routeModel("anthropic/claude-sonnet-5-5", "chat", providers, OFFERED);
89 assert.ok(claude.to === "own" && claude.provider.id === "con_anthropic" && claude.model === "claude-sonnet-5-5");
90 // Without one, Claude is g1t's.
91 const hosted = routeModel("claude-sonnet-5-5", "chat", [OLLAMA], OFFERED);
92 assert.ok(hosted.to === "g1t" && hosted.api === "anthropic" && hosted.model === "claude-sonnet-5-5");
93 const open = routeModel("workers-ai/@cf/openai/gpt-oss-120b", "chat", [], OFFERED);
94 assert.ok(open.to === "g1t" && open.api === "openai" && open.model === "@cf/openai/gpt-oss-120b");
95});
96
97test("a model nobody offers, or of the wrong kind, goes nowhere and says why", () => {
98 const none = routeModel("gpt-5.5", "chat", [], OFFERED);
99 assert.ok(none.to === "none");
100 assert.equal(none.status, 404);
101 assert.match(none.message, /gpt-5\.5 is not offered/);
102 assert.match(none.message, /anthropic\/claude-haiku-5-5, anthropic\/claude-sonnet-5-5/);
103 assert.doesNotMatch(none.message, /bge-m3/);
104 assert.match(none.message, /connect the workspace's own provider/);
105 const embedChat = routeModel("@cf/baai/bge-m3", "chat", [], OFFERED);
106 assert.ok(embedChat.to === "none" && embedChat.status === 400 && /embeddings model/.test(embedChat.message));
107 const chatEmbed = routeModel("claude-haiku-5-5", "embeddings", [], OFFERED);
108 assert.ok(chatEmbed.to === "none" && /chat model/.test(chatEmbed.message));
109 const missing = routeModel(undefined, "chat", [], OFFERED);
110 assert.ok(missing.to === "none" && missing.status === 400);
111 const anthropicEmbed = routeModel("claude-x", "embeddings", [ANTHROPIC], OFFERED);
112 assert.ok(anthropicEmbed.to === "none" && /no embeddings/.test(anthropicEmbed.message));
113});
114
115test("the models list: the workspace's own, then the catalogue with g1t's prices, each once", () => {
116 const listed = listModels([OLLAMA, OPENAI], OFFERED);
117 const ids = listed.map((m) => m.id);
118 assert.deepEqual(ids, [
119 "ollama/llama3.3",
120 "ollama/qwen3",
121 "gpt-5.5",
122 "gpt-5.5-mini",
123 "anthropic/claude-haiku-5-5",
124 "anthropic/claude-sonnet-5-5",
125 "anthropic/claude-haiku-4-5",
126 "workers-ai/@cf/openai/gpt-oss-120b",
127 "workers-ai/@cf/baai/bge-m3",
128 ]);
129 const ollama = listed[0]!;
130 assert.equal(ollama.billed_to, "workspace");
131 assert.equal(ollama.connection, "Ollama");
132 assert.equal(ollama.pricing, null);
133 const haiku = listed.find((m) => m.id === "anthropic/claude-haiku-5-5")!;
134 assert.equal(haiku.billed_to, "g1t");
135 assert.deepEqual(haiku.pricing, {
136 currency: "usd",
137 input: 0.1,
138 output: 0.5,
139 cache_read: 0.01,
140 cache_write: 0.125,
141 cache_write_1h: 0.2,
142 long_prompt: { above_tokens: 100_000, input: 0.5, output: 2.5, cache_read: 0.05, cache_write: 0.625, cache_write_1h: 1 },
143 });
144 assert.equal(listed.find((m) => m.id.endsWith("bge-m3"))!.kind, "embeddings");
145 // With its own Anthropic key, the catalogue's Claude is the workspace's.
146 const mine = listModels([ANTHROPIC], OFFERED).find((m) => m.id === "anthropic/claude-sonnet-5-5")!;
147 assert.equal(mine.billed_to, "workspace");
148 assert.equal(mine.pricing, null);
149});
150
151// --- Serving, end to end, with no network ---------------------------------------
152
153const WORKSPACE_TOKEN: User = {
154 id: "wsp_1",
155 username: "acme",
156 kind: "workspace",
157 workspaces: [{ slug: "acme", role: "member" }],
158 token: { token_id: "tok_1", scopes: ["models:write"], name: "ci" },
159};
160
161type Sent = { url: string; headers: Headers; body: Record<string, unknown> };
162
163function harness(options: {
164 providers?: GatewayProvider[];
165 admit?: string | null;
166 user?: User | null;
167 answer: (sent: Sent) => Response;
168}) {
169 const sent: Sent[] = [];
170 const records: GatewayRecord[] = [];
171 const pending: Promise<unknown>[] = [];
172 let admitted = 0;
173 const deps: GatewayDeps = {
174 hosted: { AI_GATEWAY_ID: "g1t", CLOUDFLARE_ACCOUNT_ID: "acct", AI_GATEWAY_TOKEN: "aig-secret-token", WORKERS_AI_TOKEN: "wai-secret-token" },
175 caller: async () => (options.user === undefined ? WORKSPACE_TOKEN : options.user),
176 providers: async () => options.providers ?? [],
177 offered: async () => OFFERED,
178 admit: async () => {
179 admitted += 1;
180 return options.admit ?? null;
181 },
182 record: async (record) => {
183 records.push(record);
184 },
185 fetch: async (url, init) => {
186 const one = { url, headers: new Headers(init.headers), body: JSON.parse(String(init.body)) as Record<string, unknown> };
187 sent.push(one);
188 return options.answer(one);
189 },
190 waitUntil: (promise) => {
191 pending.push(promise);
192 },
193 };
194 return {
195 deps,
196 sent,
197 records,
198 admitted: () => admitted,
199 settled: () => Promise.all(pending),
200 };
201}
202
203const post = (path: string, body: unknown, headers: Record<string, string> = {}) =>
204 new Request(`https://models.g1t.sh${path}`, {
205 method: "POST",
206 headers: { authorization: "Bearer g1t_workspace_token", "content-type": "application/json", ...headers },
207 body: JSON.stringify(body),
208 });
209
210const json = (body: unknown, status = 200) => Response.json(body, { status });
211const stream = (text: string) => new Response(text, { headers: { "content-type": "text/event-stream" } });
212const sse = (events: object[]) => events.map((e) => `event: ${(e as { type: string }).type}\ndata: ${JSON.stringify(e)}\n\n`).join("");
213
214const MESSAGE = {
215 id: "msg_1",
216 type: "message",
217 role: "assistant",
218 model: "claude-haiku-5-5",
219 content: [{ type: "text", text: "Hello." }],
220 stop_reason: "end_turn",
221 usage: { input_tokens: 12, output_tokens: 4, cache_read_input_tokens: 100, cache_creation_input_tokens: 20, cache_creation: { ephemeral_5m_input_tokens: 5, ephemeral_1h_input_tokens: 15 } },
222};
223
224test("OpenAI's format reaches g1t's Claude, translated, and is charged by Claude's own usage", async () => {
225 const h = harness({ answer: () => json(MESSAGE) });
226 const response = await serveGateway(
227 post("/openai/v1/chat/completions", { model: "anthropic/claude-haiku-5-5", messages: [{ role: "user", content: "Hi" }], reasoning_effort: "low" }),
228 "g1t_workspace_token",
229 h.deps,
230 );
231 assert.equal(response.status, 200);
232 const completion = (await response.json()) as { object: string; model: string; choices: { message: { content: string } }[]; usage: unknown };
233 assert.equal(completion.object, "chat.completion");
234 assert.equal(completion.model, "anthropic/claude-haiku-5-5");
235 assert.equal(completion.choices[0]!.message.content, "Hello.");
236 assert.match(response.headers.get("x-g1t-request-id") ?? "", /^gw_/);
237 await h.settled();
238
239 const [sent] = h.sent;
240 assert.equal(sent!.url, "https://gateway.ai.cloudflare.com/v1/acct/g1t/anthropic/v1/messages");
241 assert.equal(sent!.body.model, "claude-haiku-5-5");
242 assert.deepEqual(sent!.body.output_config, { effort: "low" });
243 assert.equal(sent!.headers.get("authorization"), null);
244 assert.equal(sent!.headers.get("anthropic-version"), "2023-06-01");
245 assert.equal(sent!.headers.get("cf-aig-authorization"), "Bearer aig-secret-token");
246 assert.equal(h.admitted(), 1);
247
248 const [record] = h.records;
249 assert.equal(record!.format, "openai");
250 assert.equal(record!.provider, "anthropic");
251 assert.equal(record!.model, "claude-haiku-5-5");
252 assert.equal(record!.ownKey, false);
253 assert.equal(record!.connection, null);
254 assert.deepEqual([record!.input, record!.output, record!.cacheRead, record!.cacheWrite, record!.cacheWriteHour], [12, 4, 100, 20, 15]);
255});
256
257test("a streamed OpenAI-format request to Claude streams chat chunks and counts the tokens", async () => {
258 const events = sse([
259 { type: "message_start", message: { id: "msg_2", usage: { input_tokens: 7, cache_read_input_tokens: 50, output_tokens: 1 } } },
260 { type: "content_block_start", index: 0, content_block: { type: "text", text: "" } },
261 { type: "content_block_delta", index: 0, delta: { type: "text_delta", text: "Hi there" } },
262 { type: "content_block_stop", index: 0 },
263 { type: "message_delta", delta: { stop_reason: "end_turn" }, usage: { output_tokens: 9 } },
264 { type: "message_stop" },
265 ]);
266 const h = harness({ answer: () => stream(events) });
267 const response = await serveGateway(
268 post("/openai/v1/chat/completions", { model: "claude-sonnet-5-5", stream: true, stream_options: { include_usage: true }, messages: [{ role: "user", content: "Hi" }] }),
269 "g1t_workspace_token",
270 h.deps,
271 );
272 assert.equal(response.headers.get("content-type"), "text/event-stream");
273 const text = await response.text();
274 assert.match(text, /"content":"Hi there"/);
275 assert.match(text, /"finish_reason":"stop"/);
276 assert.match(text, /"prompt_tokens":57/);
277 assert.ok(text.trimEnd().endsWith("data: [DONE]"));
278 await h.settled();
279 assert.equal(h.sent[0]!.body.stream, true);
280 const [record] = h.records;
281 assert.deepEqual([record!.input, record!.output, record!.cacheRead], [7, 9, 50]);
282 assert.equal(record!.streamed, true);
283});
284
285test("Anthropic's format reaches an open model on Workers AI, translated both ways", async () => {
286 const h = harness({
287 answer: () =>
288 json({
289 id: "chatcmpl-1",
290 model: "@cf/openai/gpt-oss-120b",
291 choices: [{ message: { role: "assistant", content: null, tool_calls: [{ id: "call_1", type: "function", function: { name: "f", arguments: '{"a":1}' } }] }, finish_reason: "tool_calls" }],
292 usage: { prompt_tokens: 200, completion_tokens: 30, prompt_tokens_details: { cached_tokens: 50 } },
293 }),
294 });
295 const response = await serveGateway(
296 post("/anthropic/v1/messages", {
297 model: "workers-ai/@cf/openai/gpt-oss-120b",
298 max_tokens: 100,
299 tools: [{ name: "f", input_schema: { type: "object" } }],
300 messages: [{ role: "user", content: "call f" }],
301 }),
302 "g1t_workspace_token",
303 h.deps,
304 );
305 const message = (await response.json()) as { type: string; stop_reason: string; content: { type: string; input?: unknown }[]; usage: unknown };
306 assert.equal(message.type, "message");
307 assert.equal(message.stop_reason, "tool_use");
308 assert.deepEqual(message.content[0], { type: "tool_use", id: "call_1", name: "f", input: { a: 1 } });
309 assert.deepEqual(message.usage, { input_tokens: 150, output_tokens: 30, cache_read_input_tokens: 50 });
310 await h.settled();
311 const [sent] = h.sent;
312 assert.equal(sent!.url, "https://gateway.ai.cloudflare.com/v1/acct/g1t/workers-ai/v1/chat/completions");
313 assert.equal(sent!.headers.get("authorization"), "Bearer wai-secret-token");
314 assert.equal(sent!.body.model, "@cf/openai/gpt-oss-120b");
315 assert.equal((sent!.body.tools as unknown[]).length, 1);
316 const [record] = h.records;
317 assert.equal(record!.format, "anthropic");
318 assert.equal(record!.provider, "workers-ai");
319 assert.equal(record!.model, "@cf/openai/gpt-oss-120b");
320 assert.deepEqual([record!.input, record!.output, record!.cacheRead], [150, 30, 50]);
321});
322
323test("the workspace's own endpoint gets the request with its key, never admitted or charged", async () => {
324 const h = harness({
325 providers: [OLLAMA],
326 answer: () => json({ id: "c", model: "qwen3", choices: [{ message: { role: "assistant", content: "ok" }, finish_reason: "stop" }], usage: { prompt_tokens: 5, completion_tokens: 2 } }),
327 });
328 const response = await serveGateway(post("/openai/v1/chat/completions", { model: "ollama/qwen3", messages: [{ role: "user", content: "hi" }] }), "g1t_workspace_token", h.deps);
329 assert.equal(response.status, 200);
330 const body = await response.text();
331 assert.doesNotMatch(body, new RegExp(KEY));
332 await h.settled();
333 const [sent] = h.sent;
334 assert.equal(sent!.url, "https://llm.acme.dev/v1/chat/completions");
335 assert.equal(sent!.body.model, "qwen3");
336 assert.equal(sent!.headers.get("authorization"), `Bearer ${KEY}`);
337 assert.equal(sent!.headers.get("cf-aig-metadata"), null);
338 assert.equal(h.admitted(), 0);
339 const [record] = h.records;
340 assert.equal(record!.ownKey, true);
341 assert.equal(record!.provider, "openai_endpoint");
342 assert.equal(record!.connection, "Ollama");
343 assert.equal(record!.model, "qwen3");
344 assert.deepEqual([record!.input, record!.output], [5, 2]);
345});
346
347test("an Anthropic-format request on the workspace's own Anthropic key passes through", async () => {
348 const h = harness({ providers: [ANTHROPIC], answer: () => json(MESSAGE) });
349 const response = await serveGateway(
350 post("/anthropic/v1/messages", { model: "claude-haiku-5-5", max_tokens: 10, tools: [{ type: "web_search_20260209", name: "web_search" }], messages: [] }, { "x-api-key": "g1t_workspace_token", "anthropic-beta": "x" }),
351 "g1t_workspace_token",
352 h.deps,
353 );
354 // Server tools are the provider's to bill on the workspace's own key.
355 assert.equal(response.status, 200);
356 assert.deepEqual(await response.json(), MESSAGE);
357 await h.settled();
358 const [sent] = h.sent;
359 assert.equal(sent!.url, "https://api.anthropic.com/v1/messages");
360 assert.equal(sent!.headers.get("x-api-key"), KEY);
361 assert.equal(sent!.headers.get("anthropic-beta"), "x");
362 assert.equal(h.admitted(), 0);
363 assert.equal(h.records[0]!.ownKey, true);
364});
365
366test("an own provider's error never shows its key, to the caller or the log", async () => {
367 const leak = { error: { message: `Incorrect API key provided: ${KEY}. Also ${KEY.slice(0, 20)}...`, type: "invalid_request_error", code: "invalid_api_key" } };
368 const h = harness({ providers: [OPENAI], answer: () => json(leak, 401) });
369 const response = await serveGateway(post("/openai/v1/chat/completions", { model: "gpt-5.5", messages: [{ role: "user", content: "x" }] }), "g1t_workspace_token", h.deps);
370 assert.equal(response.status, 401);
371 const text = await response.text();
372 assert.doesNotMatch(text, new RegExp(KEY.slice(0, 12)));
373 assert.match(text, /OpenAI refused the workspace's key/);
374 assert.match(text, /\[redacted\]/);
375 await h.settled();
376 assert.doesNotMatch(h.records[0]!.error ?? "", new RegExp(KEY.slice(0, 12)));
377 assert.equal(h.records[0]!.status, 401);
378 // And the Anthropic-format view of the same.
379 const a = harness({ providers: [ANTHROPIC], answer: () => json({ type: "error", error: { type: "authentication_error", message: `invalid x-api-key ${KEY}` } }, 401) });
380 const refused = await serveGateway(post("/anthropic/v1/messages", { model: "claude-sonnet-5-5", max_tokens: 5, messages: [] }), "g1t_workspace_token", a.deps);
381 const body = (await refused.json()) as { type: string; error: { type: string; message: string } };
382 assert.equal(body.error.type, "authentication_error");
383 assert.doesNotMatch(body.error.message, new RegExp(KEY.slice(0, 12)));
384});
385
386test("refusals: unknown model, unpriced features, no credit, a token without the scope", async () => {
387 const never = () => {
388 throw new Error("nothing should be sent");
389 };
390 const unknown = harness({ answer: never });
391 const missing = await serveGateway(post("/openai/v1/chat/completions", { model: "gpt-5.5", messages: [] }), "g1t_workspace_token", unknown.deps);
392 assert.equal(missing.status, 404);
393 const error = ((await missing.json()) as { error: { code: string; message: string } }).error;
394 assert.equal(error.code, "model_not_found");
395 assert.match(error.message, /not offered/);
396 await unknown.settled();
397 assert.equal(unknown.records[0]!.status, 404);
398 assert.equal(unknown.records[0]!.format, "openai");
399
400 const search = harness({ answer: never });
401 const searched = await serveGateway(post("/openai/v1/chat/completions", { model: "claude-sonnet-5-5", web_search_options: {}, messages: [] }), "g1t_workspace_token", search.deps);
402 assert.equal(searched.status, 400);
403 assert.equal(search.admitted(), 0);
404
405 const fast = harness({ answer: never });
406 const fastAnswer = await serveGateway(post("/anthropic/v1/messages", { model: "claude-sonnet-5-5", speed: "fast", max_tokens: 1, messages: [] }), "g1t_workspace_token", fast.deps);
407 assert.equal(fastAnswer.status, 400);
408 assert.equal(((await fastAnswer.json()) as { error: { type: string } }).error.type, "invalid_request_error");
409
410 const broke = harness({ admit: "The acme workspace is out of AI credit.", answer: never });
411 const poor = await serveGateway(post("/openai/v1/chat/completions", { model: "claude-sonnet-5-5", messages: [] }), "g1t_workspace_token", broke.deps);
412 assert.equal(poor.status, 402);
413 assert.equal(((await poor.json()) as { error: { type: string } }).error.type, "insufficient_quota");
414 const poorAnthropic = await serveGateway(post("/anthropic/v1/messages", { model: "claude-sonnet-5-5", max_tokens: 1, messages: [] }), "g1t_workspace_token", broke.deps);
415 assert.equal(((await poorAnthropic.json()) as { error: { type: string } }).error.type, "billing_error");
416
417 const reader = harness({ user: { ...WORKSPACE_TOKEN, token: { token_id: "tok_2", scopes: ["models:read"] } }, answer: never });
418 const forbidden = await serveGateway(post("/openai/v1/chat/completions", { model: "claude-sonnet-5-5", messages: [] }), "g1t_x", reader.deps);
419 assert.equal(forbidden.status, 403);
420 await reader.settled();
421 assert.equal(reader.records.length, 0);
422
423 const translation = harness({ answer: never });
424 const many = await serveGateway(post("/openai/v1/chat/completions", { model: "claude-sonnet-5-5", n: 3, messages: [] }), "g1t_workspace_token", translation.deps);
425 assert.equal(many.status, 400);
426 assert.match(((await many.json()) as { error: { message: string } }).error.message, /answer once/);
427});
428
429test("the models route lists what the workspace can use, in OpenAI's shape", async () => {
430 const h = harness({ providers: [OPENAI], answer: () => json({}) });
431 const response = await serveGateway(
432 new Request("https://models.g1t.sh/openai/v1/models", { headers: { authorization: "Bearer g1t_workspace_token" } }),
433 "g1t_workspace_token",
434 h.deps,
435 );
436 const listed = (await response.json()) as { object: string; data: { id: string; billed_to: string }[] };
437 assert.equal(listed.object, "list");
438 assert.equal(listed.data[0]!.id, "gpt-5.5");
439 assert.ok(listed.data.some((m) => m.id === "anthropic/claude-haiku-5-5" && m.billed_to === "g1t"));
440 assert.equal(h.sent.length, 0);
441});
442
443test("embeddings go to Workers AI and are charged by their input", async () => {
444 const h = harness({ answer: () => json({ object: "list", data: [{ object: "embedding", index: 0, embedding: [0.1, 0.2] }], model: "@cf/baai/bge-m3", usage: { prompt_tokens: 8, total_tokens: 8 } }) });
445 const response = await serveGateway(post("/openai/v1/embeddings", { model: "workers-ai/@cf/baai/bge-m3", input: "hello" }), "g1t_workspace_token", h.deps);
446 assert.equal(response.status, 200);
447 assert.equal(((await response.json()) as { data: unknown[] }).data.length, 1);
448 await h.settled();
449 assert.equal(h.sent[0]!.url, "https://gateway.ai.cloudflare.com/v1/acct/g1t/workers-ai/v1/embeddings");
450 assert.equal(h.records[0]!.model, "@cf/baai/bge-m3");
451 assert.equal(h.records[0]!.input, 8);
452});
453
454test("counting tokens for an open model is estimated, and never logged", async () => {
455 const h = harness({ answer: () => json({}) });
456 const response = await serveGateway(
457 post("/anthropic/v1/messages/count_tokens", { model: "@cf/openai/gpt-oss-120b", messages: [{ role: "user", content: "x".repeat(400) }] }),
458 "g1t_workspace_token",
459 h.deps,
460 );
461 const counted = (await response.json()) as { input_tokens: number };
462 assert.ok(counted.input_tokens > 100);
463 await h.settled();
464 assert.equal(h.sent.length, 0);
465 assert.equal(h.records.length, 0);
466});
467
468test("a route the gateway does not have is answered in the caller's format", async () => {
469 const h = harness({ answer: () => json({}) });
470 const openai = await serveGateway(post("/openai/v1/responses", {}), "g1t_workspace_token", h.deps);
471 assert.equal(openai.status, 404);
472 assert.match(((await openai.json()) as { error: { message: string } }).error.message, /chat\/completions/);
473 const anthropic = await serveGateway(post("/anthropic/v1/messages/batches", {}), "g1t_workspace_token", h.deps);
474 assert.equal(((await anthropic.json()) as { type: string }).type, "error");
475});