Skip to content

g1t/services/models/src/gateway.ts

364 lines15,832 bytesCodeBlame
1/**
2 * The AI Gateway: a workspace's own model requests, at the same address as
3 * its sandboxes' (`models.g1t.sh`), in either format:
4 *
5 * - `/anthropic/v1/messages` (and `/count_tokens`): Anthropic's Messages API.
6 * - `/openai/v1/chat/completions`, `/openai/v1/embeddings` and
7 * `/openai/v1/models`: OpenAI's.
8 *
9 * A request carries one of the workspace's access tokens (`g1t_…`) with
10 * the `models:write` scope, as `x-api-key` or `Authorization: Bearer`, so
11 * either format's SDKs and tools need only a base URL and a key. The model
12 * it names decides where it goes (`catalogue.ts`): one of the workspace's
13 * own providers under Integrations, where it costs nothing, or g1t's
14 * catalogue (Claude on Anthropic, open models on Workers AI), admitted by
15 * billing first (spend limit, AI credit) and charged at the model's price
16 * afterwards. Either format reaches either kind of provider: the proxy
17 * translates (`openai.ts`, `chat.ts`). Every request is logged, with its
18 * tokens, never its prompt or answer.
19 *
20 * What is here is the part that decides; `serve.ts` sends.
21 */
22import type { GatewayProvider, GatewayRecord, User } from "@g1t/contracts";
23
24import type { Dialect } from "./openai.ts";
25import type { HostedRouting } from "./route.ts";
26import { passedHeaders } from "./route.ts";
27import type { Tokens } from "./usage.ts";
28
29/** Anthropic's error types, by the statuses the gateway answers with. */
30export type ErrorType =
31 | "invalid_request_error"
32 | "authentication_error"
33 | "billing_error"
34 | "permission_error"
35 | "not_found_error"
36 | "rate_limit_error"
37 | "api_error";
38
39/** An error in the shape Anthropic's API and SDKs use. */
40export function anthropicError(status: number, type: ErrorType, message: string): Response {
41 return Response.json({ type: "error", error: { type, message } }, { status });
42}
43
44/** Anthropic's error type for a status. */
45export function anthropicErrorType(status: number): ErrorType {
46 if (status === 401) return "authentication_error";
47 if (status === 402) return "billing_error";
48 if (status === 403) return "permission_error";
49 if (status === 404) return "not_found_error";
50 if (status === 429) return "rate_limit_error";
51 if (status >= 500) return "api_error";
52 return "invalid_request_error";
53}
54
55/** Who a gateway request is for, from its token. */
56export type Caller = { workspace: string; tokenId: string; tokenName: string | null };
57
58/** The scope a token needs to send requests. */
59export const GATEWAY_SCOPE = "models:write";
60
61/**
62 * Who a token stands for, or why it cannot use the gateway: it is unknown
63 * or expired, it is not a workspace's own token, or it lacks
64 * `models:write`. A token with full access has every scope.
65 */
66export function callerOf(viewer: User | null): { caller: Caller } | { status: number; type: ErrorType; message: string } {
67 if (!viewer) {
68 return { status: 401, type: "authentication_error", message: "This access token is not valid, or it has expired or been deleted." };
69 }
70 if (viewer.kind !== "workspace" || !viewer.token) {
71 return {
72 status: 403,
73 type: "permission_error",
74 message: "The AI Gateway takes a workspace's access token, which its usage is charged to. An owner can make one under the workspace's Settings, Access tokens, with the models:write scope.",
75 };
76 }
77 const scopes = viewer.token.scopes;
78 if (scopes && !scopes.includes(GATEWAY_SCOPE)) {
79 return { status: 403, type: "permission_error", message: `This access token needs the ${GATEWAY_SCOPE} scope to use the AI Gateway.` };
80 }
81 return { caller: { workspace: viewer.username.toLowerCase(), tokenId: viewer.token.token_id, tokenName: viewer.token.name ?? null } };
82}
83
84/** The request formats, by their path. */
85export type Format = "anthropic" | "openai";
86
87/** What a request asks. */
88export type Operation = "messages" | "count_tokens" | "chat" | "embeddings" | "models";
89
90/** Which of the gateway's routes a path (after `/anthropic`) is, or null. */
91export function gatewayRoute(path: string): "messages" | "count_tokens" | null {
92 const bare = path.split("?")[0]!.replace(/\/+$/, "");
93 if (bare === "/v1/messages") return "messages";
94 if (bare === "/v1/messages/count_tokens") return "count_tokens";
95 return null;
96}
97
98/**
99 * The format and operation of a request by its whole path and method, or
100 * null for one the gateway does not answer.
101 */
102export function gatewayOperation(path: string, method: string): { format: Format; op: Operation } | null {
103 const bare = path.split("?")[0]!.replace(/\/+$/, "");
104 if (bare.startsWith("/anthropic/")) {
105 const op = method === "POST" ? gatewayRoute(bare.slice("/anthropic".length)) : null;
106 return op ? { format: "anthropic", op } : null;
107 }
108 if (bare === "/openai/v1/chat/completions" && method === "POST") return { format: "openai", op: "chat" };
109 if (bare === "/openai/v1/embeddings" && method === "POST") return { format: "openai", op: "embeddings" };
110 if (bare === "/openai/v1/models" && method === "GET") return { format: "openai", op: "models" };
111 return null;
112}
113
114/** What the gateway answers, per format, for a request to a route it does not have. */
115export const ROUTES: Record<Format, string> = {
116 anthropic: "The AI Gateway answers POST /anthropic/v1/messages and POST /anthropic/v1/messages/count_tokens in Anthropic's format.",
117 openai: "The AI Gateway answers POST /openai/v1/chat/completions, POST /openai/v1/embeddings and GET /openai/v1/models in OpenAI's format.",
118};
119
120/**
121 * Tool types that run on the caller's side, so cost only their tokens. A
122 * tool with no type is the caller's own.
123 */
124const CLIENT_TOOLS = ["custom", "bash_", "text_editor_", "computer_", "memory_"];
125
126const OWN = "Use it with the workspace's own Anthropic key, under Integrations.";
127
128/**
129 * Why an Anthropic-format request to g1t's models asks for something
130 * charged other than by its tokens at the model's price, which the gateway
131 * cannot charge for yet, or null. On the workspace's own key the provider
132 * bills it, so anything goes there.
133 */
134export function unpriced(body: Record<string, unknown>): string | null {
135 if (body.speed != null && body.speed !== "standard") {
136 return `Fast mode is not offered on the AI Gateway on g1t's models yet. ${OWN}`;
137 }
138 if (body.inference_geo != null && body.inference_geo !== "global") {
139 return `Only global inference is offered on the AI Gateway on g1t's models yet: leave out inference_geo. ${OWN}`;
140 }
141 if (body.fallbacks != null) {
142 return `Server-side fallbacks are not offered on the AI Gateway on g1t's models yet: leave out fallbacks. ${OWN}`;
143 }
144 if (body.container != null) {
145 return `Containers and skills are not offered on the AI Gateway on g1t's models yet. ${OWN}`;
146 }
147 const tools = Array.isArray(body.tools) ? (body.tools as unknown[]) : [];
148 for (const tool of tools) {
149 const type = (tool as { type?: unknown } | null)?.type;
150 if (type == null || (typeof type === "string" && CLIENT_TOOLS.some((prefix) => type === prefix || type.startsWith(prefix)))) continue;
151 return `Server tools such as web search and code execution (${String(type)}) are not offered on the AI Gateway on g1t's models yet. ${OWN}`;
152 }
153 return null;
154}
155
156/**
157 * The same for an OpenAI-format request to g1t's models: web search and
158 * tools other than functions are billed other than by tokens.
159 */
160export function unpricedChat(body: Record<string, unknown>): string | null {
161 const own = "Use it with the workspace's own provider, under Integrations.";
162 if (body.web_search_options != null) {
163 return `Web search is not offered on the AI Gateway on g1t's models yet: leave out web_search_options. ${own}`;
164 }
165 const tools = Array.isArray(body.tools) ? (body.tools as unknown[]) : [];
166 for (const tool of tools) {
167 const type = (tool as { type?: unknown } | null)?.type;
168 if (type === "function") continue;
169 return `Only function tools are offered on the AI Gateway on g1t's models (not ${String(type)}). ${own}`;
170 }
171 return null;
172}
173
174/** A request's id: `gw_` and 24 random hex digits. */
175export function requestId(): string {
176 const bytes = crypto.getRandomValues(new Uint8Array(12));
177 return `gw_${[...bytes].map((b) => b.toString(16).padStart(2, "0")).join("")}`;
178}
179
180/**
181 * The session a request is logged under at Cloudflare's AI Gateway: one per
182 * token per UTC hour, so the gateway's own logs can be read back by token
183 * and hour.
184 */
185export function sessionOf(tokenId: string, now: Date): string {
186 const hour = now.toISOString().slice(0, 13).replace(/[-T]/g, "");
187 return `gw_${tokenId}_${hour}`;
188}
189
190/**
191 * Where a request to g1t's Claude models goes and what it carries: the
192 * caller's request, without its token, to g1t's AI Gateway with g1t's
193 * credentials and tags for the workspace, token and session.
194 */
195export function hostedRequest(
196 hosted: HostedRouting,
197 path: string,
198 incoming: Headers,
199 caller: Caller,
200 session: string,
201): { url: string; headers: Headers } {
202 const target = hostedTarget(hosted, "anthropic", incoming, caller, session);
203 return { url: `${target!.base}${path}`, headers: target!.headers };
204}
205
206/** Where a request goes, how, and who answers it. */
207export type Target = {
208 api: "anthropic" | "openai";
209 /** For Anthropic's API, without `/v1`; for OpenAI's, with it. */
210 base: string;
211 headers: Headers;
212 /** Who serves it, as the log names it: `anthropic`, `workers-ai`, or the connection's provider. */
213 provider: string;
214 dialect: Dialect;
215 ownKey: boolean;
216 /** On the workspace's own provider: the connection's name. */
217 connection: string | null;
218 /** What must never reach the caller or the log: the keys this request carries. */
219 secrets: string[];
220};
221
222/** The address of one operation at a target. */
223export function targetUrl(target: Target, op: Operation): string {
224 if (target.api === "anthropic") return `${target.base}/v1/messages${op === "count_tokens" ? "/count_tokens" : ""}`;
225 return `${target.base}/${op === "embeddings" ? "embeddings" : "chat/completions"}`;
226}
227
228/**
229 * g1t's own way to a catalogue provider's models, through its Cloudflare AI
230 * Gateway, tagged for the workspace, token and session. Null when this g1t
231 * has no way to that provider (Workers AI with no token).
232 */
233export function hostedTarget(
234 hosted: HostedRouting,
235 provider: string,
236 incoming: Headers,
237 caller: Caller,
238 session: string,
239): Target | null {
240 const headers = passedHeaders(incoming);
241 const metadata = JSON.stringify({ task: "gateway", workspace: caller.workspace, token: caller.tokenId, session });
242 const secrets = [hosted.AI_GATEWAY_TOKEN, hosted.ANTHROPIC_API_KEY, hosted.WORKERS_AI_TOKEN].filter((s): s is string => !!s);
243 const common = { ownKey: false, connection: null, secrets, provider };
244 if (provider === "workers-ai") {
245 const token = hosted.WORKERS_AI_TOKEN || hosted.AI_GATEWAY_TOKEN;
246 if (!token || !hosted.CLOUDFLARE_ACCOUNT_ID) return null;
247 headers.set("authorization", `Bearer ${token}`);
248 headers.set("content-type", "application/json");
249 if (!hosted.AI_GATEWAY_ID) {
250 return { ...common, api: "openai", dialect: { official: false, provider }, headers, base: `https://api.cloudflare.com/client/v4/accounts/${hosted.CLOUDFLARE_ACCOUNT_ID}/ai/v1` };
251 }
252 headers.set("cf-aig-metadata", metadata);
253 if (hosted.AI_GATEWAY_TOKEN) headers.set("cf-aig-authorization", `Bearer ${hosted.AI_GATEWAY_TOKEN}`);
254 return {
255 ...common,
256 api: "openai",
257 dialect: { official: false, provider },
258 headers,
259 base: `https://gateway.ai.cloudflare.com/v1/${hosted.CLOUDFLARE_ACCOUNT_ID}/${hosted.AI_GATEWAY_ID}/workers-ai/v1`,
260 };
261 }
262 if (provider !== "anthropic") return null;
263 if (!headers.has("anthropic-version")) headers.set("anthropic-version", "2023-06-01");
264 headers.set("content-type", "application/json");
265 const anthropic = { ...common, api: "anthropic" as const, dialect: { official: false, provider }, headers };
266 if (!hosted.AI_GATEWAY_ID) {
267 if (hosted.ANTHROPIC_API_KEY) headers.set("x-api-key", hosted.ANTHROPIC_API_KEY);
268 return { ...anthropic, base: "https://api.anthropic.com" };
269 }
270 headers.set("cf-aig-metadata", metadata);
271 if (hosted.AI_GATEWAY_TOKEN) headers.set("cf-aig-authorization", `Bearer ${hosted.AI_GATEWAY_TOKEN}`);
272 if (hosted.ANTHROPIC_API_KEY) headers.set("x-api-key", hosted.ANTHROPIC_API_KEY);
273 return { ...anthropic, base: `https://gateway.ai.cloudflare.com/v1/${hosted.CLOUDFLARE_ACCOUNT_ID}/${hosted.AI_GATEWAY_ID}/anthropic` };
274}
275
276/** The workspace's own provider, with its key: never g1t's gateway. */
277export function ownTarget(provider: GatewayProvider, incoming: Headers): Target {
278 const headers = passedHeaders(incoming);
279 headers.set("content-type", "application/json");
280 const key = provider.apiKey;
281 if (key) {
282 const header = provider.authHeader || (provider.api === "anthropic" ? "x-api-key" : "authorization");
283 headers.set(header, header === "authorization" ? `Bearer ${key}` : key);
284 }
285 if (provider.gatewayToken) headers.set("cf-aig-authorization", `Bearer ${provider.gatewayToken}`);
286 if (provider.api === "anthropic" && !headers.has("anthropic-version")) headers.set("anthropic-version", "2023-06-01");
287 const fallback = provider.api === "anthropic" ? "https://api.anthropic.com" : "";
288 return {
289 api: provider.api,
290 base: (provider.baseUrl || fallback).replace(/\/+$/, ""),
291 headers,
292 provider: provider.provider,
293 dialect: { official: provider.official, provider: provider.provider },
294 ownKey: true,
295 connection: provider.name,
296 secrets: [provider.apiKey, provider.gatewayToken].filter((s): s is string => !!s),
297 };
298}
299
300/**
301 * A text with every secret it might carry taken out, as a provider's error
302 * can quote the key it refused.
303 */
304export function scrub(text: string, secrets: string[]): string {
305 let out = text;
306 for (const secret of secrets) {
307 if (secret.length < 8) continue;
308 out = out.split(secret).join("[redacted]");
309 // A key quoted with its end cut off is still most of the key.
310 const head = secret.slice(0, Math.max(8, Math.floor(secret.length * 0.75)));
311 out = out.split(head).join("[redacted]");
312 }
313 return out;
314}
315
316/** The message of a provider's error body, either format's shape, or the status. */
317export function errorMessage(status: number, body: string): string {
318 try {
319 const parsed = JSON.parse(body) as { error?: { message?: unknown } | string } | { error?: { message?: unknown } }[];
320 const error = Array.isArray(parsed) ? parsed[0]?.error : parsed.error;
321 if (typeof error === "string") return error.slice(0, 500);
322 if (typeof error?.message === "string") return error.message.slice(0, 500);
323 } catch {
324 // Not JSON: the status says enough.
325 }
326 return `The model provider answered ${status}.`;
327}
328
329/** What billing is told about one request. */
330export function gatewayRecord(input: {
331 id: string;
332 caller: Caller;
333 model: string;
334 tokens: Tokens;
335 status: number;
336 ownKey: boolean;
337 streamed: boolean;
338 durationMs: number;
339 error?: string | null;
340 format?: Format;
341 provider?: string;
342 connection?: string | null;
343}): GatewayRecord {
344 return {
345 id: input.id,
346 workspace: input.caller.workspace,
347 tokenId: input.caller.tokenId,
348 tokenName: input.caller.tokenName,
349 model: input.model.slice(0, 200) || "unknown",
350 input: input.tokens.input,
351 output: input.tokens.output,
352 cacheRead: input.tokens.cacheRead,
353 cacheWrite: input.tokens.cacheWrite,
354 cacheWriteHour: input.tokens.cacheWrite1h ?? 0,
355 status: input.status,
356 ownKey: input.ownKey,
357 format: input.format ?? "anthropic",
358 provider: input.provider ?? "",
359 connection: input.connection ?? null,
360 streamed: input.streamed,
361 durationMs: Math.max(0, Math.round(input.durationMs)),
362 error: input.error ?? null,
363 };
364}