| 1 | /** |
| 2 | * The AI Gateway: a workspace's own model requests, at the same address as |
| 3 | * its sandboxes' (`models.g1t.sh`), in either format: |
| 4 | * |
| 5 | * - `/anthropic/v1/messages` (and `/count_tokens`): Anthropic's Messages API. |
| 6 | * - `/openai/v1/chat/completions`, `/openai/v1/embeddings` and |
| 7 | * `/openai/v1/models`: OpenAI's. |
| 8 | * |
| 9 | * A request carries one of the workspace's access tokens (`g1t_…`) with |
| 10 | * the `models:write` scope, as `x-api-key` or `Authorization: Bearer`, so |
| 11 | * either format's SDKs and tools need only a base URL and a key. The model |
| 12 | * it names decides where it goes (`catalogue.ts`): one of the workspace's |
| 13 | * own providers under Integrations, where it costs nothing, or g1t's |
| 14 | * catalogue (Claude on Anthropic, open models on Workers AI), admitted by |
| 15 | * billing first (spend limit, AI credit) and charged at the model's price |
| 16 | * afterwards. Either format reaches either kind of provider: the proxy |
| 17 | * translates (`openai.ts`, `chat.ts`). Every request is logged, with its |
| 18 | * tokens, never its prompt or answer. |
| 19 | * |
| 20 | * What is here is the part that decides; `serve.ts` sends. |
| 21 | */ |
| 22 | import type { GatewayProvider, GatewayRecord, User } from "@g1t/contracts"; |
| 23 | |
| 24 | import type { Dialect } from "./openai.ts"; |
| 25 | import type { HostedRouting } from "./route.ts"; |
| 26 | import { passedHeaders } from "./route.ts"; |
| 27 | import type { Tokens } from "./usage.ts"; |
| 28 | |
| 29 | /** Anthropic's error types, by the statuses the gateway answers with. */ |
| 30 | export type ErrorType = |
| 31 | | "invalid_request_error" |
| 32 | | "authentication_error" |
| 33 | | "billing_error" |
| 34 | | "permission_error" |
| 35 | | "not_found_error" |
| 36 | | "rate_limit_error" |
| 37 | | "api_error"; |
| 38 | |
| 39 | /** An error in the shape Anthropic's API and SDKs use. */ |
| 40 | export function anthropicError(status: number, type: ErrorType, message: string): Response { |
| 41 | return Response.json({ type: "error", error: { type, message } }, { status }); |
| 42 | } |
| 43 | |
| 44 | /** Anthropic's error type for a status. */ |
| 45 | export function anthropicErrorType(status: number): ErrorType { |
| 46 | if (status === 401) return "authentication_error"; |
| 47 | if (status === 402) return "billing_error"; |
| 48 | if (status === 403) return "permission_error"; |
| 49 | if (status === 404) return "not_found_error"; |
| 50 | if (status === 429) return "rate_limit_error"; |
| 51 | if (status >= 500) return "api_error"; |
| 52 | return "invalid_request_error"; |
| 53 | } |
| 54 | |
| 55 | /** Who a gateway request is for, from its token. */ |
| 56 | export type Caller = { workspace: string; tokenId: string; tokenName: string | null }; |
| 57 | |
| 58 | /** The scope a token needs to send requests. */ |
| 59 | export const GATEWAY_SCOPE = "models:write"; |
| 60 | |
| 61 | /** |
| 62 | * Who a token stands for, or why it cannot use the gateway: it is unknown |
| 63 | * or expired, it is not a workspace's own token, or it lacks |
| 64 | * `models:write`. A token with full access has every scope. |
| 65 | */ |
| 66 | export function callerOf(viewer: User | null): { caller: Caller } | { status: number; type: ErrorType; message: string } { |
| 67 | if (!viewer) { |
| 68 | return { status: 401, type: "authentication_error", message: "This access token is not valid, or it has expired or been deleted." }; |
| 69 | } |
| 70 | if (viewer.kind !== "workspace" || !viewer.token) { |
| 71 | return { |
| 72 | status: 403, |
| 73 | type: "permission_error", |
| 74 | message: "The AI Gateway takes a workspace's access token, which its usage is charged to. An owner can make one under the workspace's Settings, Access tokens, with the models:write scope.", |
| 75 | }; |
| 76 | } |
| 77 | const scopes = viewer.token.scopes; |
| 78 | if (scopes && !scopes.includes(GATEWAY_SCOPE)) { |
| 79 | return { status: 403, type: "permission_error", message: `This access token needs the ${GATEWAY_SCOPE} scope to use the AI Gateway.` }; |
| 80 | } |
| 81 | return { caller: { workspace: viewer.username.toLowerCase(), tokenId: viewer.token.token_id, tokenName: viewer.token.name ?? null } }; |
| 82 | } |
| 83 | |
| 84 | /** The request formats, by their path. */ |
| 85 | export type Format = "anthropic" | "openai"; |
| 86 | |
| 87 | /** What a request asks. */ |
| 88 | export type Operation = "messages" | "count_tokens" | "chat" | "embeddings" | "models"; |
| 89 | |
| 90 | /** Which of the gateway's routes a path (after `/anthropic`) is, or null. */ |
| 91 | export function gatewayRoute(path: string): "messages" | "count_tokens" | null { |
| 92 | const bare = path.split("?")[0]!.replace(/\/+$/, ""); |
| 93 | if (bare === "/v1/messages") return "messages"; |
| 94 | if (bare === "/v1/messages/count_tokens") return "count_tokens"; |
| 95 | return null; |
| 96 | } |
| 97 | |
| 98 | /** |
| 99 | * The format and operation of a request by its whole path and method, or |
| 100 | * null for one the gateway does not answer. |
| 101 | */ |
| 102 | export function gatewayOperation(path: string, method: string): { format: Format; op: Operation } | null { |
| 103 | const bare = path.split("?")[0]!.replace(/\/+$/, ""); |
| 104 | if (bare.startsWith("/anthropic/")) { |
| 105 | const op = method === "POST" ? gatewayRoute(bare.slice("/anthropic".length)) : null; |
| 106 | return op ? { format: "anthropic", op } : null; |
| 107 | } |
| 108 | if (bare === "/openai/v1/chat/completions" && method === "POST") return { format: "openai", op: "chat" }; |
| 109 | if (bare === "/openai/v1/embeddings" && method === "POST") return { format: "openai", op: "embeddings" }; |
| 110 | if (bare === "/openai/v1/models" && method === "GET") return { format: "openai", op: "models" }; |
| 111 | return null; |
| 112 | } |
| 113 | |
| 114 | /** What the gateway answers, per format, for a request to a route it does not have. */ |
| 115 | export const ROUTES: Record<Format, string> = { |
| 116 | anthropic: "The AI Gateway answers POST /anthropic/v1/messages and POST /anthropic/v1/messages/count_tokens in Anthropic's format.", |
| 117 | openai: "The AI Gateway answers POST /openai/v1/chat/completions, POST /openai/v1/embeddings and GET /openai/v1/models in OpenAI's format.", |
| 118 | }; |
| 119 | |
| 120 | /** |
| 121 | * Tool types that run on the caller's side, so cost only their tokens. A |
| 122 | * tool with no type is the caller's own. |
| 123 | */ |
| 124 | const CLIENT_TOOLS = ["custom", "bash_", "text_editor_", "computer_", "memory_"]; |
| 125 | |
| 126 | const OWN = "Use it with the workspace's own Anthropic key, under Integrations."; |
| 127 | |
| 128 | /** |
| 129 | * Why an Anthropic-format request to g1t's models asks for something |
| 130 | * charged other than by its tokens at the model's price, which the gateway |
| 131 | * cannot charge for yet, or null. On the workspace's own key the provider |
| 132 | * bills it, so anything goes there. |
| 133 | */ |
| 134 | export function unpriced(body: Record<string, unknown>): string | null { |
| 135 | if (body.speed != null && body.speed !== "standard") { |
| 136 | return `Fast mode is not offered on the AI Gateway on g1t's models yet. ${OWN}`; |
| 137 | } |
| 138 | if (body.inference_geo != null && body.inference_geo !== "global") { |
| 139 | return `Only global inference is offered on the AI Gateway on g1t's models yet: leave out inference_geo. ${OWN}`; |
| 140 | } |
| 141 | if (body.fallbacks != null) { |
| 142 | return `Server-side fallbacks are not offered on the AI Gateway on g1t's models yet: leave out fallbacks. ${OWN}`; |
| 143 | } |
| 144 | if (body.container != null) { |
| 145 | return `Containers and skills are not offered on the AI Gateway on g1t's models yet. ${OWN}`; |
| 146 | } |
| 147 | const tools = Array.isArray(body.tools) ? (body.tools as unknown[]) : []; |
| 148 | for (const tool of tools) { |
| 149 | const type = (tool as { type?: unknown } | null)?.type; |
| 150 | if (type == null || (typeof type === "string" && CLIENT_TOOLS.some((prefix) => type === prefix || type.startsWith(prefix)))) continue; |
| 151 | return `Server tools such as web search and code execution (${String(type)}) are not offered on the AI Gateway on g1t's models yet. ${OWN}`; |
| 152 | } |
| 153 | return null; |
| 154 | } |
| 155 | |
| 156 | /** |
| 157 | * The same for an OpenAI-format request to g1t's models: web search and |
| 158 | * tools other than functions are billed other than by tokens. |
| 159 | */ |
| 160 | export function unpricedChat(body: Record<string, unknown>): string | null { |
| 161 | const own = "Use it with the workspace's own provider, under Integrations."; |
| 162 | if (body.web_search_options != null) { |
| 163 | return `Web search is not offered on the AI Gateway on g1t's models yet: leave out web_search_options. ${own}`; |
| 164 | } |
| 165 | const tools = Array.isArray(body.tools) ? (body.tools as unknown[]) : []; |
| 166 | for (const tool of tools) { |
| 167 | const type = (tool as { type?: unknown } | null)?.type; |
| 168 | if (type === "function") continue; |
| 169 | return `Only function tools are offered on the AI Gateway on g1t's models (not ${String(type)}). ${own}`; |
| 170 | } |
| 171 | return null; |
| 172 | } |
| 173 | |
| 174 | /** A request's id: `gw_` and 24 random hex digits. */ |
| 175 | export function requestId(): string { |
| 176 | const bytes = crypto.getRandomValues(new Uint8Array(12)); |
| 177 | return `gw_${[...bytes].map((b) => b.toString(16).padStart(2, "0")).join("")}`; |
| 178 | } |
| 179 | |
| 180 | /** |
| 181 | * The session a request is logged under at Cloudflare's AI Gateway: one per |
| 182 | * token per UTC hour, so the gateway's own logs can be read back by token |
| 183 | * and hour. |
| 184 | */ |
| 185 | export function sessionOf(tokenId: string, now: Date): string { |
| 186 | const hour = now.toISOString().slice(0, 13).replace(/[-T]/g, ""); |
| 187 | return `gw_${tokenId}_${hour}`; |
| 188 | } |
| 189 | |
| 190 | /** |
| 191 | * Where a request to g1t's Claude models goes and what it carries: the |
| 192 | * caller's request, without its token, to g1t's AI Gateway with g1t's |
| 193 | * credentials and tags for the workspace, token and session. |
| 194 | */ |
| 195 | export function hostedRequest( |
| 196 | hosted: HostedRouting, |
| 197 | path: string, |
| 198 | incoming: Headers, |
| 199 | caller: Caller, |
| 200 | session: string, |
| 201 | ): { url: string; headers: Headers } { |
| 202 | const target = hostedTarget(hosted, "anthropic", incoming, caller, session); |
| 203 | return { url: `${target!.base}${path}`, headers: target!.headers }; |
| 204 | } |
| 205 | |
| 206 | /** Where a request goes, how, and who answers it. */ |
| 207 | export type Target = { |
| 208 | api: "anthropic" | "openai"; |
| 209 | /** For Anthropic's API, without `/v1`; for OpenAI's, with it. */ |
| 210 | base: string; |
| 211 | headers: Headers; |
| 212 | /** Who serves it, as the log names it: `anthropic`, `workers-ai`, or the connection's provider. */ |
| 213 | provider: string; |
| 214 | dialect: Dialect; |
| 215 | ownKey: boolean; |
| 216 | /** On the workspace's own provider: the connection's name. */ |
| 217 | connection: string | null; |
| 218 | /** What must never reach the caller or the log: the keys this request carries. */ |
| 219 | secrets: string[]; |
| 220 | }; |
| 221 | |
| 222 | /** The address of one operation at a target. */ |
| 223 | export function targetUrl(target: Target, op: Operation): string { |
| 224 | if (target.api === "anthropic") return `${target.base}/v1/messages${op === "count_tokens" ? "/count_tokens" : ""}`; |
| 225 | return `${target.base}/${op === "embeddings" ? "embeddings" : "chat/completions"}`; |
| 226 | } |
| 227 | |
| 228 | /** |
| 229 | * g1t's own way to a catalogue provider's models, through its Cloudflare AI |
| 230 | * Gateway, tagged for the workspace, token and session. Null when this g1t |
| 231 | * has no way to that provider (Workers AI with no token). |
| 232 | */ |
| 233 | export function hostedTarget( |
| 234 | hosted: HostedRouting, |
| 235 | provider: string, |
| 236 | incoming: Headers, |
| 237 | caller: Caller, |
| 238 | session: string, |
| 239 | ): Target | null { |
| 240 | const headers = passedHeaders(incoming); |
| 241 | const metadata = JSON.stringify({ task: "gateway", workspace: caller.workspace, token: caller.tokenId, session }); |
| 242 | const secrets = [hosted.AI_GATEWAY_TOKEN, hosted.ANTHROPIC_API_KEY, hosted.WORKERS_AI_TOKEN].filter((s): s is string => !!s); |
| 243 | const common = { ownKey: false, connection: null, secrets, provider }; |
| 244 | if (provider === "workers-ai") { |
| 245 | const token = hosted.WORKERS_AI_TOKEN || hosted.AI_GATEWAY_TOKEN; |
| 246 | if (!token || !hosted.CLOUDFLARE_ACCOUNT_ID) return null; |
| 247 | headers.set("authorization", `Bearer ${token}`); |
| 248 | headers.set("content-type", "application/json"); |
| 249 | if (!hosted.AI_GATEWAY_ID) { |
| 250 | return { ...common, api: "openai", dialect: { official: false, provider }, headers, base: `https://api.cloudflare.com/client/v4/accounts/${hosted.CLOUDFLARE_ACCOUNT_ID}/ai/v1` }; |
| 251 | } |
| 252 | headers.set("cf-aig-metadata", metadata); |
| 253 | if (hosted.AI_GATEWAY_TOKEN) headers.set("cf-aig-authorization", `Bearer ${hosted.AI_GATEWAY_TOKEN}`); |
| 254 | return { |
| 255 | ...common, |
| 256 | api: "openai", |
| 257 | dialect: { official: false, provider }, |
| 258 | headers, |
| 259 | base: `https://gateway.ai.cloudflare.com/v1/${hosted.CLOUDFLARE_ACCOUNT_ID}/${hosted.AI_GATEWAY_ID}/workers-ai/v1`, |
| 260 | }; |
| 261 | } |
| 262 | if (provider !== "anthropic") return null; |
| 263 | if (!headers.has("anthropic-version")) headers.set("anthropic-version", "2023-06-01"); |
| 264 | headers.set("content-type", "application/json"); |
| 265 | const anthropic = { ...common, api: "anthropic" as const, dialect: { official: false, provider }, headers }; |
| 266 | if (!hosted.AI_GATEWAY_ID) { |
| 267 | if (hosted.ANTHROPIC_API_KEY) headers.set("x-api-key", hosted.ANTHROPIC_API_KEY); |
| 268 | return { ...anthropic, base: "https://api.anthropic.com" }; |
| 269 | } |
| 270 | headers.set("cf-aig-metadata", metadata); |
| 271 | if (hosted.AI_GATEWAY_TOKEN) headers.set("cf-aig-authorization", `Bearer ${hosted.AI_GATEWAY_TOKEN}`); |
| 272 | if (hosted.ANTHROPIC_API_KEY) headers.set("x-api-key", hosted.ANTHROPIC_API_KEY); |
| 273 | return { ...anthropic, base: `https://gateway.ai.cloudflare.com/v1/${hosted.CLOUDFLARE_ACCOUNT_ID}/${hosted.AI_GATEWAY_ID}/anthropic` }; |
| 274 | } |
| 275 | |
| 276 | /** The workspace's own provider, with its key: never g1t's gateway. */ |
| 277 | export function ownTarget(provider: GatewayProvider, incoming: Headers): Target { |
| 278 | const headers = passedHeaders(incoming); |
| 279 | headers.set("content-type", "application/json"); |
| 280 | const key = provider.apiKey; |
| 281 | if (key) { |
| 282 | const header = provider.authHeader || (provider.api === "anthropic" ? "x-api-key" : "authorization"); |
| 283 | headers.set(header, header === "authorization" ? `Bearer ${key}` : key); |
| 284 | } |
| 285 | if (provider.gatewayToken) headers.set("cf-aig-authorization", `Bearer ${provider.gatewayToken}`); |
| 286 | if (provider.api === "anthropic" && !headers.has("anthropic-version")) headers.set("anthropic-version", "2023-06-01"); |
| 287 | const fallback = provider.api === "anthropic" ? "https://api.anthropic.com" : ""; |
| 288 | return { |
| 289 | api: provider.api, |
| 290 | base: (provider.baseUrl || fallback).replace(/\/+$/, ""), |
| 291 | headers, |
| 292 | provider: provider.provider, |
| 293 | dialect: { official: provider.official, provider: provider.provider }, |
| 294 | ownKey: true, |
| 295 | connection: provider.name, |
| 296 | secrets: [provider.apiKey, provider.gatewayToken].filter((s): s is string => !!s), |
| 297 | }; |
| 298 | } |
| 299 | |
| 300 | /** |
| 301 | * A text with every secret it might carry taken out, as a provider's error |
| 302 | * can quote the key it refused. |
| 303 | */ |
| 304 | export function scrub(text: string, secrets: string[]): string { |
| 305 | let out = text; |
| 306 | for (const secret of secrets) { |
| 307 | if (secret.length < 8) continue; |
| 308 | out = out.split(secret).join("[redacted]"); |
| 309 | // A key quoted with its end cut off is still most of the key. |
| 310 | const head = secret.slice(0, Math.max(8, Math.floor(secret.length * 0.75))); |
| 311 | out = out.split(head).join("[redacted]"); |
| 312 | } |
| 313 | return out; |
| 314 | } |
| 315 | |
| 316 | /** The message of a provider's error body, either format's shape, or the status. */ |
| 317 | export function errorMessage(status: number, body: string): string { |
| 318 | try { |
| 319 | const parsed = JSON.parse(body) as { error?: { message?: unknown } | string } | { error?: { message?: unknown } }[]; |
| 320 | const error = Array.isArray(parsed) ? parsed[0]?.error : parsed.error; |
| 321 | if (typeof error === "string") return error.slice(0, 500); |
| 322 | if (typeof error?.message === "string") return error.message.slice(0, 500); |
| 323 | } catch { |
| 324 | // Not JSON: the status says enough. |
| 325 | } |
| 326 | return `The model provider answered ${status}.`; |
| 327 | } |
| 328 | |
| 329 | /** What billing is told about one request. */ |
| 330 | export function gatewayRecord(input: { |
| 331 | id: string; |
| 332 | caller: Caller; |
| 333 | model: string; |
| 334 | tokens: Tokens; |
| 335 | status: number; |
| 336 | ownKey: boolean; |
| 337 | streamed: boolean; |
| 338 | durationMs: number; |
| 339 | error?: string | null; |
| 340 | format?: Format; |
| 341 | provider?: string; |
| 342 | connection?: string | null; |
| 343 | }): GatewayRecord { |
| 344 | return { |
| 345 | id: input.id, |
| 346 | workspace: input.caller.workspace, |
| 347 | tokenId: input.caller.tokenId, |
| 348 | tokenName: input.caller.tokenName, |
| 349 | model: input.model.slice(0, 200) || "unknown", |
| 350 | input: input.tokens.input, |
| 351 | output: input.tokens.output, |
| 352 | cacheRead: input.tokens.cacheRead, |
| 353 | cacheWrite: input.tokens.cacheWrite, |
| 354 | cacheWriteHour: input.tokens.cacheWrite1h ?? 0, |
| 355 | status: input.status, |
| 356 | ownKey: input.ownKey, |
| 357 | format: input.format ?? "anthropic", |
| 358 | provider: input.provider ?? "", |
| 359 | connection: input.connection ?? null, |
| 360 | streamed: input.streamed, |
| 361 | durationMs: Math.max(0, Math.round(input.durationMs)), |
| 362 | error: input.error ?? null, |
| 363 | }; |
| 364 | } |