| 1 | /** |
| 2 | * The AI Gateway: a workspace's own model requests, in Anthropic's Messages |
| 3 | * format, at the same address as its sandboxes' (`models.g1t.sh/anthropic`). |
| 4 | * |
| 5 | * A request carries one of the workspace's access tokens (`g1t_…`) with |
| 6 | * the `models:write` scope, as `x-api-key` or `Authorization: Bearer`, so |
| 7 | * an Anthropic SDK or Claude Code needs only a base URL and a key. When the |
| 8 | * workspace has its own Anthropic key under Integrations, requests go there |
| 9 | * and cost nothing; otherwise they go to g1t's models through its AI |
| 10 | * Gateway, are admitted by billing first (spend limit, AI credit), and are |
| 11 | * charged at the model's price afterwards. Every request is logged, with |
| 12 | * its tokens, never its prompt or answer. |
| 13 | * |
| 14 | * What is here is the part that decides; `index.ts` sends. |
| 15 | */ |
| 16 | import type { GatewayModel, GatewayRecord, User } from "@g1t/contracts"; |
| 17 | |
| 18 | import type { HostedRouting } from "./route.ts"; |
| 19 | import { passedHeaders } from "./route.ts"; |
| 20 | import type { Tokens } from "./usage.ts"; |
| 21 | |
| 22 | /** Anthropic's error types, by the statuses the gateway answers with. */ |
| 23 | export type ErrorType = |
| 24 | | "invalid_request_error" |
| 25 | | "authentication_error" |
| 26 | | "billing_error" |
| 27 | | "permission_error" |
| 28 | | "not_found_error" |
| 29 | | "api_error"; |
| 30 | |
| 31 | /** An error in the shape Anthropic's API and SDKs use. */ |
| 32 | export function anthropicError(status: number, type: ErrorType, message: string): Response { |
| 33 | return Response.json({ type: "error", error: { type, message } }, { status }); |
| 34 | } |
| 35 | |
| 36 | /** Who a gateway request is for, from its token. */ |
| 37 | export type Caller = { workspace: string; tokenId: string; tokenName: string | null }; |
| 38 | |
| 39 | /** The scope a token needs to send requests. */ |
| 40 | export const GATEWAY_SCOPE = "models:write"; |
| 41 | |
| 42 | /** |
| 43 | * Who a token stands for, or why it cannot use the gateway: it is unknown |
| 44 | * or expired, it is not a workspace's own token, or it lacks |
| 45 | * `models:write`. A token with full access has every scope. |
| 46 | */ |
| 47 | export function callerOf(viewer: User | null): { caller: Caller } | { status: number; type: ErrorType; message: string } { |
| 48 | if (!viewer) { |
| 49 | return { status: 401, type: "authentication_error", message: "This access token is not valid, or it has expired or been deleted." }; |
| 50 | } |
| 51 | if (viewer.kind !== "workspace" || !viewer.token) { |
| 52 | return { |
| 53 | status: 403, |
| 54 | type: "permission_error", |
| 55 | message: "The AI Gateway takes a workspace's access token, which its usage is charged to. An owner can make one under the workspace's Settings, Access tokens, with the models:write scope.", |
| 56 | }; |
| 57 | } |
| 58 | const scopes = viewer.token.scopes; |
| 59 | if (scopes && !scopes.includes(GATEWAY_SCOPE)) { |
| 60 | return { status: 403, type: "permission_error", message: `This access token needs the ${GATEWAY_SCOPE} scope to use the AI Gateway.` }; |
| 61 | } |
| 62 | return { caller: { workspace: viewer.username.toLowerCase(), tokenId: viewer.token.token_id, tokenName: viewer.token.name ?? null } }; |
| 63 | } |
| 64 | |
| 65 | /** Which of the gateway's routes a path (after `/anthropic`) is, or null. */ |
| 66 | export function gatewayRoute(path: string): "messages" | "count_tokens" | null { |
| 67 | const bare = path.split("?")[0]!.replace(/\/+$/, ""); |
| 68 | if (bare === "/v1/messages") return "messages"; |
| 69 | if (bare === "/v1/messages/count_tokens") return "count_tokens"; |
| 70 | return null; |
| 71 | } |
| 72 | |
| 73 | /** Why a model is not offered on g1t's key, or null when it is. */ |
| 74 | export function unoffered(model: unknown, offered: GatewayModel[]): string | null { |
| 75 | if (typeof model !== "string" || !model.trim()) return "Name a model: `model` is required."; |
| 76 | if (offered.some((row) => row.model === model)) return null; |
| 77 | const names = offered.map((row) => row.model).filter((id, i, all) => all.indexOf(id) === i); |
| 78 | return `${model} is not offered on the AI Gateway. It offers ${names.join(", ")}. See https://docs.g1t.sh/guides/ai-gateway/#models`; |
| 79 | } |
| 80 | |
| 81 | /** |
| 82 | * Tool types that run on the caller's side, so cost only their tokens. A |
| 83 | * tool with no type is the caller's own. |
| 84 | */ |
| 85 | const CLIENT_TOOLS = ["custom", "bash_", "text_editor_", "computer_", "memory_"]; |
| 86 | |
| 87 | /** |
| 88 | * Why a request to g1t's models asks for something charged other than by |
| 89 | * its tokens at the model's price, which the gateway cannot charge for |
| 90 | * yet, or null. On the workspace's own key the provider bills it, so |
| 91 | * anything goes there. |
| 92 | */ |
| 93 | export function unpriced(body: Record<string, unknown>): string | null { |
| 94 | const own = "Use it with the workspace's own Anthropic key, under Integrations."; |
| 95 | if (body.speed != null && body.speed !== "standard") { |
| 96 | return `Fast mode is not offered on the AI Gateway on g1t's models yet. ${own}`; |
| 97 | } |
| 98 | if (body.inference_geo != null && body.inference_geo !== "global") { |
| 99 | return `Only global inference is offered on the AI Gateway on g1t's models yet: leave out inference_geo. ${own}`; |
| 100 | } |
| 101 | if (body.fallbacks != null) { |
| 102 | return `Server-side fallbacks are not offered on the AI Gateway on g1t's models yet: leave out fallbacks. ${own}`; |
| 103 | } |
| 104 | if (body.container != null) { |
| 105 | return `Containers and skills are not offered on the AI Gateway on g1t's models yet. ${own}`; |
| 106 | } |
| 107 | const tools = Array.isArray(body.tools) ? (body.tools as unknown[]) : []; |
| 108 | for (const tool of tools) { |
| 109 | const type = (tool as { type?: unknown } | null)?.type; |
| 110 | if (type == null || (typeof type === "string" && CLIENT_TOOLS.some((prefix) => type === prefix || type.startsWith(prefix)))) continue; |
| 111 | return `Server tools such as web search and code execution (${String(type)}) are not offered on the AI Gateway on g1t's models yet. ${own}`; |
| 112 | } |
| 113 | return null; |
| 114 | } |
| 115 | |
| 116 | /** A request's id: `gw_` and 24 random hex digits. */ |
| 117 | export function requestId(): string { |
| 118 | const bytes = crypto.getRandomValues(new Uint8Array(12)); |
| 119 | return `gw_${[...bytes].map((b) => b.toString(16).padStart(2, "0")).join("")}`; |
| 120 | } |
| 121 | |
| 122 | /** |
| 123 | * The session a request is logged under at Cloudflare's AI Gateway: one per |
| 124 | * token per UTC hour, so the gateway's own logs can be read back by token |
| 125 | * and hour. |
| 126 | */ |
| 127 | export function sessionOf(tokenId: string, now: Date): string { |
| 128 | const hour = now.toISOString().slice(0, 13).replace(/[-T]/g, ""); |
| 129 | return `gw_${tokenId}_${hour}`; |
| 130 | } |
| 131 | |
| 132 | /** |
| 133 | * Where a request to g1t's models goes and what it carries: the caller's |
| 134 | * request, without its token, to g1t's AI Gateway with g1t's credentials |
| 135 | * and tags for the workspace, token and session. |
| 136 | */ |
| 137 | export function hostedRequest( |
| 138 | hosted: HostedRouting, |
| 139 | path: string, |
| 140 | incoming: Headers, |
| 141 | caller: Caller, |
| 142 | session: string, |
| 143 | ): { url: string; headers: Headers } { |
| 144 | const headers = passedHeaders(incoming); |
| 145 | if (!hosted.AI_GATEWAY_ID) { |
| 146 | if (hosted.ANTHROPIC_API_KEY) headers.set("x-api-key", hosted.ANTHROPIC_API_KEY); |
| 147 | return { url: `https://api.anthropic.com${path}`, headers }; |
| 148 | } |
| 149 | headers.set("cf-aig-metadata", JSON.stringify({ task: "gateway", workspace: caller.workspace, token: caller.tokenId, session })); |
| 150 | if (hosted.AI_GATEWAY_TOKEN) headers.set("cf-aig-authorization", `Bearer ${hosted.AI_GATEWAY_TOKEN}`); |
| 151 | if (hosted.ANTHROPIC_API_KEY) headers.set("x-api-key", hosted.ANTHROPIC_API_KEY); |
| 152 | return { |
| 153 | url: `https://gateway.ai.cloudflare.com/v1/${hosted.CLOUDFLARE_ACCOUNT_ID}/${hosted.AI_GATEWAY_ID}/anthropic${path}`, |
| 154 | headers, |
| 155 | }; |
| 156 | } |
| 157 | |
| 158 | /** The message of an Anthropic-shaped error body, or the status. */ |
| 159 | export function errorMessage(status: number, body: string): string { |
| 160 | try { |
| 161 | const parsed = JSON.parse(body) as { error?: { message?: unknown } }; |
| 162 | if (typeof parsed.error?.message === "string") return parsed.error.message.slice(0, 500); |
| 163 | } catch { |
| 164 | // Not JSON: the status says enough. |
| 165 | } |
| 166 | return `The model provider answered ${status}.`; |
| 167 | } |
| 168 | |
| 169 | /** What billing is told about one request. */ |
| 170 | export function gatewayRecord(input: { |
| 171 | id: string; |
| 172 | caller: Caller; |
| 173 | model: string; |
| 174 | tokens: Tokens; |
| 175 | status: number; |
| 176 | ownKey: boolean; |
| 177 | streamed: boolean; |
| 178 | durationMs: number; |
| 179 | error?: string | null; |
| 180 | }): GatewayRecord { |
| 181 | return { |
| 182 | id: input.id, |
| 183 | workspace: input.caller.workspace, |
| 184 | tokenId: input.caller.tokenId, |
| 185 | tokenName: input.caller.tokenName, |
| 186 | model: input.model.slice(0, 200) || "unknown", |
| 187 | input: input.tokens.input, |
| 188 | output: input.tokens.output, |
| 189 | cacheRead: input.tokens.cacheRead, |
| 190 | cacheWrite: input.tokens.cacheWrite, |
| 191 | status: input.status, |
| 192 | ownKey: input.ownKey, |
| 193 | streamed: input.streamed, |
| 194 | durationMs: Math.max(0, Math.round(input.durationMs)), |
| 195 | error: input.error ?? null, |
| 196 | }; |
| 197 | } |