Skip to content

g1t/services/models/src/gateway.ts

197 lines8,384 bytesCodeBlame
1/**
2 * The AI Gateway: a workspace's own model requests, in Anthropic's Messages
3 * format, at the same address as its sandboxes' (`models.g1t.sh/anthropic`).
4 *
5 * A request carries one of the workspace's access tokens (`g1t_…`) with
6 * the `models:write` scope, as `x-api-key` or `Authorization: Bearer`, so
7 * an Anthropic SDK or Claude Code needs only a base URL and a key. When the
8 * workspace has its own Anthropic key under Integrations, requests go there
9 * and cost nothing; otherwise they go to g1t's models through its AI
10 * Gateway, are admitted by billing first (spend limit, AI credit), and are
11 * charged at the model's price afterwards. Every request is logged, with
12 * its tokens, never its prompt or answer.
13 *
14 * What is here is the part that decides; `index.ts` sends.
15 */
16import type { GatewayModel, GatewayRecord, User } from "@g1t/contracts";
17
18import type { HostedRouting } from "./route.ts";
19import { passedHeaders } from "./route.ts";
20import type { Tokens } from "./usage.ts";
21
22/** Anthropic's error types, by the statuses the gateway answers with. */
23export type ErrorType =
24 | "invalid_request_error"
25 | "authentication_error"
26 | "billing_error"
27 | "permission_error"
28 | "not_found_error"
29 | "api_error";
30
31/** An error in the shape Anthropic's API and SDKs use. */
32export function anthropicError(status: number, type: ErrorType, message: string): Response {
33 return Response.json({ type: "error", error: { type, message } }, { status });
34}
35
36/** Who a gateway request is for, from its token. */
37export type Caller = { workspace: string; tokenId: string; tokenName: string | null };
38
39/** The scope a token needs to send requests. */
40export const GATEWAY_SCOPE = "models:write";
41
42/**
43 * Who a token stands for, or why it cannot use the gateway: it is unknown
44 * or expired, it is not a workspace's own token, or it lacks
45 * `models:write`. A token with full access has every scope.
46 */
47export function callerOf(viewer: User | null): { caller: Caller } | { status: number; type: ErrorType; message: string } {
48 if (!viewer) {
49 return { status: 401, type: "authentication_error", message: "This access token is not valid, or it has expired or been deleted." };
50 }
51 if (viewer.kind !== "workspace" || !viewer.token) {
52 return {
53 status: 403,
54 type: "permission_error",
55 message: "The AI Gateway takes a workspace's access token, which its usage is charged to. An owner can make one under the workspace's Settings, Access tokens, with the models:write scope.",
56 };
57 }
58 const scopes = viewer.token.scopes;
59 if (scopes && !scopes.includes(GATEWAY_SCOPE)) {
60 return { status: 403, type: "permission_error", message: `This access token needs the ${GATEWAY_SCOPE} scope to use the AI Gateway.` };
61 }
62 return { caller: { workspace: viewer.username.toLowerCase(), tokenId: viewer.token.token_id, tokenName: viewer.token.name ?? null } };
63}
64
65/** Which of the gateway's routes a path (after `/anthropic`) is, or null. */
66export function gatewayRoute(path: string): "messages" | "count_tokens" | null {
67 const bare = path.split("?")[0]!.replace(/\/+$/, "");
68 if (bare === "/v1/messages") return "messages";
69 if (bare === "/v1/messages/count_tokens") return "count_tokens";
70 return null;
71}
72
73/** Why a model is not offered on g1t's key, or null when it is. */
74export function unoffered(model: unknown, offered: GatewayModel[]): string | null {
75 if (typeof model !== "string" || !model.trim()) return "Name a model: `model` is required.";
76 if (offered.some((row) => row.model === model)) return null;
77 const names = offered.map((row) => row.model).filter((id, i, all) => all.indexOf(id) === i);
78 return `${model} is not offered on the AI Gateway. It offers ${names.join(", ")}. See https://docs.g1t.sh/guides/ai-gateway/#models`;
79}
80
81/**
82 * Tool types that run on the caller's side, so cost only their tokens. A
83 * tool with no type is the caller's own.
84 */
85const CLIENT_TOOLS = ["custom", "bash_", "text_editor_", "computer_", "memory_"];
86
87/**
88 * Why a request to g1t's models asks for something charged other than by
89 * its tokens at the model's price, which the gateway cannot charge for
90 * yet, or null. On the workspace's own key the provider bills it, so
91 * anything goes there.
92 */
93export function unpriced(body: Record<string, unknown>): string | null {
94 const own = "Use it with the workspace's own Anthropic key, under Integrations.";
95 if (body.speed != null && body.speed !== "standard") {
96 return `Fast mode is not offered on the AI Gateway on g1t's models yet. ${own}`;
97 }
98 if (body.inference_geo != null && body.inference_geo !== "global") {
99 return `Only global inference is offered on the AI Gateway on g1t's models yet: leave out inference_geo. ${own}`;
100 }
101 if (body.fallbacks != null) {
102 return `Server-side fallbacks are not offered on the AI Gateway on g1t's models yet: leave out fallbacks. ${own}`;
103 }
104 if (body.container != null) {
105 return `Containers and skills are not offered on the AI Gateway on g1t's models yet. ${own}`;
106 }
107 const tools = Array.isArray(body.tools) ? (body.tools as unknown[]) : [];
108 for (const tool of tools) {
109 const type = (tool as { type?: unknown } | null)?.type;
110 if (type == null || (typeof type === "string" && CLIENT_TOOLS.some((prefix) => type === prefix || type.startsWith(prefix)))) continue;
111 return `Server tools such as web search and code execution (${String(type)}) are not offered on the AI Gateway on g1t's models yet. ${own}`;
112 }
113 return null;
114}
115
116/** A request's id: `gw_` and 24 random hex digits. */
117export function requestId(): string {
118 const bytes = crypto.getRandomValues(new Uint8Array(12));
119 return `gw_${[...bytes].map((b) => b.toString(16).padStart(2, "0")).join("")}`;
120}
121
122/**
123 * The session a request is logged under at Cloudflare's AI Gateway: one per
124 * token per UTC hour, so the gateway's own logs can be read back by token
125 * and hour.
126 */
127export function sessionOf(tokenId: string, now: Date): string {
128 const hour = now.toISOString().slice(0, 13).replace(/[-T]/g, "");
129 return `gw_${tokenId}_${hour}`;
130}
131
132/**
133 * Where a request to g1t's models goes and what it carries: the caller's
134 * request, without its token, to g1t's AI Gateway with g1t's credentials
135 * and tags for the workspace, token and session.
136 */
137export function hostedRequest(
138 hosted: HostedRouting,
139 path: string,
140 incoming: Headers,
141 caller: Caller,
142 session: string,
143): { url: string; headers: Headers } {
144 const headers = passedHeaders(incoming);
145 if (!hosted.AI_GATEWAY_ID) {
146 if (hosted.ANTHROPIC_API_KEY) headers.set("x-api-key", hosted.ANTHROPIC_API_KEY);
147 return { url: `https://api.anthropic.com${path}`, headers };
148 }
149 headers.set("cf-aig-metadata", JSON.stringify({ task: "gateway", workspace: caller.workspace, token: caller.tokenId, session }));
150 if (hosted.AI_GATEWAY_TOKEN) headers.set("cf-aig-authorization", `Bearer ${hosted.AI_GATEWAY_TOKEN}`);
151 if (hosted.ANTHROPIC_API_KEY) headers.set("x-api-key", hosted.ANTHROPIC_API_KEY);
152 return {
153 url: `https://gateway.ai.cloudflare.com/v1/${hosted.CLOUDFLARE_ACCOUNT_ID}/${hosted.AI_GATEWAY_ID}/anthropic${path}`,
154 headers,
155 };
156}
157
158/** The message of an Anthropic-shaped error body, or the status. */
159export function errorMessage(status: number, body: string): string {
160 try {
161 const parsed = JSON.parse(body) as { error?: { message?: unknown } };
162 if (typeof parsed.error?.message === "string") return parsed.error.message.slice(0, 500);
163 } catch {
164 // Not JSON: the status says enough.
165 }
166 return `The model provider answered ${status}.`;
167}
168
169/** What billing is told about one request. */
170export function gatewayRecord(input: {
171 id: string;
172 caller: Caller;
173 model: string;
174 tokens: Tokens;
175 status: number;
176 ownKey: boolean;
177 streamed: boolean;
178 durationMs: number;
179 error?: string | null;
180}): GatewayRecord {
181 return {
182 id: input.id,
183 workspace: input.caller.workspace,
184 tokenId: input.caller.tokenId,
185 tokenName: input.caller.tokenName,
186 model: input.model.slice(0, 200) || "unknown",
187 input: input.tokens.input,
188 output: input.tokens.output,
189 cacheRead: input.tokens.cacheRead,
190 cacheWrite: input.tokens.cacheWrite,
191 status: input.status,
192 ownKey: input.ownKey,
193 streamed: input.streamed,
194 durationMs: Math.max(0, Math.round(input.durationMs)),
195 error: input.error ?? null,
196 };
197}