Skip to content

g1t/services/models/src/openai.ts

397 lines15,854 bytesCodeBlame

Pick any line to see why it is the way it is: the commit, the pull request and issue it came from, and what the agent was thinking.

Models per workspace: several providers, routed by kind of work1/**
2 * Speaking OpenAI's Chat Completions API on behalf of a harness that speaks
3 * Anthropic's Messages API.
4 *
5 * g1t's agents run Claude Code, which only speaks Anthropic's API. A
6 * workspace whose provider speaks OpenAI's (OpenAI itself, Gemini's
7 * compatible endpoint, OpenRouter, Groq, vLLM, Ollama…) still gets agents:
8 * the proxy turns each request into a chat completion and turns the answer,
9 * streamed or not, back into what Anthropic would have sent, tool calls
10 * included.
11 */
12
13type Json = Record<string, unknown>;
14
15type AnthropicBlock =
16 | { type: "text"; text: string }
17 | { type: "image"; source: { type: "base64"; media_type: string; data: string } | { type: "url"; url: string } }
18 | { type: "tool_use"; id: string; name: string; input: unknown }
19 | { type: "tool_result"; tool_use_id: string; content?: string | AnthropicBlock[]; is_error?: boolean }
20 | { type: "thinking" | "redacted_thinking"; [key: string]: unknown };
21
22type AnthropicMessage = { role: "user" | "assistant"; content: string | AnthropicBlock[] };
23
24export type AnthropicRequest = {
25 model?: string;
26 system?: string | { type: "text"; text: string }[];
27 messages: AnthropicMessage[];
28 tools?: { name: string; description?: string; input_schema?: unknown; type?: string }[];
29 tool_choice?: { type: "auto" | "any" | "tool" | "none"; name?: string };
30 max_tokens?: number;
31 temperature?: number;
32 top_p?: number;
33 stop_sequences?: string[];
34 stream?: boolean;
AI Gateway: OpenAI's format, open models, and your own providers35 /** Effort, and a JSON schema the answer follows. */
36 output_config?: { effort?: string; format?: { type?: string; schema?: unknown } };
Models per workspace: several providers, routed by kind of work37};
38
AI Gateway: OpenAI's format, open models, and your own providers39/**
40 * Anthropic's effort as OpenAI's `reasoning_effort`, which stops at `high`.
41 */
42export function reasoningEffort(effort: unknown): string | undefined {
43 if (effort === "low" || effort === "medium" || effort === "high") return effort;
44 if (effort === "xhigh" || effort === "max") return "high";
45 return undefined;
46}
47
Models per workspace: several providers, routed by kind of work48type ChatMessage =
49 | { role: "system"; content: string }
50 | { role: "user"; content: string | Json[] }
51 | { role: "assistant"; content: string | null; tool_calls?: Json[] }
52 | { role: "tool"; tool_call_id: string; content: string };
53
54/** How the provider wants the request shaped. */
55export type Dialect = {
56 /** OpenAI's own API: `max_completion_tokens`, and no temperature for its reasoning models. */
57 official: boolean;
A catalogue of model providers, and settings that feel like settings58 /** Which provider, by name, for the quirks of each. */
59 provider?: string;
Models per workspace: several providers, routed by kind of work60};
61
A catalogue of model providers, and settings that feel like settings62/** The most output tokens a provider accepts in one answer, where it caps them below what the harness asks. */
63const MAX_OUTPUT: Record<string, number> = { deepseek: 8192, groq: 32768, cerebras: 32768 };
64
65/**
66 * Some providers attach data to a tool call that has to come back with it
67 * on the next turn, such as Gemini's thought signatures. The harness keeps
68 * nothing but the call's id, so the data rides in the id.
69 */
70const CARRIED = "__g1t_";
71
72function toBase64Url(text: string): string {
73 let binary = "";
74 for (const byte of new TextEncoder().encode(text)) binary += String.fromCharCode(byte);
75 return btoa(binary).replace(/\+/g, "-").replace(/\//g, "_").replace(/=+$/, "");
76}
77
78function fromBase64Url(encoded: string): string {
79 const binary = atob(encoded.replace(/-/g, "+").replace(/_/g, "/"));
80 return new TextDecoder().decode(Uint8Array.from(binary, (c) => c.charCodeAt(0)));
81}
82
83/** A tool call's id, carrying whatever the provider attached to the call. */
84export function carryId(id: string, extra: unknown): string {
85 return extra == null ? id : `${id}${CARRIED}${toBase64Url(JSON.stringify(extra))}`;
86}
87
88/** The provider's own id, and what it attached, from a carried id. */
89export function uncarryId(id: string): { id: string; extra: unknown } {
90 const at = id.indexOf(CARRIED);
91 if (at < 0) return { id, extra: undefined };
92 try {
93 return { id: id.slice(0, at), extra: JSON.parse(fromBase64Url(id.slice(at + CARRIED.length))) };
94 } catch {
95 return { id: id.slice(0, at), extra: undefined };
96 }
97}
98
Models per workspace: several providers, routed by kind of work99function text(content: string | AnthropicBlock[] | undefined): string {
100 if (content == null) return "";
101 if (typeof content === "string") return content;
102 return content
103 .map((block) => (block.type === "text" ? block.text : ""))
104 .filter(Boolean)
105 .join("\n");
106}
107
108/** A chat completion request saying what the Anthropic request said. */
109export function toChat(request: AnthropicRequest, model: string, dialect: Dialect): Json {
110 const messages: ChatMessage[] = [];
111 const system = typeof request.system === "string" ? request.system : text(request.system as AnthropicBlock[] | undefined);
112 if (system) messages.push({ role: "system", content: system });
113
114 for (const message of request.messages) {
115 const blocks: AnthropicBlock[] =
116 typeof message.content === "string" ? [{ type: "text", text: message.content }] : message.content;
117 if (message.role === "assistant") {
118 const said = blocks.filter((b) => b.type === "text").map((b) => (b as { text: string }).text).join("");
119 const calls = blocks
120 .filter((b): b is Extract<AnthropicBlock, { type: "tool_use" }> => b.type === "tool_use")
A catalogue of model providers, and settings that feel like settings121 .map((b) => {
122 const { id, extra } = uncarryId(b.id);
123 return {
124 id,
125 type: "function",
126 function: { name: b.name, arguments: JSON.stringify(b.input ?? {}) },
127 ...(extra === undefined ? {} : { extra_content: extra }),
128 };
129 });
Models per workspace: several providers, routed by kind of work130 messages.push({ role: "assistant", content: said || null, ...(calls.length ? { tool_calls: calls } : {}) });
131 continue;
132 }
133 // Tool results answer the assistant's calls, so they go first, each as
134 // its own message; whatever else the user said follows.
135 for (const block of blocks) {
136 if (block.type !== "tool_result") continue;
137 const result = text(block.content);
A catalogue of model providers, and settings that feel like settings138 messages.push({
139 role: "tool",
140 tool_call_id: uncarryId(block.tool_use_id).id,
141 content: block.is_error ? `Error: ${result}` : result,
142 });
Models per workspace: several providers, routed by kind of work143 }
144 const parts: Json[] = [];
145 for (const block of blocks) {
146 if (block.type === "text" && block.text) parts.push({ type: "text", text: block.text });
147 if (block.type === "image") {
148 const url = block.source.type === "base64" ? `data:${block.source.media_type};base64,${block.source.data}` : block.source.url;
149 parts.push({ type: "image_url", image_url: { url } });
150 }
151 }
152 if (parts.length === 1 && parts[0].type === "text") messages.push({ role: "user", content: parts[0].text as string });
153 else if (parts.length > 0) messages.push({ role: "user", content: parts });
154 }
155
156 const body: Json = { model, messages, stream: request.stream === true };
A catalogue of model providers, and settings that feel like settings157 // Usage at the end of a stream; Mistral refuses the option.
158 if (request.stream && dialect.provider !== "mistral") body.stream_options = { include_usage: true };
159 const cap = MAX_OUTPUT[dialect.provider ?? ""];
160 const maxTokens = request.max_tokens && cap ? Math.min(request.max_tokens, cap) : request.max_tokens;
161 if (maxTokens) body[dialect.official ? "max_completion_tokens" : "max_tokens"] = maxTokens;
Models per workspace: several providers, routed by kind of work162 if (!dialect.official && request.temperature != null) body.temperature = request.temperature;
163 if (!dialect.official && request.top_p != null) body.top_p = request.top_p;
164 if (request.stop_sequences?.length) body.stop = request.stop_sequences.slice(0, 4);
AI Gateway: OpenAI's format, open models, and your own providers165 const effort = reasoningEffort(request.output_config?.effort);
166 if (effort) body.reasoning_effort = effort;
167 const format = request.output_config?.format;
168 if (format?.type === "json_schema" && format.schema) {
169 body.response_format = { type: "json_schema", json_schema: { name: "answer", schema: format.schema, strict: true } };
170 }
Models per workspace: several providers, routed by kind of work171 // Anthropic's own server tools (web search and the like) have no
172 // counterpart; functions do.
173 const tools = (request.tools ?? []).filter((tool) => tool.input_schema != null);
174 if (tools.length) {
175 body.tools = tools.map((tool) => ({
176 type: "function",
177 function: { name: tool.name, description: tool.description ?? "", parameters: tool.input_schema },
178 }));
179 const choice = request.tool_choice;
A catalogue of model providers, and settings that feel like settings180 if (choice?.type === "any") body.tool_choice = dialect.provider === "mistral" ? "any" : "required";
Models per workspace: several providers, routed by kind of work181 else if (choice?.type === "tool" && choice.name) body.tool_choice = { type: "function", function: { name: choice.name } };
182 else if (choice?.type === "none") body.tool_choice = "none";
183 }
184 return body;
185}
186
187const STOP: Record<string, string> = {
188 stop: "end_turn",
189 length: "max_tokens",
190 tool_calls: "tool_use",
191 function_call: "tool_use",
192 content_filter: "end_turn",
193};
194
195function parseArguments(raw: string): unknown {
196 try {
197 return raw ? JSON.parse(raw) : {};
198 } catch {
199 return {};
200 }
201}
202
203/** An Anthropic message saying what a (non-streamed) chat completion said. */
204export function fromChat(completion: Json, model: string): Json {
205 const choice = ((completion.choices as Json[] | undefined) ?? [])[0] ?? {};
206 const message = (choice.message as Json | undefined) ?? {};
207 const content: Json[] = [];
208 if (typeof message.content === "string" && message.content) content.push({ type: "text", text: message.content });
209 for (const call of (message.tool_calls as Json[] | undefined) ?? []) {
210 const fn = call.function as Json;
A catalogue of model providers, and settings that feel like settings211 content.push({
212 type: "tool_use",
213 id: carryId(String(call.id), call.extra_content),
214 name: fn.name,
215 input: parseArguments(String(fn.arguments ?? "")),
216 });
Models per workspace: several providers, routed by kind of work217 }
218 return {
219 id: `msg_${String(completion.id ?? crypto.randomUUID()).replace(/[^A-Za-z0-9_-]/g, "")}`,
220 type: "message",
221 role: "assistant",
222 model,
223 content,
224 stop_reason: STOP[String(choice.finish_reason)] ?? "end_turn",
225 stop_sequence: null,
AI Gateway: OpenAI's format, open models, and your own providers226 usage: anthropicUsage(completion.usage as Json | undefined),
227 };
228}
229
230/**
231 * A chat completion's usage as Anthropic counts it: prompt tokens read
232 * from the cache are cache reads, the rest input.
233 */
234export function anthropicUsage(usage: Json | undefined | null): Json {
235 const n = (value: unknown) => (typeof value === "number" && Number.isFinite(value) && value > 0 ? value : 0);
236 const prompt = n(usage?.prompt_tokens);
237 const details = usage?.prompt_tokens_details as Json | undefined;
238 const cached = Math.min(prompt, n(details?.cached_tokens) || n(usage?.prompt_cache_hit_tokens));
239 return {
240 input_tokens: prompt - cached,
241 output_tokens: n(usage?.completion_tokens),
242 ...(cached ? { cache_read_input_tokens: cached } : {}),
Models per workspace: several providers, routed by kind of work243 };
244}
245
246function event(name: string, data: Json): string {
247 return `event: ${name}\ndata: ${JSON.stringify({ type: name, ...data })}\n\n`;
248}
249
250/**
251 * Turns a chat completion's server-sent events into the events Anthropic's
252 * API streams: a message, its content blocks one at a time (text, or a tool
253 * call whose input arrives in pieces), then why it stopped.
254 */
255export class StreamTranslator {
256 private started = false;
257 private open: { kind: "text" } | { kind: "tool"; slot: number } | null = null;
258 private index = -1;
259 private stopReason = "end_turn";
AI Gateway: OpenAI's format, open models, and your own providers260 private usage: Json = { input_tokens: 0, output_tokens: 0 };
Models per workspace: several providers, routed by kind of work261 private buffer = "";
262 private finished = false;
263
264 private readonly model: string;
265
266 constructor(model: string) {
267 this.model = model;
268 }
269
270 private start(id: string): string {
271 if (this.started) return "";
272 this.started = true;
273 return event("message_start", {
274 message: {
275 id: `msg_${id.replace(/[^A-Za-z0-9_-]/g, "") || crypto.randomUUID()}`,
276 type: "message",
277 role: "assistant",
278 model: this.model,
279 content: [],
280 stop_reason: null,
281 stop_sequence: null,
282 usage: { input_tokens: 0, output_tokens: 0 },
283 },
284 });
285 }
286
287 private close(): string {
288 if (!this.open) return "";
289 this.open = null;
290 return event("content_block_stop", { index: this.index });
291 }
292
293 private chunk(data: Json): string {
294 let out = this.start(String(data.id ?? ""));
295 const usage = data.usage as Json | undefined;
296 if (usage) {
AI Gateway: OpenAI's format, open models, and your own providers297 this.usage = anthropicUsage(usage);
Models per workspace: several providers, routed by kind of work298 }
299 for (const choice of (data.choices as Json[] | undefined) ?? []) {
300 const delta = (choice.delta as Json | undefined) ?? {};
301 if (typeof delta.content === "string" && delta.content) {
302 if (this.open?.kind !== "text") {
303 out += this.close();
304 this.index += 1;
305 this.open = { kind: "text" };
306 out += event("content_block_start", { index: this.index, content_block: { type: "text", text: "" } });
307 }
308 out += event("content_block_delta", { index: this.index, delta: { type: "text_delta", text: delta.content } });
309 }
310 for (const call of (delta.tool_calls as Json[] | undefined) ?? []) {
311 const slot = Number(call.index ?? 0);
312 const fn = (call.function as Json | undefined) ?? {};
313 if (!(this.open?.kind === "tool" && this.open.slot === slot)) {
314 out += this.close();
315 this.index += 1;
316 this.open = { kind: "tool", slot };
317 out += event("content_block_start", {
318 index: this.index,
A catalogue of model providers, and settings that feel like settings319 content_block: {
320 type: "tool_use",
321 id: carryId(String(call.id ?? `call_${this.index}`), call.extra_content),
322 name: String(fn.name ?? ""),
323 input: {},
324 },
Models per workspace: several providers, routed by kind of work325 });
326 }
327 if (typeof fn.arguments === "string" && fn.arguments) {
328 out += event("content_block_delta", {
329 index: this.index,
330 delta: { type: "input_json_delta", partial_json: fn.arguments },
331 });
332 }
333 }
334 if (choice.finish_reason) this.stopReason = STOP[String(choice.finish_reason)] ?? "end_turn";
335 }
336 return out;
337 }
338
339 /** Takes raw bytes of the upstream stream; returns Anthropic events to send. */
340 push(piece: string): string {
341 this.buffer += piece;
342 let out = "";
343 let at: number;
344 while ((at = this.buffer.indexOf("\n")) >= 0) {
345 const line = this.buffer.slice(0, at).trim();
346 this.buffer = this.buffer.slice(at + 1);
347 if (!line.startsWith("data:")) continue;
348 const payload = line.slice(5).trim();
349 if (payload === "[DONE]") {
350 out += this.finish();
351 continue;
352 }
353 try {
354 out += this.chunk(JSON.parse(payload) as Json);
355 } catch {
356 // A line that is not JSON carries nothing to translate.
357 }
358 }
359 return out;
360 }
361
362 /** The closing events, once. */
363 finish(): string {
364 if (this.finished) return "";
365 this.finished = true;
366 return (
367 this.start("") +
368 this.close() +
369 event("message_delta", {
370 delta: { stop_reason: this.stopReason, stop_sequence: null },
AI Gateway: OpenAI's format, open models, and your own providers371 usage: this.usage,
Models per workspace: several providers, routed by kind of work372 }) +
373 event("message_stop", {})
374 );
375 }
376}
377
378/** A provider's error, in the shape Anthropic's API uses. */
379export function errorFromChat(status: number, body: string): Json {
380 let message = body.slice(0, 500);
381 try {
382 const parsed = JSON.parse(body) as Json;
383 const error = (Array.isArray(parsed) ? parsed[0]?.error : parsed.error) as Json | string | undefined;
384 if (typeof error === "string") message = error;
385 else if (error && typeof error.message === "string") message = error.message;
386 } catch {
387 // Not JSON: keep the text.
388 }
389 const type =
390 status === 401 ? "authentication_error" : status === 429 ? "rate_limit_error" : status === 404 ? "not_found_error" : status >= 500 ? "api_error" : "invalid_request_error";
391 return { type: "error", error: { type, message: `The provider said: ${message}` } };
392}
393
394/** A rough count, for the harness's token counting, which chat APIs lack. */
395export function estimateTokens(request: AnthropicRequest): number {
396 return Math.ceil(JSON.stringify({ system: request.system, messages: request.messages, tools: request.tools }).length / 4);
397}