g1t/services/models/src/openai.ts
Pick any line to see why it is the way it is: the commit, the pull request and issue it came from, and what the agent was thinking.
| Models per workspace: several providers, routed by kind of work | 1 | /** |
| 2 | * Speaking OpenAI's Chat Completions API on behalf of a harness that speaks | |
| 3 | * Anthropic's Messages API. | |
| 4 | * | |
| 5 | * g1t's agents run Claude Code, which only speaks Anthropic's API. A | |
| 6 | * workspace whose provider speaks OpenAI's (OpenAI itself, Gemini's | |
| 7 | * compatible endpoint, OpenRouter, Groq, vLLM, Ollama…) still gets agents: | |
| 8 | * the proxy turns each request into a chat completion and turns the answer, | |
| 9 | * streamed or not, back into what Anthropic would have sent, tool calls | |
| 10 | * included. | |
| 11 | */ | |
| 12 | ||
| 13 | type Json = Record<string, unknown>; | |
| 14 | ||
| 15 | type AnthropicBlock = | |
| 16 | | { type: "text"; text: string } | |
| 17 | | { type: "image"; source: { type: "base64"; media_type: string; data: string } | { type: "url"; url: string } } | |
| 18 | | { type: "tool_use"; id: string; name: string; input: unknown } | |
| 19 | | { type: "tool_result"; tool_use_id: string; content?: string | AnthropicBlock[]; is_error?: boolean } | |
| 20 | | { type: "thinking" | "redacted_thinking"; [key: string]: unknown }; | |
| 21 | ||
| 22 | type AnthropicMessage = { role: "user" | "assistant"; content: string | AnthropicBlock[] }; | |
| 23 | ||
| 24 | export type AnthropicRequest = { | |
| 25 | model?: string; | |
| 26 | system?: string | { type: "text"; text: string }[]; | |
| 27 | messages: AnthropicMessage[]; | |
| 28 | tools?: { name: string; description?: string; input_schema?: unknown; type?: string }[]; | |
| 29 | tool_choice?: { type: "auto" | "any" | "tool" | "none"; name?: string }; | |
| 30 | max_tokens?: number; | |
| 31 | temperature?: number; | |
| 32 | top_p?: number; | |
| 33 | stop_sequences?: string[]; | |
| 34 | stream?: boolean; | |
| 35 | }; | |
| 36 | ||
| 37 | type ChatMessage = | |
| 38 | | { role: "system"; content: string } | |
| 39 | | { role: "user"; content: string | Json[] } | |
| 40 | | { role: "assistant"; content: string | null; tool_calls?: Json[] } | |
| 41 | | { role: "tool"; tool_call_id: string; content: string }; | |
| 42 | ||
| 43 | /** How the provider wants the request shaped. */ | |
| 44 | export type Dialect = { | |
| 45 | /** OpenAI's own API: `max_completion_tokens`, and no temperature for its reasoning models. */ | |
| 46 | official: boolean; | |
| A catalogue of model providers, and settings that feel like settings | 47 | /** Which provider, by name, for the quirks of each. */ |
| 48 | provider?: string; | |
| Models per workspace: several providers, routed by kind of work | 49 | }; |
| 50 | ||
| A catalogue of model providers, and settings that feel like settings | 51 | /** The most output tokens a provider accepts in one answer, where it caps them below what the harness asks. */ |
| 52 | const MAX_OUTPUT: Record<string, number> = { deepseek: 8192, groq: 32768, cerebras: 32768 }; | |
| 53 | ||
| 54 | /** | |
| 55 | * Some providers attach data to a tool call that has to come back with it | |
| 56 | * on the next turn, such as Gemini's thought signatures. The harness keeps | |
| 57 | * nothing but the call's id, so the data rides in the id. | |
| 58 | */ | |
| 59 | const CARRIED = "__g1t_"; | |
| 60 | ||
| 61 | function toBase64Url(text: string): string { | |
| 62 | let binary = ""; | |
| 63 | for (const byte of new TextEncoder().encode(text)) binary += String.fromCharCode(byte); | |
| 64 | return btoa(binary).replace(/\+/g, "-").replace(/\//g, "_").replace(/=+$/, ""); | |
| 65 | } | |
| 66 | ||
| 67 | function fromBase64Url(encoded: string): string { | |
| 68 | const binary = atob(encoded.replace(/-/g, "+").replace(/_/g, "/")); | |
| 69 | return new TextDecoder().decode(Uint8Array.from(binary, (c) => c.charCodeAt(0))); | |
| 70 | } | |
| 71 | ||
| 72 | /** A tool call's id, carrying whatever the provider attached to the call. */ | |
| 73 | export function carryId(id: string, extra: unknown): string { | |
| 74 | return extra == null ? id : `${id}${CARRIED}${toBase64Url(JSON.stringify(extra))}`; | |
| 75 | } | |
| 76 | ||
| 77 | /** The provider's own id, and what it attached, from a carried id. */ | |
| 78 | export function uncarryId(id: string): { id: string; extra: unknown } { | |
| 79 | const at = id.indexOf(CARRIED); | |
| 80 | if (at < 0) return { id, extra: undefined }; | |
| 81 | try { | |
| 82 | return { id: id.slice(0, at), extra: JSON.parse(fromBase64Url(id.slice(at + CARRIED.length))) }; | |
| 83 | } catch { | |
| 84 | return { id: id.slice(0, at), extra: undefined }; | |
| 85 | } | |
| 86 | } | |
| 87 | ||
| Models per workspace: several providers, routed by kind of work | 88 | function text(content: string | AnthropicBlock[] | undefined): string { |
| 89 | if (content == null) return ""; | |
| 90 | if (typeof content === "string") return content; | |
| 91 | return content | |
| 92 | .map((block) => (block.type === "text" ? block.text : "")) | |
| 93 | .filter(Boolean) | |
| 94 | .join("\n"); | |
| 95 | } | |
| 96 | ||
| 97 | /** A chat completion request saying what the Anthropic request said. */ | |
| 98 | export function toChat(request: AnthropicRequest, model: string, dialect: Dialect): Json { | |
| 99 | const messages: ChatMessage[] = []; | |
| 100 | const system = typeof request.system === "string" ? request.system : text(request.system as AnthropicBlock[] | undefined); | |
| 101 | if (system) messages.push({ role: "system", content: system }); | |
| 102 | ||
| 103 | for (const message of request.messages) { | |
| 104 | const blocks: AnthropicBlock[] = | |
| 105 | typeof message.content === "string" ? [{ type: "text", text: message.content }] : message.content; | |
| 106 | if (message.role === "assistant") { | |
| 107 | const said = blocks.filter((b) => b.type === "text").map((b) => (b as { text: string }).text).join(""); | |
| 108 | const calls = blocks | |
| 109 | .filter((b): b is Extract<AnthropicBlock, { type: "tool_use" }> => b.type === "tool_use") | |
| A catalogue of model providers, and settings that feel like settings | 110 | .map((b) => { |
| 111 | const { id, extra } = uncarryId(b.id); | |
| 112 | return { | |
| 113 | id, | |
| 114 | type: "function", | |
| 115 | function: { name: b.name, arguments: JSON.stringify(b.input ?? {}) }, | |
| 116 | ...(extra === undefined ? {} : { extra_content: extra }), | |
| 117 | }; | |
| 118 | }); | |
| Models per workspace: several providers, routed by kind of work | 119 | messages.push({ role: "assistant", content: said || null, ...(calls.length ? { tool_calls: calls } : {}) }); |
| 120 | continue; | |
| 121 | } | |
| 122 | // Tool results answer the assistant's calls, so they go first, each as | |
| 123 | // its own message; whatever else the user said follows. | |
| 124 | for (const block of blocks) { | |
| 125 | if (block.type !== "tool_result") continue; | |
| 126 | const result = text(block.content); | |
| A catalogue of model providers, and settings that feel like settings | 127 | messages.push({ |
| 128 | role: "tool", | |
| 129 | tool_call_id: uncarryId(block.tool_use_id).id, | |
| 130 | content: block.is_error ? `Error: ${result}` : result, | |
| 131 | }); | |
| Models per workspace: several providers, routed by kind of work | 132 | } |
| 133 | const parts: Json[] = []; | |
| 134 | for (const block of blocks) { | |
| 135 | if (block.type === "text" && block.text) parts.push({ type: "text", text: block.text }); | |
| 136 | if (block.type === "image") { | |
| 137 | const url = block.source.type === "base64" ? `data:${block.source.media_type};base64,${block.source.data}` : block.source.url; | |
| 138 | parts.push({ type: "image_url", image_url: { url } }); | |
| 139 | } | |
| 140 | } | |
| 141 | if (parts.length === 1 && parts[0].type === "text") messages.push({ role: "user", content: parts[0].text as string }); | |
| 142 | else if (parts.length > 0) messages.push({ role: "user", content: parts }); | |
| 143 | } | |
| 144 | ||
| 145 | const body: Json = { model, messages, stream: request.stream === true }; | |
| A catalogue of model providers, and settings that feel like settings | 146 | // Usage at the end of a stream; Mistral refuses the option. |
| 147 | if (request.stream && dialect.provider !== "mistral") body.stream_options = { include_usage: true }; | |
| 148 | const cap = MAX_OUTPUT[dialect.provider ?? ""]; | |
| 149 | const maxTokens = request.max_tokens && cap ? Math.min(request.max_tokens, cap) : request.max_tokens; | |
| 150 | if (maxTokens) body[dialect.official ? "max_completion_tokens" : "max_tokens"] = maxTokens; | |
| Models per workspace: several providers, routed by kind of work | 151 | if (!dialect.official && request.temperature != null) body.temperature = request.temperature; |
| 152 | if (!dialect.official && request.top_p != null) body.top_p = request.top_p; | |
| 153 | if (request.stop_sequences?.length) body.stop = request.stop_sequences.slice(0, 4); | |
| 154 | // Anthropic's own server tools (web search and the like) have no | |
| 155 | // counterpart; functions do. | |
| 156 | const tools = (request.tools ?? []).filter((tool) => tool.input_schema != null); | |
| 157 | if (tools.length) { | |
| 158 | body.tools = tools.map((tool) => ({ | |
| 159 | type: "function", | |
| 160 | function: { name: tool.name, description: tool.description ?? "", parameters: tool.input_schema }, | |
| 161 | })); | |
| 162 | const choice = request.tool_choice; | |
| A catalogue of model providers, and settings that feel like settings | 163 | if (choice?.type === "any") body.tool_choice = dialect.provider === "mistral" ? "any" : "required"; |
| Models per workspace: several providers, routed by kind of work | 164 | else if (choice?.type === "tool" && choice.name) body.tool_choice = { type: "function", function: { name: choice.name } }; |
| 165 | else if (choice?.type === "none") body.tool_choice = "none"; | |
| 166 | } | |
| 167 | return body; | |
| 168 | } | |
| 169 | ||
| 170 | const STOP: Record<string, string> = { | |
| 171 | stop: "end_turn", | |
| 172 | length: "max_tokens", | |
| 173 | tool_calls: "tool_use", | |
| 174 | function_call: "tool_use", | |
| 175 | content_filter: "end_turn", | |
| 176 | }; | |
| 177 | ||
| 178 | function parseArguments(raw: string): unknown { | |
| 179 | try { | |
| 180 | return raw ? JSON.parse(raw) : {}; | |
| 181 | } catch { | |
| 182 | return {}; | |
| 183 | } | |
| 184 | } | |
| 185 | ||
| 186 | /** An Anthropic message saying what a (non-streamed) chat completion said. */ | |
| 187 | export function fromChat(completion: Json, model: string): Json { | |
| 188 | const choice = ((completion.choices as Json[] | undefined) ?? [])[0] ?? {}; | |
| 189 | const message = (choice.message as Json | undefined) ?? {}; | |
| 190 | const content: Json[] = []; | |
| 191 | if (typeof message.content === "string" && message.content) content.push({ type: "text", text: message.content }); | |
| 192 | for (const call of (message.tool_calls as Json[] | undefined) ?? []) { | |
| 193 | const fn = call.function as Json; | |
| A catalogue of model providers, and settings that feel like settings | 194 | content.push({ |
| 195 | type: "tool_use", | |
| 196 | id: carryId(String(call.id), call.extra_content), | |
| 197 | name: fn.name, | |
| 198 | input: parseArguments(String(fn.arguments ?? "")), | |
| 199 | }); | |
| Models per workspace: several providers, routed by kind of work | 200 | } |
| 201 | const usage = (completion.usage as Json | undefined) ?? {}; | |
| 202 | return { | |
| 203 | id: `msg_${String(completion.id ?? crypto.randomUUID()).replace(/[^A-Za-z0-9_-]/g, "")}`, | |
| 204 | type: "message", | |
| 205 | role: "assistant", | |
| 206 | model, | |
| 207 | content, | |
| 208 | stop_reason: STOP[String(choice.finish_reason)] ?? "end_turn", | |
| 209 | stop_sequence: null, | |
| 210 | usage: { input_tokens: Number(usage.prompt_tokens ?? 0), output_tokens: Number(usage.completion_tokens ?? 0) }, | |
| 211 | }; | |
| 212 | } | |
| 213 | ||
| 214 | function event(name: string, data: Json): string { | |
| 215 | return `event: ${name}\ndata: ${JSON.stringify({ type: name, ...data })}\n\n`; | |
| 216 | } | |
| 217 | ||
| 218 | /** | |
| 219 | * Turns a chat completion's server-sent events into the events Anthropic's | |
| 220 | * API streams: a message, its content blocks one at a time (text, or a tool | |
| 221 | * call whose input arrives in pieces), then why it stopped. | |
| 222 | */ | |
| 223 | export class StreamTranslator { | |
| 224 | private started = false; | |
| 225 | private open: { kind: "text" } | { kind: "tool"; slot: number } | null = null; | |
| 226 | private index = -1; | |
| 227 | private stopReason = "end_turn"; | |
| 228 | private inputTokens = 0; | |
| 229 | private outputTokens = 0; | |
| 230 | private buffer = ""; | |
| 231 | private finished = false; | |
| 232 | ||
| 233 | private readonly model: string; | |
| 234 | ||
| 235 | constructor(model: string) { | |
| 236 | this.model = model; | |
| 237 | } | |
| 238 | ||
| 239 | private start(id: string): string { | |
| 240 | if (this.started) return ""; | |
| 241 | this.started = true; | |
| 242 | return event("message_start", { | |
| 243 | message: { | |
| 244 | id: `msg_${id.replace(/[^A-Za-z0-9_-]/g, "") || crypto.randomUUID()}`, | |
| 245 | type: "message", | |
| 246 | role: "assistant", | |
| 247 | model: this.model, | |
| 248 | content: [], | |
| 249 | stop_reason: null, | |
| 250 | stop_sequence: null, | |
| 251 | usage: { input_tokens: 0, output_tokens: 0 }, | |
| 252 | }, | |
| 253 | }); | |
| 254 | } | |
| 255 | ||
| 256 | private close(): string { | |
| 257 | if (!this.open) return ""; | |
| 258 | this.open = null; | |
| 259 | return event("content_block_stop", { index: this.index }); | |
| 260 | } | |
| 261 | ||
| 262 | private chunk(data: Json): string { | |
| 263 | let out = this.start(String(data.id ?? "")); | |
| 264 | const usage = data.usage as Json | undefined; | |
| 265 | if (usage) { | |
| 266 | this.inputTokens = Number(usage.prompt_tokens ?? this.inputTokens); | |
| 267 | this.outputTokens = Number(usage.completion_tokens ?? this.outputTokens); | |
| 268 | } | |
| 269 | for (const choice of (data.choices as Json[] | undefined) ?? []) { | |
| 270 | const delta = (choice.delta as Json | undefined) ?? {}; | |
| 271 | if (typeof delta.content === "string" && delta.content) { | |
| 272 | if (this.open?.kind !== "text") { | |
| 273 | out += this.close(); | |
| 274 | this.index += 1; | |
| 275 | this.open = { kind: "text" }; | |
| 276 | out += event("content_block_start", { index: this.index, content_block: { type: "text", text: "" } }); | |
| 277 | } | |
| 278 | out += event("content_block_delta", { index: this.index, delta: { type: "text_delta", text: delta.content } }); | |
| 279 | } | |
| 280 | for (const call of (delta.tool_calls as Json[] | undefined) ?? []) { | |
| 281 | const slot = Number(call.index ?? 0); | |
| 282 | const fn = (call.function as Json | undefined) ?? {}; | |
| 283 | if (!(this.open?.kind === "tool" && this.open.slot === slot)) { | |
| 284 | out += this.close(); | |
| 285 | this.index += 1; | |
| 286 | this.open = { kind: "tool", slot }; | |
| 287 | out += event("content_block_start", { | |
| 288 | index: this.index, | |
| A catalogue of model providers, and settings that feel like settings | 289 | content_block: { |
| 290 | type: "tool_use", | |
| 291 | id: carryId(String(call.id ?? `call_${this.index}`), call.extra_content), | |
| 292 | name: String(fn.name ?? ""), | |
| 293 | input: {}, | |
| 294 | }, | |
| Models per workspace: several providers, routed by kind of work | 295 | }); |
| 296 | } | |
| 297 | if (typeof fn.arguments === "string" && fn.arguments) { | |
| 298 | out += event("content_block_delta", { | |
| 299 | index: this.index, | |
| 300 | delta: { type: "input_json_delta", partial_json: fn.arguments }, | |
| 301 | }); | |
| 302 | } | |
| 303 | } | |
| 304 | if (choice.finish_reason) this.stopReason = STOP[String(choice.finish_reason)] ?? "end_turn"; | |
| 305 | } | |
| 306 | return out; | |
| 307 | } | |
| 308 | ||
| 309 | /** Takes raw bytes of the upstream stream; returns Anthropic events to send. */ | |
| 310 | push(piece: string): string { | |
| 311 | this.buffer += piece; | |
| 312 | let out = ""; | |
| 313 | let at: number; | |
| 314 | while ((at = this.buffer.indexOf("\n")) >= 0) { | |
| 315 | const line = this.buffer.slice(0, at).trim(); | |
| 316 | this.buffer = this.buffer.slice(at + 1); | |
| 317 | if (!line.startsWith("data:")) continue; | |
| 318 | const payload = line.slice(5).trim(); | |
| 319 | if (payload === "[DONE]") { | |
| 320 | out += this.finish(); | |
| 321 | continue; | |
| 322 | } | |
| 323 | try { | |
| 324 | out += this.chunk(JSON.parse(payload) as Json); | |
| 325 | } catch { | |
| 326 | // A line that is not JSON carries nothing to translate. | |
| 327 | } | |
| 328 | } | |
| 329 | return out; | |
| 330 | } | |
| 331 | ||
| 332 | /** The closing events, once. */ | |
| 333 | finish(): string { | |
| 334 | if (this.finished) return ""; | |
| 335 | this.finished = true; | |
| 336 | return ( | |
| 337 | this.start("") + | |
| 338 | this.close() + | |
| 339 | event("message_delta", { | |
| 340 | delta: { stop_reason: this.stopReason, stop_sequence: null }, | |
| 341 | usage: { input_tokens: this.inputTokens, output_tokens: this.outputTokens }, | |
| 342 | }) + | |
| 343 | event("message_stop", {}) | |
| 344 | ); | |
| 345 | } | |
| 346 | } | |
| 347 | ||
| 348 | /** A provider's error, in the shape Anthropic's API uses. */ | |
| 349 | export function errorFromChat(status: number, body: string): Json { | |
| 350 | let message = body.slice(0, 500); | |
| 351 | try { | |
| 352 | const parsed = JSON.parse(body) as Json; | |
| 353 | const error = (Array.isArray(parsed) ? parsed[0]?.error : parsed.error) as Json | string | undefined; | |
| 354 | if (typeof error === "string") message = error; | |
| 355 | else if (error && typeof error.message === "string") message = error.message; | |
| 356 | } catch { | |
| 357 | // Not JSON: keep the text. | |
| 358 | } | |
| 359 | const type = | |
| 360 | status === 401 ? "authentication_error" : status === 429 ? "rate_limit_error" : status === 404 ? "not_found_error" : status >= 500 ? "api_error" : "invalid_request_error"; | |
| 361 | return { type: "error", error: { type, message: `The provider said: ${message}` } }; | |
| 362 | } | |
| 363 | ||
| 364 | /** A rough count, for the harness's token counting, which chat APIs lack. */ | |
| 365 | export function estimateTokens(request: AnthropicRequest): number { | |
| 366 | return Math.ceil(JSON.stringify({ system: request.system, messages: request.messages, tools: request.tools }).length / 4); | |
| 367 | } |