| 1 | /** |
| 2 | * Speaking Anthropic's Messages API on behalf of a caller that speaks |
| 3 | * OpenAI's Chat Completions API: the AI Gateway's `/openai/v1` format, sent |
| 4 | * to a Claude model. |
| 5 | * |
| 6 | * `openai.ts` goes the other way, for a caller that speaks Anthropic's API |
| 7 | * and a provider that speaks OpenAI's. Between them, any model is |
| 8 | * reachable from either format. |
| 9 | * |
| 10 | * - A request becomes a message: system and developer messages become the |
| 11 | * system prompt, tool messages become tool results, functions become |
| 12 | * tools, `reasoning_effort` becomes effort, `response_format` a JSON |
| 13 | * schema (or an instruction, for `json_object`). |
| 14 | * - The answer, streamed or not, becomes a chat completion, tool calls |
| 15 | * included, with Claude's thinking as `reasoning_content`. |
| 16 | * - Claude's thinking blocks must come back with the tool calls they led |
| 17 | * to, and a chat client keeps nothing but the call's id, so they ride in |
| 18 | * the first call's id, as `openai.ts` carries a provider's data. |
| 19 | */ |
| 20 | |
| 21 | import { carryId, uncarryId } from "./openai.ts"; |
| 22 | |
| 23 | type Json = Record<string, unknown>; |
| 24 | |
| 25 | /** A request this translation cannot say in Anthropic's terms, and why. */ |
| 26 | export class Untranslatable extends Error {} |
| 27 | |
| 28 | /** The most output tokens asked for when a request names none: Anthropic requires a number. */ |
| 29 | export const DEFAULT_MAX_TOKENS = 8192; |
| 30 | |
| 31 | /** OpenAI's `reasoning_effort` as Anthropic's effort. */ |
| 32 | export function anthropicEffort(effort: unknown): string | undefined { |
| 33 | if (effort === "none" || effort === "minimal" || effort === "low") return "low"; |
| 34 | if (effort === "medium" || effort === "high" || effort === "xhigh" || effort === "max") return effort; |
| 35 | return undefined; |
| 36 | } |
| 37 | |
| 38 | function textOf(content: unknown): string { |
| 39 | if (typeof content === "string") return content; |
| 40 | if (!Array.isArray(content)) return ""; |
| 41 | return content |
| 42 | .map((part) => ((part as Json)?.type === "text" ? String((part as Json).text ?? "") : "")) |
| 43 | .filter(Boolean) |
| 44 | .join("\n"); |
| 45 | } |
| 46 | |
| 47 | /** A data: URL's media type and base64 data, or null. */ |
| 48 | function dataUrl(url: string): { media_type: string; data: string } | null { |
| 49 | const match = /^data:([^;,]+);base64,(.*)$/s.exec(url); |
| 50 | return match ? { media_type: match[1]!, data: match[2]! } : null; |
| 51 | } |
| 52 | |
| 53 | /** A user message's parts as Anthropic content blocks. */ |
| 54 | function userBlocks(content: unknown): Json[] { |
| 55 | if (typeof content === "string") return content ? [{ type: "text", text: content }] : []; |
| 56 | if (!Array.isArray(content)) return []; |
| 57 | const blocks: Json[] = []; |
| 58 | for (const raw of content) { |
| 59 | const part = (raw ?? {}) as Json; |
| 60 | if (part.type === "text") { |
| 61 | if (part.text) blocks.push({ type: "text", text: String(part.text) }); |
| 62 | } else if (part.type === "image_url") { |
| 63 | const url = String(((part.image_url ?? {}) as Json).url ?? ""); |
| 64 | const inline = dataUrl(url); |
| 65 | blocks.push({ type: "image", source: inline ? { type: "base64", ...inline } : { type: "url", url } }); |
| 66 | } else if (part.type === "file") { |
| 67 | const file = (part.file ?? {}) as Json; |
| 68 | const inline = dataUrl(String(file.file_data ?? "")); |
| 69 | if (!inline) throw new Untranslatable("A file part needs its contents inline, as file_data: a data: URL."); |
| 70 | blocks.push({ type: "document", source: { type: "base64", ...inline } }); |
| 71 | } else { |
| 72 | throw new Untranslatable(`Message parts of type ${String(part.type)} are not supported with Claude models; send text, image_url or file.`); |
| 73 | } |
| 74 | } |
| 75 | return blocks; |
| 76 | } |
| 77 | |
| 78 | /** What a tool message said, as a tool result's content. */ |
| 79 | function toolContent(content: unknown): string | Json[] { |
| 80 | if (typeof content === "string") return content; |
| 81 | return userBlocks(content); |
| 82 | } |
| 83 | |
| 84 | /** An Anthropic Messages request saying what a chat completion request said. */ |
| 85 | export function chatToAnthropic(request: Json, model: string): Json { |
| 86 | if (typeof request.n === "number" && request.n > 1) { |
| 87 | throw new Untranslatable("Claude models answer once per request: leave out n, or set it to 1."); |
| 88 | } |
| 89 | const system: string[] = []; |
| 90 | const messages: { role: "user" | "assistant"; content: Json[] }[] = []; |
| 91 | const add = (role: "user" | "assistant", blocks: Json[]) => { |
| 92 | if (blocks.length === 0) return; |
| 93 | const last = messages[messages.length - 1]; |
| 94 | // Tool results and what follows them make one user turn. |
| 95 | if (last && last.role === role) last.content.push(...blocks); |
| 96 | else messages.push({ role, content: blocks }); |
| 97 | }; |
| 98 | |
| 99 | for (const raw of (request.messages as unknown[] | undefined) ?? []) { |
| 100 | const message = (raw ?? {}) as Json; |
| 101 | switch (message.role) { |
| 102 | case "system": |
| 103 | case "developer": { |
| 104 | const said = textOf(message.content); |
| 105 | if (said) system.push(said); |
| 106 | break; |
| 107 | } |
| 108 | case "user": |
| 109 | add("user", userBlocks(message.content)); |
| 110 | break; |
| 111 | case "assistant": { |
| 112 | const blocks: Json[] = []; |
| 113 | const calls = (message.tool_calls as Json[] | undefined) ?? []; |
| 114 | // The thinking that led to these calls, carried in the first call's id. |
| 115 | const carried = calls.length ? uncarryId(String(calls[0]!.id ?? "")).extra : undefined; |
| 116 | const thinking = (carried as { thinking?: unknown } | undefined)?.thinking; |
| 117 | if (Array.isArray(thinking)) blocks.push(...(thinking as Json[])); |
| 118 | const said = textOf(message.content); |
| 119 | if (said) blocks.push({ type: "text", text: said }); |
| 120 | for (const call of calls) { |
| 121 | const fn = (call.function ?? {}) as Json; |
| 122 | let input: unknown = {}; |
| 123 | try { |
| 124 | input = fn.arguments ? JSON.parse(String(fn.arguments)) : {}; |
| 125 | } catch { |
| 126 | input = {}; |
| 127 | } |
| 128 | blocks.push({ type: "tool_use", id: uncarryId(String(call.id ?? "")).id, name: String(fn.name ?? ""), input }); |
| 129 | } |
| 130 | add("assistant", blocks); |
| 131 | break; |
| 132 | } |
| 133 | case "tool": |
| 134 | add("user", [ |
| 135 | { |
| 136 | type: "tool_result", |
| 137 | tool_use_id: uncarryId(String(message.tool_call_id ?? "")).id, |
| 138 | content: toolContent(message.content), |
| 139 | }, |
| 140 | ]); |
| 141 | break; |
| 142 | default: |
| 143 | throw new Untranslatable(`Messages with the role ${String(message.role)} are not supported with Claude models.`); |
| 144 | } |
| 145 | } |
| 146 | |
| 147 | const body: Json = { |
| 148 | model, |
| 149 | messages, |
| 150 | max_tokens: Number(request.max_completion_tokens ?? request.max_tokens ?? DEFAULT_MAX_TOKENS) || DEFAULT_MAX_TOKENS, |
| 151 | }; |
| 152 | if (request.stream === true) body.stream = true; |
| 153 | if (typeof request.temperature === "number") body.temperature = request.temperature; |
| 154 | if (typeof request.top_p === "number") body.top_p = request.top_p; |
| 155 | const stop = typeof request.stop === "string" ? [request.stop] : Array.isArray(request.stop) ? request.stop : []; |
| 156 | if (stop.length) body.stop_sequences = stop.map(String); |
| 157 | if (typeof request.user === "string" && request.user) body.metadata = { user_id: request.user.slice(0, 256) }; |
| 158 | |
| 159 | const output: Json = {}; |
| 160 | const effort = anthropicEffort(request.reasoning_effort); |
| 161 | if (effort) output.effort = effort; |
| 162 | const format = (request.response_format ?? {}) as Json; |
| 163 | if (format.type === "json_schema") { |
| 164 | const schema = ((format.json_schema ?? {}) as Json).schema; |
| 165 | if (schema) output.format = { type: "json_schema", schema }; |
| 166 | } else if (format.type === "json_object") { |
| 167 | system.push("Answer with a single JSON object and nothing else."); |
| 168 | } |
| 169 | if (Object.keys(output).length) body.output_config = output; |
| 170 | // Anthropic's own thinking settings, for a caller that knows them. |
| 171 | if (request.thinking && typeof request.thinking === "object") body.thinking = request.thinking; |
| 172 | if (system.length) body.system = system.join("\n\n"); |
| 173 | |
| 174 | const tools = (request.tools as Json[] | undefined) ?? []; |
| 175 | if (tools.length) { |
| 176 | body.tools = tools.map((tool) => { |
| 177 | if (tool.type !== "function") { |
| 178 | throw new Untranslatable(`Tools of type ${String(tool.type)} are not supported with Claude models in this format; send functions.`); |
| 179 | } |
| 180 | const fn = (tool.function ?? {}) as Json; |
| 181 | return { |
| 182 | name: String(fn.name ?? ""), |
| 183 | ...(fn.description ? { description: String(fn.description) } : {}), |
| 184 | input_schema: fn.parameters ?? { type: "object", properties: {} }, |
| 185 | ...(fn.strict === true ? { strict: true } : {}), |
| 186 | }; |
| 187 | }); |
| 188 | } |
| 189 | const choice = request.tool_choice; |
| 190 | const parallel = request.parallel_tool_calls === false ? { disable_parallel_tool_use: true } : {}; |
| 191 | if (choice === "none") body.tool_choice = { type: "none" }; |
| 192 | else if (choice === "required") body.tool_choice = { type: "any", ...parallel }; |
| 193 | else if (choice && typeof choice === "object") { |
| 194 | const name = ((choice as Json).function as Json | undefined)?.name; |
| 195 | if (name) body.tool_choice = { type: "tool", name: String(name), ...parallel }; |
| 196 | } else if (tools.length && request.parallel_tool_calls === false) body.tool_choice = { type: "auto", ...parallel }; |
| 197 | return body; |
| 198 | } |
| 199 | |
| 200 | const FINISH: Record<string, string> = { |
| 201 | end_turn: "stop", |
| 202 | stop_sequence: "stop", |
| 203 | pause_turn: "stop", |
| 204 | max_tokens: "length", |
| 205 | tool_use: "tool_calls", |
| 206 | refusal: "content_filter", |
| 207 | }; |
| 208 | |
| 209 | /** A finish reason for an Anthropic stop reason. */ |
| 210 | export function finishReason(stop: unknown): string { |
| 211 | return FINISH[String(stop)] ?? "stop"; |
| 212 | } |
| 213 | |
| 214 | /** Anthropic's usage, as a chat completion reports it: cached tokens are part of the prompt. */ |
| 215 | export function chatUsage(usage: Json | undefined | null): Json { |
| 216 | const n = (value: unknown) => (typeof value === "number" && Number.isFinite(value) && value > 0 ? value : 0); |
| 217 | const cached = n(usage?.cache_read_input_tokens); |
| 218 | const prompt = n(usage?.input_tokens) + cached + n(usage?.cache_creation_input_tokens); |
| 219 | const completion = n(usage?.output_tokens); |
| 220 | return { |
| 221 | prompt_tokens: prompt, |
| 222 | completion_tokens: completion, |
| 223 | total_tokens: prompt + completion, |
| 224 | prompt_tokens_details: { cached_tokens: cached }, |
| 225 | }; |
| 226 | } |
| 227 | |
| 228 | /** The thinking blocks among an answer's content, to carry with its first tool call. */ |
| 229 | function thinkingOf(content: Json[]): Json[] { |
| 230 | return content.filter((block) => block.type === "thinking" || block.type === "redacted_thinking"); |
| 231 | } |
| 232 | |
| 233 | /** A chat completion saying what an Anthropic message said. */ |
| 234 | export function anthropicToChat(message: Json, model: string): Json { |
| 235 | const content = (message.content as Json[] | undefined) ?? []; |
| 236 | const text = content |
| 237 | .filter((block) => block.type === "text") |
| 238 | .map((block) => String(block.text ?? "")) |
| 239 | .join(""); |
| 240 | const reasoning = content |
| 241 | .filter((block) => block.type === "thinking" && block.thinking) |
| 242 | .map((block) => String(block.thinking)) |
| 243 | .join("\n"); |
| 244 | const thinking = thinkingOf(content); |
| 245 | const calls = content |
| 246 | .filter((block) => block.type === "tool_use") |
| 247 | .map((block, index) => ({ |
| 248 | id: index === 0 && thinking.length ? carryId(String(block.id), { thinking }) : String(block.id), |
| 249 | type: "function", |
| 250 | function: { name: String(block.name ?? ""), arguments: JSON.stringify(block.input ?? {}) }, |
| 251 | })); |
| 252 | return { |
| 253 | id: `chatcmpl-${String(message.id ?? crypto.randomUUID()).replace(/^msg_/, "")}`, |
| 254 | object: "chat.completion", |
| 255 | created: Math.floor(Date.now() / 1000), |
| 256 | model, |
| 257 | choices: [ |
| 258 | { |
| 259 | index: 0, |
| 260 | message: { |
| 261 | role: "assistant", |
| 262 | content: text || (calls.length ? null : ""), |
| 263 | ...(reasoning ? { reasoning_content: reasoning } : {}), |
| 264 | ...(calls.length ? { tool_calls: calls } : {}), |
| 265 | refusal: null, |
| 266 | }, |
| 267 | logprobs: null, |
| 268 | finish_reason: finishReason(message.stop_reason), |
| 269 | }, |
| 270 | ], |
| 271 | usage: chatUsage(message.usage as Json | undefined), |
| 272 | }; |
| 273 | } |
| 274 | |
| 275 | /** |
| 276 | * Turns Anthropic's streamed events into a chat completion's chunks: the |
| 277 | * role first, then text, reasoning and tool calls as they arrive, then why |
| 278 | * it stopped, the usage when the caller asked for it, and `[DONE]`. |
| 279 | */ |
| 280 | export class ChatStreamTranslator { |
| 281 | private readonly model: string; |
| 282 | private readonly includeUsage: boolean; |
| 283 | private id = `chatcmpl-${crypto.randomUUID().replace(/-/g, "")}`; |
| 284 | private readonly created = Math.floor(Date.now() / 1000); |
| 285 | private buffer = ""; |
| 286 | private started = false; |
| 287 | private finished = false; |
| 288 | private usage: Json = {}; |
| 289 | private stop = "end_turn"; |
| 290 | /** Content blocks by their index: what kind each is, and its tool call's slot. */ |
| 291 | private blocks = new Map<number, { kind: string; slot?: number; block?: Json }>(); |
| 292 | private calls = 0; |
| 293 | /** Thinking blocks so far, with their signatures, to carry with the first call. */ |
| 294 | private thinking: Json[] = []; |
| 295 | |
| 296 | constructor(model: string, includeUsage: boolean) { |
| 297 | this.model = model; |
| 298 | this.includeUsage = includeUsage; |
| 299 | } |
| 300 | |
| 301 | private chunk(delta: Json, finish: string | null = null): string { |
| 302 | const data = { |
| 303 | id: this.id, |
| 304 | object: "chat.completion.chunk", |
| 305 | created: this.created, |
| 306 | model: this.model, |
| 307 | choices: [{ index: 0, delta, logprobs: null, finish_reason: finish }], |
| 308 | }; |
| 309 | return `data: ${JSON.stringify(data)}\n\n`; |
| 310 | } |
| 311 | |
| 312 | private start(): string { |
| 313 | if (this.started) return ""; |
| 314 | this.started = true; |
| 315 | return this.chunk({ role: "assistant", content: "" }); |
| 316 | } |
| 317 | |
| 318 | private event(data: Json): string { |
| 319 | switch (data.type) { |
| 320 | case "message_start": { |
| 321 | const message = (data.message ?? {}) as Json; |
| 322 | if (typeof message.id === "string") this.id = `chatcmpl-${message.id.replace(/^msg_/, "")}`; |
| 323 | this.usage = { ...((message.usage as Json | undefined) ?? {}) }; |
| 324 | return this.start(); |
| 325 | } |
| 326 | case "content_block_start": { |
| 327 | const index = Number(data.index ?? 0); |
| 328 | const block = (data.content_block ?? {}) as Json; |
| 329 | if (block.type === "tool_use") { |
| 330 | const slot = this.calls++; |
| 331 | this.blocks.set(index, { kind: "tool_use", slot }); |
| 332 | const id = slot === 0 && this.thinking.length ? carryId(String(block.id), { thinking: this.thinking }) : String(block.id); |
| 333 | return ( |
| 334 | this.start() + |
| 335 | this.chunk({ tool_calls: [{ index: slot, id, type: "function", function: { name: String(block.name ?? ""), arguments: "" } }] }) |
| 336 | ); |
| 337 | } |
| 338 | if (block.type === "thinking" || block.type === "redacted_thinking") { |
| 339 | const kept: Json = { ...block }; |
| 340 | this.blocks.set(index, { kind: String(block.type), block: kept }); |
| 341 | this.thinking.push(kept); |
| 342 | return this.start(); |
| 343 | } |
| 344 | this.blocks.set(index, { kind: String(block.type) }); |
| 345 | return this.start() + (block.type === "text" && block.text ? this.chunk({ content: String(block.text) }) : ""); |
| 346 | } |
| 347 | case "content_block_delta": { |
| 348 | const at = this.blocks.get(Number(data.index ?? 0)); |
| 349 | const delta = (data.delta ?? {}) as Json; |
| 350 | if (delta.type === "text_delta") return this.chunk({ content: String(delta.text ?? "") }); |
| 351 | if (delta.type === "input_json_delta" && at?.slot != null) { |
| 352 | return this.chunk({ tool_calls: [{ index: at.slot, function: { arguments: String(delta.partial_json ?? "") } }] }); |
| 353 | } |
| 354 | if (delta.type === "thinking_delta" && at?.block) { |
| 355 | at.block.thinking = String(at.block.thinking ?? "") + String(delta.thinking ?? ""); |
| 356 | return delta.thinking ? this.chunk({ reasoning_content: String(delta.thinking) }) : ""; |
| 357 | } |
| 358 | if (delta.type === "signature_delta" && at?.block) { |
| 359 | at.block.signature = String(at.block.signature ?? "") + String(delta.signature ?? ""); |
| 360 | } |
| 361 | return ""; |
| 362 | } |
| 363 | case "message_delta": { |
| 364 | const delta = (data.delta ?? {}) as Json; |
| 365 | if (delta.stop_reason) this.stop = String(delta.stop_reason); |
| 366 | const usage = (data.usage ?? {}) as Json; |
| 367 | for (const [key, value] of Object.entries(usage)) if (typeof value === "number" && value > 0) this.usage[key] = value; |
| 368 | return ""; |
| 369 | } |
| 370 | case "message_stop": |
| 371 | return this.finish(); |
| 372 | case "error": { |
| 373 | const error = (data.error ?? {}) as Json; |
| 374 | this.finished = true; |
| 375 | return `data: ${JSON.stringify({ error: { message: String(error.message ?? "The model failed."), type: String(error.type ?? "api_error"), param: null, code: null } })}\n\ndata: [DONE]\n\n`; |
| 376 | } |
| 377 | default: |
| 378 | return ""; |
| 379 | } |
| 380 | } |
| 381 | |
| 382 | /** Takes raw bytes of Anthropic's stream; returns chunks to send. */ |
| 383 | push(piece: string): string { |
| 384 | this.buffer += piece; |
| 385 | let out = ""; |
| 386 | let at: number; |
| 387 | while ((at = this.buffer.indexOf("\n")) >= 0) { |
| 388 | const line = this.buffer.slice(0, at).trim(); |
| 389 | this.buffer = this.buffer.slice(at + 1); |
| 390 | if (!line.startsWith("data:")) continue; |
| 391 | try { |
| 392 | out += this.event(JSON.parse(line.slice(5).trim()) as Json); |
| 393 | } catch { |
| 394 | // A line that is not JSON carries nothing to translate. |
| 395 | } |
| 396 | } |
| 397 | return out; |
| 398 | } |
| 399 | |
| 400 | /** The closing chunks, once. */ |
| 401 | finish(): string { |
| 402 | if (this.finished) return ""; |
| 403 | this.finished = true; |
| 404 | let out = this.start() + this.chunk({}, finishReason(this.stop)); |
| 405 | if (this.includeUsage) { |
| 406 | const data = { id: this.id, object: "chat.completion.chunk", created: this.created, model: this.model, choices: [], usage: chatUsage(this.usage) }; |
| 407 | out += `data: ${JSON.stringify(data)}\n\n`; |
| 408 | } |
| 409 | return `${out}data: [DONE]\n\n`; |
| 410 | } |
| 411 | } |
| 412 | |
| 413 | /** OpenAI's error types, by the statuses the gateway answers with. */ |
| 414 | export function openaiErrorType(status: number): { type: string; code: string | null } { |
| 415 | if (status === 401) return { type: "authentication_error", code: "invalid_api_key" }; |
| 416 | if (status === 402) return { type: "insufficient_quota", code: "insufficient_quota" }; |
| 417 | if (status === 403) return { type: "permission_error", code: null }; |
| 418 | if (status === 404) return { type: "invalid_request_error", code: "model_not_found" }; |
| 419 | if (status === 429) return { type: "rate_limit_error", code: "rate_limit_exceeded" }; |
| 420 | if (status >= 500) return { type: "api_error", code: null }; |
| 421 | return { type: "invalid_request_error", code: null }; |
| 422 | } |
| 423 | |
| 424 | /** An error in the shape OpenAI's API and SDKs use. */ |
| 425 | export function openaiError(status: number, message: string, code?: string | null): Response { |
| 426 | const kind = openaiErrorType(status); |
| 427 | return Response.json( |
| 428 | { error: { message, type: kind.type, param: null, code: code === undefined ? kind.code : code } }, |
| 429 | { status }, |
| 430 | ); |
| 431 | } |