@velum-labs/routekit-gateway 0.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (99) hide show
  1. package/LICENSE +201 -0
  2. package/README.md +28 -0
  3. package/dist/acp-agent.d.ts +38 -0
  4. package/dist/acp-agent.js +142 -0
  5. package/dist/acp-registry.d.ts +36 -0
  6. package/dist/acp-registry.js +85 -0
  7. package/dist/adapters/anthropic.d.ts +131 -0
  8. package/dist/adapters/anthropic.js +1195 -0
  9. package/dist/adapters/chat.d.ts +14 -0
  10. package/dist/adapters/chat.js +34 -0
  11. package/dist/adapters/cursor.d.ts +34 -0
  12. package/dist/adapters/cursor.js +305 -0
  13. package/dist/adapters/dropped.d.ts +10 -0
  14. package/dist/adapters/dropped.js +24 -0
  15. package/dist/adapters/openai-chat-wire.d.ts +93 -0
  16. package/dist/adapters/openai-chat-wire.js +143 -0
  17. package/dist/adapters/responses-stream.d.ts +7 -0
  18. package/dist/adapters/responses-stream.js +597 -0
  19. package/dist/adapters/responses.d.ts +174 -0
  20. package/dist/adapters/responses.js +778 -0
  21. package/dist/adapters/server-tool-loop.d.ts +94 -0
  22. package/dist/adapters/server-tool-loop.js +477 -0
  23. package/dist/adapters/upstream-error.d.ts +14 -0
  24. package/dist/adapters/upstream-error.js +25 -0
  25. package/dist/adapters/validate.d.ts +27 -0
  26. package/dist/adapters/validate.js +180 -0
  27. package/dist/adapters/web-search.d.ts +46 -0
  28. package/dist/adapters/web-search.js +151 -0
  29. package/dist/auth.d.ts +10 -0
  30. package/dist/auth.js +28 -0
  31. package/dist/backend.d.ts +151 -0
  32. package/dist/backend.js +143 -0
  33. package/dist/capacity-pool.d.ts +31 -0
  34. package/dist/capacity-pool.js +99 -0
  35. package/dist/cost.d.ts +49 -0
  36. package/dist/cost.js +112 -0
  37. package/dist/endpoint-health.d.ts +54 -0
  38. package/dist/endpoint-health.js +123 -0
  39. package/dist/index.d.ts +40 -0
  40. package/dist/index.js +23 -0
  41. package/dist/provenance.d.ts +31 -0
  42. package/dist/provenance.js +191 -0
  43. package/dist/provider-backends.d.ts +40 -0
  44. package/dist/provider-backends.js +1050 -0
  45. package/dist/provider-source.d.ts +40 -0
  46. package/dist/provider-source.js +293 -0
  47. package/dist/router.d.ts +168 -0
  48. package/dist/router.js +474 -0
  49. package/dist/server.d.ts +67 -0
  50. package/dist/server.js +930 -0
  51. package/dist/sse/chat-assembler.d.ts +45 -0
  52. package/dist/sse/chat-assembler.js +190 -0
  53. package/dist/sse/parse.d.ts +50 -0
  54. package/dist/sse/parse.js +149 -0
  55. package/dist/sse-wire.d.ts +10 -0
  56. package/dist/sse-wire.js +31 -0
  57. package/dist/switching-proxy.d.ts +15 -0
  58. package/dist/switching-proxy.js +232 -0
  59. package/dist/test/acp-agent.test.d.ts +1 -0
  60. package/dist/test/acp-agent.test.js +66 -0
  61. package/dist/test/acp-registry.test.d.ts +1 -0
  62. package/dist/test/acp-registry.test.js +70 -0
  63. package/dist/test/anthropic.test.d.ts +1 -0
  64. package/dist/test/anthropic.test.js +793 -0
  65. package/dist/test/auth.test.d.ts +1 -0
  66. package/dist/test/auth.test.js +25 -0
  67. package/dist/test/boundary.test.d.ts +1 -0
  68. package/dist/test/boundary.test.js +32 -0
  69. package/dist/test/chat.test.d.ts +1 -0
  70. package/dist/test/chat.test.js +418 -0
  71. package/dist/test/cost.test.d.ts +1 -0
  72. package/dist/test/cost.test.js +60 -0
  73. package/dist/test/cursor.test.d.ts +1 -0
  74. package/dist/test/cursor.test.js +100 -0
  75. package/dist/test/drain.test.d.ts +1 -0
  76. package/dist/test/drain.test.js +116 -0
  77. package/dist/test/dropped.test.d.ts +1 -0
  78. package/dist/test/dropped.test.js +80 -0
  79. package/dist/test/endpoint-health.test.d.ts +1 -0
  80. package/dist/test/endpoint-health.test.js +73 -0
  81. package/dist/test/provenance.test.d.ts +1 -0
  82. package/dist/test/provenance.test.js +176 -0
  83. package/dist/test/provider-backends.test.d.ts +1 -0
  84. package/dist/test/provider-backends.test.js +699 -0
  85. package/dist/test/responses.test.d.ts +1 -0
  86. package/dist/test/responses.test.js +813 -0
  87. package/dist/test/routed-backend.test.d.ts +1 -0
  88. package/dist/test/routed-backend.test.js +39 -0
  89. package/dist/test/router.test.d.ts +1 -0
  90. package/dist/test/router.test.js +297 -0
  91. package/dist/test/server-resilience.test.d.ts +1 -0
  92. package/dist/test/server-resilience.test.js +169 -0
  93. package/dist/test/sse-codec.test.d.ts +1 -0
  94. package/dist/test/sse-codec.test.js +186 -0
  95. package/dist/test/web-search-loop.test.d.ts +1 -0
  96. package/dist/test/web-search-loop.test.js +469 -0
  97. package/dist/test/wire-validation.test.d.ts +1 -0
  98. package/dist/test/wire-validation.test.js +140 -0
  99. package/package.json +48 -0
@@ -0,0 +1,1195 @@
1
+ /**
2
+ * Anthropic Messages adapter. Claude Code speaks the Anthropic Messages API to
3
+ * whatever `ANTHROPIC_BASE_URL` points at, so to back it with a local model we
4
+ * translate `/v1/messages` (and `/v1/messages/count_tokens`, and the
5
+ * `/v1/models` discovery probe) to and from the gateway's OpenAI Chat
6
+ * Completions core. The pure translation functions are exported for testing;
7
+ * the request handler wires them to a `Backend` and returns a `Response` the
8
+ * server pipes straight to the client (JSON or SSE).
9
+ */
10
+ import { estimateTokens, randomId } from "@velum-labs/routekit-runtime";
11
+ import { SseDecoder, SseParseError } from "../sse/parse.js";
12
+ import { ANTHROPIC_MESSAGE_CONTENT, ANTHROPIC_REQUEST_METADATA, attachReasoningSelection, attachReasoningSelectionError, anthropicReasoningDetailsOf } from "./openai-chat-wire.js";
13
+ import { droppedField } from "./dropped.js";
14
+ import { unwrapUpstreamError } from "./upstream-error.js";
15
+ import { composeServerToolStream, runBufferedServerToolLoop, serverToolMarkerOf } from "./server-tool-loop.js";
16
+ import { resolveWebSearchExecutor } from "./web-search.js";
17
+ const ENCODER = new TextEncoder();
18
+ /**
19
+ * Whether an Anthropic tool is *server-executed* (run by Anthropic's backend,
20
+ * e.g. `web_search_20250305` / `code_execution_*`). Nothing behind this gateway
21
+ * can execute those, so advertising them to the upstream model would only produce
22
+ * calls nobody answers. Everything else — plain client tools (no `type` /
23
+ * `custom`) and Anthropic-defined client tools (`bash_*`, `text_editor_*`,
24
+ * `computer_*`), all of which the caller executes via ordinary `tool_use`
25
+ * blocks — is projected through.
26
+ */
27
+ function isAnthropicServerTool(tool) {
28
+ const type = tool.type ?? "";
29
+ return type.startsWith("web_search") || type.startsWith("code_execution");
30
+ }
31
+ /** A server web search tool declaration (`web_search_20250305` et al.) — the
32
+ * one server tool the gateway can honor via its own web-search executor. */
33
+ function isAnthropicWebSearchTool(tool) {
34
+ return (tool.type ?? "").startsWith("web_search");
35
+ }
36
+ /** The name the gateway-executed web search tool is projected under chat-side. */
37
+ const WEB_SEARCH_TOOL_NAME = "web_search";
38
+ const WEB_SEARCH_TOOL_DESCRIPTION = "Search the web for current, factual information. The search runs server-side and " +
39
+ "returns result text with source URLs. Use it when the answer depends on information " +
40
+ "that may have changed since your training data.";
41
+ const WEB_SEARCH_TOOL_PARAMETERS = {
42
+ type: "object",
43
+ properties: {
44
+ query: { type: "string", description: "The web search query." }
45
+ },
46
+ required: ["query"],
47
+ additionalProperties: false
48
+ };
49
+ /** Render an echoed `web_search_tool_result`'s content as a chat tool message.
50
+ * Bulky opaque fields (`encrypted_content`) are stripped; the upstream model
51
+ * only needs the urls/titles to remember what the search found. */
52
+ function webSearchResultText(content) {
53
+ if (!Array.isArray(content))
54
+ return JSON.stringify(content ?? null);
55
+ const results = content.map((entry) => {
56
+ if (entry === null || typeof entry !== "object")
57
+ return entry;
58
+ const { encrypted_content: _encrypted, ...rest } = entry;
59
+ return rest;
60
+ });
61
+ return JSON.stringify(results);
62
+ }
63
+ // ---- request translation ----
64
+ function systemText(system) {
65
+ if (system == null)
66
+ return "";
67
+ if (typeof system === "string")
68
+ return system;
69
+ return system
70
+ .map((block) => block !== null && typeof block === "object" && typeof block.text === "string"
71
+ ? block.text
72
+ : "")
73
+ .join("\n");
74
+ }
75
+ function blockText(content) {
76
+ if (content == null)
77
+ return "";
78
+ if (typeof content === "string")
79
+ return content;
80
+ return content
81
+ .map((block) => block !== null && typeof block === "object" && block.type === "text"
82
+ ? block.text
83
+ : "")
84
+ .join("");
85
+ }
86
+ function mapToolChoice(choice) {
87
+ switch (choice.type) {
88
+ case "auto":
89
+ return "auto";
90
+ case "any":
91
+ return "required";
92
+ case "none":
93
+ return "none";
94
+ case "tool":
95
+ return { type: "function", function: { name: choice.name ?? "" } };
96
+ default: {
97
+ const unreachable = choice.type;
98
+ return unreachable;
99
+ }
100
+ }
101
+ }
102
+ function mapThinking(thinking, outputConfig) {
103
+ if (thinking.type === "disabled")
104
+ return undefined;
105
+ const effort = outputConfig?.effort;
106
+ return typeof effort === "string" && effort.length > 0 ? effort : undefined;
107
+ }
108
+ function thinkingValidationError(body) {
109
+ const thinking = body.thinking;
110
+ if (thinking == null || thinking.type !== "enabled")
111
+ return undefined;
112
+ const budget = thinking.budget_tokens;
113
+ if (!Number.isInteger(budget) || budget < 1_024) {
114
+ return "thinking.budget_tokens must be an integer greater than or equal to 1024";
115
+ }
116
+ if (typeof body.max_tokens === "number" && budget >= body.max_tokens) {
117
+ return `thinking.budget_tokens must be less than max_tokens (${body.max_tokens})`;
118
+ }
119
+ return undefined;
120
+ }
121
+ function toolResultContent(result) {
122
+ const text = blockText(result.content);
123
+ return result.is_error === true ? `[tool_error]\n${text}` : text;
124
+ }
125
+ function attachAnthropicContent(message, content) {
126
+ Object.defineProperty(message, ANTHROPIC_MESSAGE_CONTENT, {
127
+ value: [...content],
128
+ enumerable: true
129
+ });
130
+ }
131
+ /**
132
+ * Translate an Anthropic Messages request to an OpenAI Chat Completions body.
133
+ * The upstream model is always the backend's own model (Claude Code sends a
134
+ * `claude-*` id the local server would not recognise); the requested id is
135
+ * only echoed back in the response.
136
+ */
137
+ export function anthropicToChat(body, backendModel, options = {}) {
138
+ const messages = [];
139
+ const system = systemText(body.system);
140
+ if (system.length > 0)
141
+ messages.push({ role: "system", content: system });
142
+ for (const message of body.messages) {
143
+ if (typeof message.content === "string") {
144
+ messages.push({ role: message.role, content: message.content });
145
+ continue;
146
+ }
147
+ const textParts = [];
148
+ const imageParts = [];
149
+ const toolCalls = [];
150
+ const toolResults = [];
151
+ const nativeContent = [];
152
+ let hasReplayableThinking = false;
153
+ // Echoed gateway-executed (or genuinely provider-executed, in a resumed
154
+ // session) web searches: `server_tool_use` + `web_search_tool_result`
155
+ // blocks ride in the assistant message and round-trip losslessly.
156
+ const serverToolUses = [];
157
+ const serverToolResults = [];
158
+ for (const block of message.content) {
159
+ switch (block.type) {
160
+ case "text":
161
+ textParts.push(block.text);
162
+ nativeContent.push({ type: "text", text: block.text });
163
+ break;
164
+ case "image": {
165
+ const source = block.source;
166
+ imageParts.push({
167
+ type: "image_url",
168
+ image_url: { url: `data:${source.media_type};base64,${source.data}` }
169
+ });
170
+ break;
171
+ }
172
+ case "tool_use": {
173
+ const tool = block;
174
+ toolCalls.push({
175
+ id: tool.id,
176
+ type: "function",
177
+ function: { name: tool.name, arguments: JSON.stringify(tool.input ?? {}) }
178
+ });
179
+ nativeContent.push({
180
+ type: "tool_use",
181
+ id: tool.id,
182
+ name: tool.name,
183
+ input: tool.input ?? {}
184
+ });
185
+ break;
186
+ }
187
+ case "tool_result": {
188
+ const result = block;
189
+ toolResults.push({ id: result.tool_use_id, content: toolResultContent(result) });
190
+ break;
191
+ }
192
+ case "server_tool_use": {
193
+ const tool = block;
194
+ serverToolUses.push({
195
+ id: tool.id,
196
+ type: "function",
197
+ function: { name: tool.name, arguments: JSON.stringify(tool.input ?? {}) }
198
+ });
199
+ break;
200
+ }
201
+ case "web_search_tool_result": {
202
+ const result = block;
203
+ serverToolResults.push({
204
+ id: result.tool_use_id ?? "",
205
+ content: webSearchResultText(result.content)
206
+ });
207
+ break;
208
+ }
209
+ case "thinking": {
210
+ const thinking = block;
211
+ // Only provider-issued non-empty signatures are safe to replay to
212
+ // Anthropic. Synthetic thinking emitted for another provider uses an
213
+ // empty signature and remains display-only.
214
+ if (typeof thinking.thinking === "string" &&
215
+ typeof thinking.signature === "string" &&
216
+ thinking.signature.length > 0) {
217
+ nativeContent.push({
218
+ type: "thinking",
219
+ thinking: thinking.thinking,
220
+ signature: thinking.signature
221
+ });
222
+ hasReplayableThinking = true;
223
+ }
224
+ else {
225
+ droppedField("anthropic", "thinking", "message");
226
+ }
227
+ break;
228
+ }
229
+ case "redacted_thinking": {
230
+ const redacted = block;
231
+ if (typeof redacted.data === "string" && redacted.data.length > 0) {
232
+ nativeContent.push({ type: "redacted_thinking", data: redacted.data });
233
+ hasReplayableThinking = true;
234
+ }
235
+ else {
236
+ droppedField("anthropic", "redacted_thinking", "message");
237
+ }
238
+ break;
239
+ }
240
+ default:
241
+ droppedField("anthropic", block.type, "message");
242
+ break;
243
+ }
244
+ }
245
+ if (message.role === "assistant") {
246
+ const text = textParts.join("");
247
+ if (imageParts.length > 0) {
248
+ droppedField("anthropic", "image", "assistant_message");
249
+ }
250
+ // Replay echoed server web searches as a chat tool exchange preceding
251
+ // the assistant's answer, so the upstream model remembers what was
252
+ // searched and found rather than blindly repeating it.
253
+ if (serverToolUses.length > 0) {
254
+ messages.push({ role: "assistant", content: null, tool_calls: serverToolUses });
255
+ for (const use of serverToolUses) {
256
+ const result = serverToolResults.find((entry) => entry.id === use.id);
257
+ messages.push({
258
+ role: "tool",
259
+ tool_call_id: use.id ?? "",
260
+ content: result?.content ?? "[web search results not retained]"
261
+ });
262
+ }
263
+ }
264
+ if (text.length > 0 || toolCalls.length > 0 || serverToolUses.length === 0) {
265
+ const assistant = { role: "assistant", content: text.length > 0 ? text : null };
266
+ if (toolCalls.length > 0)
267
+ assistant.tool_calls = toolCalls;
268
+ if (hasReplayableThinking)
269
+ attachAnthropicContent(assistant, nativeContent);
270
+ messages.push(assistant);
271
+ }
272
+ continue;
273
+ }
274
+ // user turn: tool results become standalone tool messages; remaining
275
+ // text/images become a user message.
276
+ for (const result of toolResults) {
277
+ messages.push({ role: "tool", tool_call_id: result.id, content: result.content });
278
+ }
279
+ const text = textParts.join("");
280
+ if (imageParts.length > 0) {
281
+ const parts = [];
282
+ if (text.length > 0)
283
+ parts.push({ type: "text", text });
284
+ parts.push(...imageParts);
285
+ messages.push({ role: "user", content: parts });
286
+ }
287
+ else if (text.length > 0 || toolResults.length === 0) {
288
+ messages.push({ role: "user", content: text });
289
+ }
290
+ }
291
+ const chat = {
292
+ model: backendModel ?? body.model ?? "",
293
+ messages,
294
+ stream: body.stream === true
295
+ };
296
+ // `max_completion_tokens`, not legacy `max_tokens`: OpenAI reasoning models
297
+ // reject the latter, and the other dialect adapters already emit the modern
298
+ // field (Claude Code always sends `max_tokens`, so this path is always hit).
299
+ if (typeof body.max_tokens === "number")
300
+ chat.max_completion_tokens = body.max_tokens;
301
+ if (typeof body.temperature === "number")
302
+ chat.temperature = body.temperature;
303
+ if (typeof body.top_p === "number")
304
+ chat.top_p = body.top_p;
305
+ if (typeof body.top_k === "number")
306
+ chat.top_k = body.top_k;
307
+ // Explicit nulls mean "unset" (see AnthropicRequest).
308
+ if (body.metadata != null)
309
+ droppedField("anthropic", "metadata");
310
+ if (body.output_config != null &&
311
+ Object.hasOwn(body.output_config, "effort") &&
312
+ body.output_config.effort !== null &&
313
+ (typeof body.output_config.effort !== "string" ||
314
+ body.output_config.effort.length === 0)) {
315
+ attachReasoningSelectionError(chat, "output_config.effort must be a non-empty string");
316
+ }
317
+ // `thinking: null` means "no extended thinking" — skip, never dereference
318
+ // (same failure class as the Responses adapter's `reasoning: null`).
319
+ if (body.thinking != null) {
320
+ const reasoningEffort = mapThinking(body.thinking, body.output_config);
321
+ if (body.thinking.type === "disabled") {
322
+ attachReasoningSelection(chat, { mode: "disabled" });
323
+ }
324
+ else if (reasoningEffort !== undefined) {
325
+ chat.reasoning_effort = reasoningEffort;
326
+ attachReasoningSelection(chat, {
327
+ mode: "effort",
328
+ effort: reasoningEffort
329
+ });
330
+ }
331
+ else if (body.thinking.type === "adaptive") {
332
+ attachReasoningSelection(chat, { mode: "adaptive" });
333
+ }
334
+ else {
335
+ attachReasoningSelection(chat, {
336
+ mode: "budget",
337
+ budgetTokens: body.thinking.budget_tokens
338
+ });
339
+ }
340
+ }
341
+ const metadata = {
342
+ ...(body.thinking != null ? { thinking: body.thinking } : {}),
343
+ ...(body.output_config !== undefined ? { output_config: body.output_config } : {})
344
+ };
345
+ if (Object.keys(metadata).length > 0) {
346
+ Object.defineProperty(chat, ANTHROPIC_REQUEST_METADATA, {
347
+ value: metadata,
348
+ enumerable: true
349
+ });
350
+ }
351
+ if (Array.isArray(body.stop_sequences) && body.stop_sequences.length > 0) {
352
+ chat.stop = body.stop_sequences;
353
+ }
354
+ if (Array.isArray(body.tools) && body.tools.length > 0) {
355
+ // Web search is honorable when an executor exists (the server-tool loop
356
+ // runs it); other server tools (`code_execution_*`) stay excluded.
357
+ const honorWebSearch = options.serverTools === true;
358
+ const excluded = body.tools.filter((tool) => isAnthropicServerTool(tool) && !(honorWebSearch && isAnthropicWebSearchTool(tool)));
359
+ if (excluded.length > 0) {
360
+ for (const tool of excluded) {
361
+ droppedField("anthropic", tool.name ?? tool.type ?? "server_tool", "tools");
362
+ }
363
+ if (process.env.ROUTEKIT_DEBUG) {
364
+ process.stderr.write(`[routekit-debug] anthropic: excluding ${excluded.length} server-executed tool(s) ` +
365
+ `from the request: ${excluded.map((tool) => tool.name).join(", ")}\n`);
366
+ }
367
+ }
368
+ const tools = body.tools
369
+ .filter((tool) => !isAnthropicServerTool(tool) && typeof tool.name === "string" && tool.name.length > 0)
370
+ .map((tool) => ({
371
+ type: "function",
372
+ function: {
373
+ name: tool.name,
374
+ ...(tool.description !== undefined ? { description: tool.description } : {}),
375
+ parameters: tool.input_schema ?? { type: "object", properties: {} }
376
+ }
377
+ }));
378
+ if (honorWebSearch &&
379
+ body.tools.some(isAnthropicWebSearchTool) &&
380
+ !tools.some((tool) => tool.function.name === WEB_SEARCH_TOOL_NAME)) {
381
+ tools.push({
382
+ type: "function",
383
+ function: {
384
+ name: WEB_SEARCH_TOOL_NAME,
385
+ description: WEB_SEARCH_TOOL_DESCRIPTION,
386
+ parameters: WEB_SEARCH_TOOL_PARAMETERS
387
+ }
388
+ });
389
+ }
390
+ if (tools.length > 0)
391
+ chat.tools = tools;
392
+ }
393
+ if (body.tool_choice != null) {
394
+ chat.tool_choice = mapToolChoice(body.tool_choice);
395
+ if (body.tool_choice.disable_parallel_tool_use === true)
396
+ chat.parallel_tool_calls = false;
397
+ }
398
+ if (body.stream === true)
399
+ chat.stream_options = { include_usage: true };
400
+ return chat;
401
+ }
402
+ // ---- response translation ----
403
+ export function mapStopReason(finishReason) {
404
+ switch (finishReason) {
405
+ case "length":
406
+ return "max_tokens";
407
+ case "tool_calls":
408
+ return "tool_use";
409
+ case "content_filter":
410
+ return "refusal";
411
+ case "stop":
412
+ case null:
413
+ case undefined:
414
+ return "end_turn";
415
+ default:
416
+ return "end_turn";
417
+ }
418
+ }
419
+ /** The native Anthropic blocks for one gateway-executed web search: the
420
+ * `server_tool_use` and its `web_search_tool_result`. Anthropic-executor
421
+ * results pass through verbatim; other executors build result blocks
422
+ * from their citations. */
423
+ function executedSearchBlocks(search) {
424
+ const resultContent = search.status !== "completed"
425
+ ? { type: "web_search_tool_result_error", error_code: "unavailable" }
426
+ : (search.outcome?.anthropicResultBlocks ??
427
+ (search.outcome?.citations ?? []).map((citation) => ({
428
+ type: "web_search_result",
429
+ url: citation.url,
430
+ ...(citation.title !== undefined ? { title: citation.title } : {})
431
+ })));
432
+ return [
433
+ { type: "server_tool_use", id: search.itemId, name: WEB_SEARCH_TOOL_NAME, input: { query: search.query } },
434
+ { type: "web_search_tool_result", tool_use_id: search.itemId, content: resultContent }
435
+ ];
436
+ }
437
+ export function chatToAnthropicMessage(openai, model, searches = [], events) {
438
+ const choice = openai.choices?.[0];
439
+ const message = choice?.message;
440
+ const content = [];
441
+ const nativeReasoning = anthropicReasoningDetailsOf(message?.reasoning_details, "message").sort((a, b) => a.index - b.index);
442
+ const appendNativeReasoning = (details) => {
443
+ for (const detail of details) {
444
+ if (detail.type === "thinking") {
445
+ content.push({
446
+ type: "thinking",
447
+ thinking: detail.thinking ?? "",
448
+ signature: detail.signature ?? ""
449
+ });
450
+ }
451
+ else {
452
+ content.push({ type: "redacted_thinking", data: detail.data });
453
+ }
454
+ }
455
+ };
456
+ // Gateway-executed steps precede the terminal model step. Preserve their
457
+ // signed/redacted reasoning in exact step order around search blocks.
458
+ if (events !== undefined) {
459
+ for (const event of events) {
460
+ if (event.kind === "reasoning") {
461
+ appendNativeReasoning(anthropicReasoningDetailsOf(event.details, "message"));
462
+ }
463
+ else {
464
+ content.push(...executedSearchBlocks(event.search));
465
+ }
466
+ }
467
+ }
468
+ else {
469
+ for (const search of searches)
470
+ content.push(...executedSearchBlocks(search));
471
+ }
472
+ appendNativeReasoning(nativeReasoning);
473
+ const rawReasoning = typeof message?.reasoning === "string" && message.reasoning.length > 0
474
+ ? message.reasoning
475
+ : "";
476
+ const narration = typeof message?.reasoning_content === "string" &&
477
+ message.reasoning_content.length > 0
478
+ ? message.reasoning_content.replace(/\*\*/g, "")
479
+ : "";
480
+ if (nativeReasoning.length === 0 && rawReasoning.length > 0) {
481
+ // Generic providers cannot produce an Anthropic-verifiable signature.
482
+ // The empty marker makes the block displayable; ingress deliberately
483
+ // refuses to replay it as native signed history.
484
+ content.push({
485
+ type: "thinking",
486
+ thinking: rawReasoning,
487
+ signature: ""
488
+ });
489
+ }
490
+ if (narration.length > 0) {
491
+ content.push({ type: "thinking", thinking: narration, signature: "" });
492
+ }
493
+ const text = typeof message?.content === "string" ? message.content : "";
494
+ if (text.length > 0)
495
+ content.push({ type: "text", text });
496
+ if (Array.isArray(message?.tool_calls)) {
497
+ for (const call of message.tool_calls) {
498
+ let input = {};
499
+ const args = call.function?.arguments;
500
+ if (typeof args === "string" && args.length > 0) {
501
+ try {
502
+ input = JSON.parse(args);
503
+ }
504
+ catch {
505
+ input = {};
506
+ }
507
+ }
508
+ content.push({
509
+ type: "tool_use",
510
+ id: call.id ?? `toolu_${randomId()}`,
511
+ name: call.function?.name ?? "",
512
+ input
513
+ });
514
+ }
515
+ }
516
+ if (content.length === 0)
517
+ content.push({ type: "text", text: "" });
518
+ const response = {
519
+ id: openai.id !== undefined ? `msg_${openai.id}` : `msg_${randomId()}`,
520
+ type: "message",
521
+ role: "assistant",
522
+ model,
523
+ content,
524
+ stop_reason: typeof choice?.anthropic_stop_reason === "string"
525
+ ? choice.anthropic_stop_reason
526
+ : mapStopReason(choice?.finish_reason),
527
+ stop_sequence: typeof choice?.anthropic_stop_sequence === "string"
528
+ ? choice.anthropic_stop_sequence
529
+ : null
530
+ };
531
+ if (openai.usage !== undefined) {
532
+ response.usage = {
533
+ ...(openai.usage.prompt_tokens !== undefined ? { input_tokens: openai.usage.prompt_tokens } : {}),
534
+ ...(openai.usage.completion_tokens !== undefined ? { output_tokens: openai.usage.completion_tokens } : {})
535
+ };
536
+ }
537
+ return response;
538
+ }
539
+ // ---- streaming translation (OpenAI chat SSE -> Anthropic Messages SSE) ----
540
+ function sse(type, data) {
541
+ return ENCODER.encode(`event: ${type}\ndata: ${JSON.stringify(data)}\n\n`);
542
+ }
543
+ export function openAiSseToAnthropic(upstream, model) {
544
+ const reader = upstream.getReader();
545
+ const sseDecoder = new SseDecoder();
546
+ // OpenAI tool-call fragments map onto Anthropic `tool_use` content blocks.
547
+ // Fragments are keyed by `index` when present, else by `id`; an id/index-less
548
+ // fragment (Anthropic/Responses translations omit `index`) appends to the last
549
+ // open call. Keying everything to index 0 used to merge parallel index-less
550
+ // calls into one block — the same bug the shared assembler now avoids.
551
+ const toolBlockByIndex = new Map();
552
+ const toolBlockById = new Map();
553
+ const toolBlocks = [];
554
+ let lastToolBlock;
555
+ const messageId = `msg_${randomId()}`;
556
+ const state = {
557
+ started: false,
558
+ textOpen: false,
559
+ textIndex: -1,
560
+ thinkingOpen: false,
561
+ thinkingIndex: -1,
562
+ thinkingSourceIndex: undefined,
563
+ pendingNarration: [],
564
+ outputStarted: false,
565
+ nextIndex: 0,
566
+ finished: false,
567
+ inputTokens: undefined,
568
+ outputTokens: undefined,
569
+ keepaliveTimer: undefined
570
+ };
571
+ const ensureStarted = (controller) => {
572
+ if (state.started)
573
+ return;
574
+ state.started = true;
575
+ controller.enqueue(sse("message_start", {
576
+ type: "message_start",
577
+ message: {
578
+ id: messageId,
579
+ type: "message",
580
+ role: "assistant",
581
+ model,
582
+ content: [],
583
+ stop_reason: null,
584
+ stop_sequence: null,
585
+ ...(state.inputTokens !== undefined ? { usage: { input_tokens: state.inputTokens } } : {})
586
+ }
587
+ }));
588
+ };
589
+ // Generic reasoning has no provider-verifiable signature. Native Anthropic
590
+ // metadata below carries its real block lifecycle and signature separately.
591
+ const ensureThinking = (controller) => {
592
+ ensureStarted(controller);
593
+ if (state.thinkingOpen || state.outputStarted)
594
+ return;
595
+ state.thinkingOpen = true;
596
+ state.thinkingSourceIndex = undefined;
597
+ state.thinkingIndex = state.nextIndex++;
598
+ controller.enqueue(sse("content_block_start", {
599
+ type: "content_block_start",
600
+ index: state.thinkingIndex,
601
+ content_block: { type: "thinking", thinking: "" }
602
+ }));
603
+ };
604
+ const closeThinking = (controller, sourceIndex) => {
605
+ if (!state.thinkingOpen)
606
+ return;
607
+ if (sourceIndex !== undefined &&
608
+ state.thinkingSourceIndex !== undefined &&
609
+ sourceIndex !== state.thinkingSourceIndex) {
610
+ return;
611
+ }
612
+ state.thinkingOpen = false;
613
+ controller.enqueue(sse("content_block_stop", { type: "content_block_stop", index: state.thinkingIndex }));
614
+ state.thinkingSourceIndex = undefined;
615
+ };
616
+ const emitNarration = (controller, text) => {
617
+ ensureThinking(controller);
618
+ if (!state.thinkingOpen || state.thinkingSourceIndex !== undefined)
619
+ return;
620
+ controller.enqueue(sse("content_block_delta", {
621
+ type: "content_block_delta",
622
+ index: state.thinkingIndex,
623
+ delta: {
624
+ type: "thinking_delta",
625
+ thinking: text.replace(/\*\*/g, "")
626
+ }
627
+ }));
628
+ };
629
+ const flushPendingNarration = (controller) => {
630
+ if (state.pendingNarration.length === 0)
631
+ return;
632
+ const pending = state.pendingNarration.join("");
633
+ state.pendingNarration = [];
634
+ emitNarration(controller, pending);
635
+ };
636
+ const ensureText = (controller) => {
637
+ ensureStarted(controller);
638
+ closeThinking(controller);
639
+ flushPendingNarration(controller);
640
+ closeThinking(controller);
641
+ state.outputStarted = true;
642
+ if (state.textOpen)
643
+ return;
644
+ state.textOpen = true;
645
+ state.textIndex = state.nextIndex++;
646
+ controller.enqueue(sse("content_block_start", {
647
+ type: "content_block_start",
648
+ index: state.textIndex,
649
+ content_block: { type: "text", text: "" }
650
+ }));
651
+ };
652
+ const closeOpenBlocks = (controller) => {
653
+ closeThinking(controller);
654
+ flushPendingNarration(controller);
655
+ closeThinking(controller);
656
+ if (state.textOpen) {
657
+ controller.enqueue(sse("content_block_stop", { type: "content_block_stop", index: state.textIndex }));
658
+ }
659
+ for (const index of toolBlocks) {
660
+ controller.enqueue(sse("content_block_stop", { type: "content_block_stop", index }));
661
+ }
662
+ };
663
+ const finalize = (controller, stopReason, stopSequence = null) => {
664
+ if (state.finished)
665
+ return;
666
+ state.finished = true;
667
+ if (state.keepaliveTimer !== undefined)
668
+ clearInterval(state.keepaliveTimer);
669
+ closeOpenBlocks(controller);
670
+ controller.enqueue(sse("message_delta", {
671
+ type: "message_delta",
672
+ delta: { stop_reason: stopReason, stop_sequence: stopSequence },
673
+ ...(state.inputTokens !== undefined || state.outputTokens !== undefined
674
+ ? {
675
+ usage: {
676
+ ...(state.inputTokens !== undefined ? { input_tokens: state.inputTokens } : {}),
677
+ ...(state.outputTokens !== undefined ? { output_tokens: state.outputTokens } : {})
678
+ }
679
+ }
680
+ : {})
681
+ }));
682
+ controller.enqueue(sse("message_stop", { type: "message_stop" }));
683
+ };
684
+ /**
685
+ * The upstream ended (reader closed or a `[DONE]` arrived) before any
686
+ * `finish_reason`. Truncation is an error, not a clean stop (WS5.2): emit an
687
+ * Anthropic `error` event rather than fabricating `stop_reason:"end_turn"`, so
688
+ * the caller sees a failed turn instead of silently accepting a partial answer.
689
+ */
690
+ const finalizeTruncated = (controller, detail) => {
691
+ if (state.finished)
692
+ return;
693
+ state.finished = true;
694
+ if (state.keepaliveTimer !== undefined)
695
+ clearInterval(state.keepaliveTimer);
696
+ closeOpenBlocks(controller);
697
+ controller.enqueue(sse("error", {
698
+ type: "error",
699
+ error: { type: "incomplete_stream", message: detail }
700
+ }));
701
+ };
702
+ const finalizeUpstreamError = (controller, error) => {
703
+ if (state.finished)
704
+ return;
705
+ state.finished = true;
706
+ if (state.keepaliveTimer !== undefined)
707
+ clearInterval(state.keepaliveTimer);
708
+ closeOpenBlocks(controller);
709
+ controller.enqueue(sse("error", {
710
+ type: "error",
711
+ error: unwrapUpstreamError(JSON.stringify({ error }))
712
+ }));
713
+ };
714
+ // The server-tool loop injects marker chunks around each gateway-executed
715
+ // web search; render them as native `server_tool_use` /
716
+ // `web_search_tool_result` blocks (each opened and closed immediately —
717
+ // their content is complete when the marker arrives).
718
+ const handleServerToolMarker = (controller, marker) => {
719
+ ensureStarted(controller);
720
+ closeThinking(controller);
721
+ flushPendingNarration(controller);
722
+ closeThinking(controller);
723
+ state.outputStarted = true;
724
+ if (marker.phase === "start") {
725
+ const index = state.nextIndex++;
726
+ controller.enqueue(sse("content_block_start", {
727
+ type: "content_block_start",
728
+ index,
729
+ content_block: { type: "server_tool_use", id: marker.item_id, name: "web_search", input: {} }
730
+ }));
731
+ controller.enqueue(sse("content_block_delta", {
732
+ type: "content_block_delta",
733
+ index,
734
+ delta: { type: "input_json_delta", partial_json: JSON.stringify({ query: marker.query }) }
735
+ }));
736
+ controller.enqueue(sse("content_block_stop", { type: "content_block_stop", index }));
737
+ return;
738
+ }
739
+ const index = state.nextIndex++;
740
+ const content = marker.status === "failed"
741
+ ? { type: "web_search_tool_result_error", error_code: "unavailable" }
742
+ : (marker.result_blocks ?? []);
743
+ controller.enqueue(sse("content_block_start", {
744
+ type: "content_block_start",
745
+ index,
746
+ content_block: { type: "web_search_tool_result", tool_use_id: marker.item_id, content }
747
+ }));
748
+ controller.enqueue(sse("content_block_stop", { type: "content_block_stop", index }));
749
+ // A completed server-side tool step is an internal model boundary, not the
750
+ // final answer. The continuation model may legitimately begin with another
751
+ // signed thinking/redacted block.
752
+ state.outputStarted = false;
753
+ };
754
+ const handleReasoningDetails = (controller, details) => {
755
+ let carriedText = false;
756
+ for (const detail of details) {
757
+ if (detail.type === "redacted_thinking") {
758
+ if (state.outputStarted)
759
+ continue;
760
+ ensureStarted(controller);
761
+ closeThinking(controller);
762
+ const index = state.nextIndex++;
763
+ controller.enqueue(sse("content_block_start", {
764
+ type: "content_block_start",
765
+ index,
766
+ content_block: { type: "redacted_thinking", data: detail.data }
767
+ }));
768
+ controller.enqueue(sse("content_block_stop", { type: "content_block_stop", index }));
769
+ continue;
770
+ }
771
+ if (detail.phase === "start") {
772
+ if (state.outputStarted)
773
+ continue;
774
+ ensureStarted(controller);
775
+ closeThinking(controller);
776
+ state.thinkingOpen = true;
777
+ state.thinkingSourceIndex = detail.index;
778
+ state.thinkingIndex = state.nextIndex++;
779
+ controller.enqueue(sse("content_block_start", {
780
+ type: "content_block_start",
781
+ index: state.thinkingIndex,
782
+ content_block: {
783
+ type: "thinking",
784
+ thinking: "",
785
+ signature: detail.signature ?? ""
786
+ }
787
+ }));
788
+ continue;
789
+ }
790
+ if (state.thinkingSourceIndex !== detail.index ||
791
+ !state.thinkingOpen ||
792
+ state.outputStarted) {
793
+ continue;
794
+ }
795
+ if (detail.phase === "delta" && typeof detail.thinking === "string") {
796
+ carriedText = true;
797
+ controller.enqueue(sse("content_block_delta", {
798
+ type: "content_block_delta",
799
+ index: state.thinkingIndex,
800
+ delta: { type: "thinking_delta", thinking: detail.thinking }
801
+ }));
802
+ }
803
+ else if (detail.phase === "signature" && typeof detail.signature === "string") {
804
+ controller.enqueue(sse("content_block_delta", {
805
+ type: "content_block_delta",
806
+ index: state.thinkingIndex,
807
+ delta: { type: "signature_delta", signature: detail.signature }
808
+ }));
809
+ }
810
+ else if (detail.phase === "stop") {
811
+ closeThinking(controller, detail.index);
812
+ }
813
+ }
814
+ return carriedText;
815
+ };
816
+ const process = (controller, chunk) => {
817
+ if (chunk.error !== undefined && chunk.error !== null) {
818
+ finalizeUpstreamError(controller, chunk.error);
819
+ return;
820
+ }
821
+ const choice = chunk.choices?.[0];
822
+ if (choice === undefined) {
823
+ if (chunk.usage?.prompt_tokens !== undefined)
824
+ state.inputTokens = chunk.usage.prompt_tokens;
825
+ if (chunk.usage?.completion_tokens !== undefined)
826
+ state.outputTokens = chunk.usage.completion_tokens;
827
+ return;
828
+ }
829
+ const delta = choice.delta ?? {};
830
+ const nativeDetails = anthropicReasoningDetailsOf(delta.reasoning_details, "stream");
831
+ const nativeCarriedText = nativeDetails.length > 0 &&
832
+ handleReasoningDetails(controller, nativeDetails);
833
+ if (state.pendingNarration.length > 0 &&
834
+ (!state.thinkingOpen || state.thinkingSourceIndex === undefined)) {
835
+ flushPendingNarration(controller);
836
+ }
837
+ if (typeof delta.reasoning_content === "string" &&
838
+ delta.reasoning_content.length > 0 &&
839
+ !state.outputStarted) {
840
+ if (state.thinkingOpen && state.thinkingSourceIndex !== undefined) {
841
+ // Never contaminate provider-signed thinking with gateway narration:
842
+ // the signature must continue to describe exactly the native text.
843
+ state.pendingNarration.push(delta.reasoning_content);
844
+ }
845
+ else {
846
+ emitNarration(controller, delta.reasoning_content);
847
+ }
848
+ }
849
+ if (!nativeCarriedText &&
850
+ typeof delta.reasoning === "string" &&
851
+ delta.reasoning.length > 0 &&
852
+ !state.outputStarted) {
853
+ // Raw model thinking tokens pass through verbatim: they are already
854
+ // plain text, and Anthropic thinking blocks stream token deltas natively.
855
+ ensureThinking(controller);
856
+ controller.enqueue(sse("content_block_delta", {
857
+ type: "content_block_delta",
858
+ index: state.thinkingIndex,
859
+ delta: { type: "thinking_delta", thinking: delta.reasoning }
860
+ }));
861
+ }
862
+ if (typeof delta.content === "string" && delta.content.length > 0) {
863
+ ensureText(controller);
864
+ controller.enqueue(sse("content_block_delta", {
865
+ type: "content_block_delta",
866
+ index: state.textIndex,
867
+ delta: { type: "text_delta", text: delta.content }
868
+ }));
869
+ }
870
+ if (Array.isArray(delta.tool_calls)) {
871
+ for (const call of delta.tool_calls) {
872
+ const indexKey = typeof call.index === "number" ? call.index : undefined;
873
+ const idKey = typeof call.id === "string" && call.id.length > 0 ? call.id : undefined;
874
+ let block = indexKey !== undefined
875
+ ? toolBlockByIndex.get(indexKey)
876
+ : idKey !== undefined
877
+ ? toolBlockById.get(idKey)
878
+ : lastToolBlock;
879
+ if (block === undefined) {
880
+ ensureStarted(controller);
881
+ closeThinking(controller);
882
+ flushPendingNarration(controller);
883
+ closeThinking(controller);
884
+ state.outputStarted = true;
885
+ block = state.nextIndex++;
886
+ toolBlocks.push(block);
887
+ controller.enqueue(sse("content_block_start", {
888
+ type: "content_block_start",
889
+ index: block,
890
+ content_block: {
891
+ type: "tool_use",
892
+ id: call.id ?? `toolu_${randomId()}`,
893
+ name: call.function?.name ?? "",
894
+ input: {}
895
+ }
896
+ }));
897
+ }
898
+ if (indexKey !== undefined && !toolBlockByIndex.has(indexKey))
899
+ toolBlockByIndex.set(indexKey, block);
900
+ if (idKey !== undefined && !toolBlockById.has(idKey))
901
+ toolBlockById.set(idKey, block);
902
+ lastToolBlock = block;
903
+ const args = call.function?.arguments;
904
+ if (typeof args === "string" && args.length > 0) {
905
+ controller.enqueue(sse("content_block_delta", {
906
+ type: "content_block_delta",
907
+ index: block,
908
+ delta: { type: "input_json_delta", partial_json: args }
909
+ }));
910
+ }
911
+ }
912
+ }
913
+ if (chunk.usage?.prompt_tokens !== undefined)
914
+ state.inputTokens = chunk.usage.prompt_tokens;
915
+ if (chunk.usage?.completion_tokens !== undefined)
916
+ state.outputTokens = chunk.usage.completion_tokens;
917
+ if (choice.finish_reason !== null && choice.finish_reason !== undefined) {
918
+ finalize(controller, typeof choice.anthropic_stop_reason === "string"
919
+ ? choice.anthropic_stop_reason
920
+ : mapStopReason(choice.finish_reason), typeof choice.anthropic_stop_sequence === "string"
921
+ ? choice.anthropic_stop_sequence
922
+ : null);
923
+ }
924
+ };
925
+ // Backpressure handshake: the pump awaits `resumePull` whenever the consumer's
926
+ // desired size drops to zero, and `pull` resolves it. This replaces the old
927
+ // "return when desiredSize changed" hack with an explicit pump that reads the
928
+ // upstream reader to completion while honoring backpressure.
929
+ let resumePull;
930
+ const awaitPull = () => new Promise((resolve) => {
931
+ resumePull = resolve;
932
+ });
933
+ const handleEvent = (controller, data) => {
934
+ if (data.length === 0)
935
+ return;
936
+ if (data === "[DONE]") {
937
+ // A `[DONE]` without a prior finish_reason is truncation, not a clean stop.
938
+ if (!state.finished)
939
+ finalizeTruncated(controller, "upstream sent [DONE] before a finish reason");
940
+ return;
941
+ }
942
+ let chunk;
943
+ try {
944
+ chunk = JSON.parse(data);
945
+ }
946
+ catch (error) {
947
+ // The live upstream stream is authoritative: a malformed payload is a
948
+ // stream error, never silently skipped (WS5). Surface it and stop.
949
+ const detail = error instanceof Error ? error.message : String(error);
950
+ throw new SseParseError(`malformed OpenAI SSE payload in Anthropic translation: ${detail}`, data.slice(0, 200));
951
+ }
952
+ const marker = serverToolMarkerOf(chunk);
953
+ if (marker !== undefined) {
954
+ handleServerToolMarker(controller, marker);
955
+ return;
956
+ }
957
+ process(controller, chunk);
958
+ };
959
+ const pump = async (controller) => {
960
+ try {
961
+ for (;;) {
962
+ if ((controller.desiredSize ?? 1) <= 0)
963
+ await awaitPull();
964
+ const { done, value } = await reader.read();
965
+ if (done) {
966
+ for (const event of sseDecoder.flush())
967
+ handleEvent(controller, event.data);
968
+ // Upstream closed with no finish_reason: incomplete, not `end_turn`.
969
+ if (!state.finished)
970
+ finalizeTruncated(controller, "upstream stream ended before a finish reason");
971
+ controller.close();
972
+ return;
973
+ }
974
+ if (value !== undefined) {
975
+ for (const event of sseDecoder.feed(value))
976
+ handleEvent(controller, event.data);
977
+ }
978
+ }
979
+ }
980
+ catch (error) {
981
+ if (state.keepaliveTimer !== undefined)
982
+ clearInterval(state.keepaliveTimer);
983
+ controller.error(error);
984
+ void reader.cancel(error).catch(() => undefined);
985
+ }
986
+ };
987
+ return new ReadableStream({
988
+ start(controller) {
989
+ // Start the message immediately and keep the connection alive with `ping`
990
+ // events while the upstream is still producing its first token. Claude
991
+ // Code times out if it sees nothing during a slow upstream phase (the
992
+ // chat-layer keepalive comments are dropped by this translator, so this
993
+ // ping is the single keepalive that reaches the client).
994
+ ensureStarted(controller);
995
+ state.keepaliveTimer = setInterval(() => {
996
+ if (state.finished)
997
+ return;
998
+ // Honor backpressure: skip the ping if the consumer's queue is full.
999
+ if ((controller.desiredSize ?? 1) <= 0)
1000
+ return;
1001
+ try {
1002
+ controller.enqueue(sse("ping", { type: "ping" }));
1003
+ }
1004
+ catch {
1005
+ // controller closed
1006
+ }
1007
+ }, 3000);
1008
+ void pump(controller);
1009
+ },
1010
+ pull() {
1011
+ resumePull?.();
1012
+ resumePull = undefined;
1013
+ },
1014
+ cancel(reason) {
1015
+ if (state.keepaliveTimer !== undefined)
1016
+ clearInterval(state.keepaliveTimer);
1017
+ resumePull?.();
1018
+ resumePull = undefined;
1019
+ return reader.cancel(reason);
1020
+ }
1021
+ });
1022
+ }
1023
+ // ---- token counting + discovery ----
1024
+ export function countTokensEstimate(body) {
1025
+ const parts = [systemText(body.system)];
1026
+ for (const message of body.messages)
1027
+ parts.push(blockText(message.content));
1028
+ return estimateTokens(...parts);
1029
+ }
1030
+ // ---- handlers (return a Response the server pipes) ----
1031
+ function jsonResponse(status, value) {
1032
+ return new Response(JSON.stringify(value), {
1033
+ status,
1034
+ headers: { "content-type": "application/json" }
1035
+ });
1036
+ }
1037
+ export async function handleAnthropicMessages(backend, body, modelCallId, signal, backendOptions = {}) {
1038
+ const invalidThinking = thinkingValidationError(body);
1039
+ if (invalidThinking !== undefined) {
1040
+ return jsonResponse(400, {
1041
+ type: "error",
1042
+ error: { type: "invalid_request_error", message: invalidThinking }
1043
+ });
1044
+ }
1045
+ const requestedModel = body.model ?? backend.defaultModel ?? "";
1046
+ const resolvedModel = backend.resolveModel?.(body.model);
1047
+ if (body.model !== undefined &&
1048
+ backend.resolveModel !== undefined &&
1049
+ resolvedModel === undefined) {
1050
+ return jsonResponse(400, {
1051
+ type: "error",
1052
+ error: {
1053
+ type: "invalid_request_error",
1054
+ message: `unknown model: ${body.model}`
1055
+ }
1056
+ });
1057
+ }
1058
+ const upstreamModel = resolvedModel ?? backend.defaultModel;
1059
+ if (upstreamModel === undefined) {
1060
+ return jsonResponse(503, {
1061
+ type: "error",
1062
+ error: {
1063
+ type: "unavailable",
1064
+ message: "no model is available; configure a provider"
1065
+ }
1066
+ });
1067
+ }
1068
+ // Server-executed web search is honored when the caller declared the server
1069
+ // tool, an executor is available, and no *client* tool already owns the
1070
+ // projected name (a client `web_search` must keep round-tripping untouched).
1071
+ const declaresWebSearch = body.tools?.some(isAnthropicWebSearchTool) === true;
1072
+ const clientNameCollision = body.tools?.some((tool) => !isAnthropicServerTool(tool) && tool.name === WEB_SEARCH_TOOL_NAME) === true;
1073
+ const executor = declaresWebSearch && !clientNameCollision ? resolveWebSearchExecutor("anthropic") : undefined;
1074
+ const serverTools = executor !== undefined;
1075
+ const chat = anthropicToChat(body, upstreamModel, { serverTools });
1076
+ const requestOptions = {
1077
+ ...backendOptions,
1078
+ modelCallId,
1079
+ // The streamed response is translated to Anthropic SSE by
1080
+ // openAiSseToAnthropic, which emits its own `ping` keepalive.
1081
+ ...(body.stream === true ? { translated: true } : {})
1082
+ };
1083
+ const upstream = await backend.chat(chat, signal, requestOptions);
1084
+ if (!upstream.ok) {
1085
+ const detail = await upstream.text();
1086
+ return jsonResponse(upstream.status, { type: "error", error: unwrapUpstreamError(detail) });
1087
+ }
1088
+ if (executor !== undefined) {
1089
+ const loopOptions = {
1090
+ chat,
1091
+ runStep: (stepChat) => backend.chat(stepChat, signal, requestOptions),
1092
+ serverToolNames: new Set([WEB_SEARCH_TOOL_NAME]),
1093
+ executor,
1094
+ ...(signal !== undefined ? { signal } : {})
1095
+ };
1096
+ if (body.stream === true) {
1097
+ const source = upstream.body;
1098
+ if (source === null)
1099
+ return jsonResponse(502, { type: "error", error: { type: "api_error", message: "no upstream stream" } });
1100
+ const composed = composeServerToolStream({ ...loopOptions, firstStep: upstream });
1101
+ return new Response(openAiSseToAnthropic(composed, requestedModel), {
1102
+ status: 200,
1103
+ headers: { "content-type": "text/event-stream", "cache-control": "no-cache" }
1104
+ });
1105
+ }
1106
+ const outcome = await runBufferedServerToolLoop({ ...loopOptions, firstStep: upstream });
1107
+ if (outcome.kind === "upstream_error") {
1108
+ const detail = await outcome.response.text();
1109
+ return jsonResponse(outcome.response.status, {
1110
+ type: "error",
1111
+ error: { type: "api_error", message: detail.slice(0, 2000) }
1112
+ });
1113
+ }
1114
+ return jsonResponse(200, chatToAnthropicMessage(outcome.openai, requestedModel, outcome.searches, outcome.events));
1115
+ }
1116
+ if (body.stream === true) {
1117
+ const source = upstream.body;
1118
+ if (source === null)
1119
+ return jsonResponse(502, { type: "error", error: { type: "api_error", message: "no upstream stream" } });
1120
+ return new Response(openAiSseToAnthropic(source, requestedModel), {
1121
+ status: 200,
1122
+ headers: { "content-type": "text/event-stream", "cache-control": "no-cache" }
1123
+ });
1124
+ }
1125
+ const openai = (await upstream.json());
1126
+ return jsonResponse(200, chatToAnthropicMessage(openai, requestedModel));
1127
+ }
1128
+ export function handleCountTokens(body) {
1129
+ return jsonResponse(200, { input_tokens: countTokensEstimate(body) });
1130
+ }
1131
+ /** Claude Code only lists models whose id begins with `claude` or `anthropic`. */
1132
+ function isAnthropicFamilyId(id) {
1133
+ return id.startsWith("claude") || id.startsWith("anthropic");
1134
+ }
1135
+ /** The `claude-` prefix used to alias non-Anthropic models past Claude's filter. */
1136
+ export const CLAUDE_ALIAS_PREFIX = "claude-";
1137
+ /**
1138
+ * The id a model is advertised under in Claude Code's `/model` picker. Claude
1139
+ * only lists ids beginning with `claude`/`anthropic`, so non-Anthropic models
1140
+ * are aliased with a `claude-` prefix; the gateway maps the alias back when
1141
+ * routing (see `resolveAlias`), and the picker shows the real id via
1142
+ * `display_name`. This is the claude-code-router trick: the `model` field is an
1143
+ * identifier we control end-to-end, so any model can be made selectable.
1144
+ */
1145
+ export function claudeModelAlias(id) {
1146
+ return isAnthropicFamilyId(id) ? id : `${CLAUDE_ALIAS_PREFIX}${id}`;
1147
+ }
1148
+ export function resolveClaudeModelAlias(requested, modelIds = []) {
1149
+ if (requested === undefined || modelIds.includes(requested))
1150
+ return requested;
1151
+ if (!requested.startsWith(CLAUDE_ALIAS_PREFIX))
1152
+ return requested;
1153
+ const candidate = requested.slice(CLAUDE_ALIAS_PREFIX.length);
1154
+ return modelIds.includes(candidate) && claudeModelAlias(candidate) === requested
1155
+ ? candidate
1156
+ : requested;
1157
+ }
1158
+ /**
1159
+ * Anthropic-shaped `/v1/models` discovery response. Every advertised model is
1160
+ * listed so it appears in Claude Code's `/model` picker: Anthropic-family ids
1161
+ * as-is, others under a `claude-`prefixed alias with the real id as
1162
+ * `display_name`. `modelIds` is the full advertised set (default model first);
1163
+ * when absent we fall back to the single backend default.
1164
+ */
1165
+ export function anthropicModelsResponse(backendModel, modelIds, modelRoutes = []) {
1166
+ const source = modelIds !== undefined && modelIds.length > 0
1167
+ ? modelIds
1168
+ : backendModel !== undefined
1169
+ ? [backendModel]
1170
+ : [];
1171
+ const seen = new Set();
1172
+ const routes = new Map(modelRoutes.map((route) => [route.publicId, route]));
1173
+ const models = [];
1174
+ for (const realId of source) {
1175
+ const route = routes.get(realId);
1176
+ const displayName = route?.provider === "claude-code" ? route.nativeId : realId;
1177
+ const id = claudeModelAlias(displayName);
1178
+ if (seen.has(id))
1179
+ continue;
1180
+ seen.add(id);
1181
+ models.push({
1182
+ type: "model",
1183
+ id,
1184
+ display_name: displayName,
1185
+ created_at: new Date(0).toISOString()
1186
+ });
1187
+ }
1188
+ const ids = models.map((model) => model.id);
1189
+ return new Response(JSON.stringify({
1190
+ data: models,
1191
+ has_more: false,
1192
+ first_id: ids[0],
1193
+ last_id: ids[ids.length - 1]
1194
+ }), { status: 200, headers: { "content-type": "application/json" } });
1195
+ }