@velum-labs/routekit-gateway 0.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +201 -0
- package/README.md +28 -0
- package/dist/acp-agent.d.ts +38 -0
- package/dist/acp-agent.js +142 -0
- package/dist/acp-registry.d.ts +36 -0
- package/dist/acp-registry.js +85 -0
- package/dist/adapters/anthropic.d.ts +131 -0
- package/dist/adapters/anthropic.js +1195 -0
- package/dist/adapters/chat.d.ts +14 -0
- package/dist/adapters/chat.js +34 -0
- package/dist/adapters/cursor.d.ts +34 -0
- package/dist/adapters/cursor.js +305 -0
- package/dist/adapters/dropped.d.ts +10 -0
- package/dist/adapters/dropped.js +24 -0
- package/dist/adapters/openai-chat-wire.d.ts +93 -0
- package/dist/adapters/openai-chat-wire.js +143 -0
- package/dist/adapters/responses-stream.d.ts +7 -0
- package/dist/adapters/responses-stream.js +597 -0
- package/dist/adapters/responses.d.ts +174 -0
- package/dist/adapters/responses.js +778 -0
- package/dist/adapters/server-tool-loop.d.ts +94 -0
- package/dist/adapters/server-tool-loop.js +477 -0
- package/dist/adapters/upstream-error.d.ts +14 -0
- package/dist/adapters/upstream-error.js +25 -0
- package/dist/adapters/validate.d.ts +27 -0
- package/dist/adapters/validate.js +180 -0
- package/dist/adapters/web-search.d.ts +46 -0
- package/dist/adapters/web-search.js +151 -0
- package/dist/auth.d.ts +10 -0
- package/dist/auth.js +28 -0
- package/dist/backend.d.ts +151 -0
- package/dist/backend.js +143 -0
- package/dist/capacity-pool.d.ts +31 -0
- package/dist/capacity-pool.js +99 -0
- package/dist/cost.d.ts +49 -0
- package/dist/cost.js +112 -0
- package/dist/endpoint-health.d.ts +54 -0
- package/dist/endpoint-health.js +123 -0
- package/dist/index.d.ts +40 -0
- package/dist/index.js +23 -0
- package/dist/provenance.d.ts +31 -0
- package/dist/provenance.js +191 -0
- package/dist/provider-backends.d.ts +40 -0
- package/dist/provider-backends.js +1050 -0
- package/dist/provider-source.d.ts +40 -0
- package/dist/provider-source.js +293 -0
- package/dist/router.d.ts +168 -0
- package/dist/router.js +474 -0
- package/dist/server.d.ts +67 -0
- package/dist/server.js +930 -0
- package/dist/sse/chat-assembler.d.ts +45 -0
- package/dist/sse/chat-assembler.js +190 -0
- package/dist/sse/parse.d.ts +50 -0
- package/dist/sse/parse.js +149 -0
- package/dist/sse-wire.d.ts +10 -0
- package/dist/sse-wire.js +31 -0
- package/dist/switching-proxy.d.ts +15 -0
- package/dist/switching-proxy.js +232 -0
- package/dist/test/acp-agent.test.d.ts +1 -0
- package/dist/test/acp-agent.test.js +66 -0
- package/dist/test/acp-registry.test.d.ts +1 -0
- package/dist/test/acp-registry.test.js +70 -0
- package/dist/test/anthropic.test.d.ts +1 -0
- package/dist/test/anthropic.test.js +793 -0
- package/dist/test/auth.test.d.ts +1 -0
- package/dist/test/auth.test.js +25 -0
- package/dist/test/boundary.test.d.ts +1 -0
- package/dist/test/boundary.test.js +32 -0
- package/dist/test/chat.test.d.ts +1 -0
- package/dist/test/chat.test.js +418 -0
- package/dist/test/cost.test.d.ts +1 -0
- package/dist/test/cost.test.js +60 -0
- package/dist/test/cursor.test.d.ts +1 -0
- package/dist/test/cursor.test.js +100 -0
- package/dist/test/drain.test.d.ts +1 -0
- package/dist/test/drain.test.js +116 -0
- package/dist/test/dropped.test.d.ts +1 -0
- package/dist/test/dropped.test.js +80 -0
- package/dist/test/endpoint-health.test.d.ts +1 -0
- package/dist/test/endpoint-health.test.js +73 -0
- package/dist/test/provenance.test.d.ts +1 -0
- package/dist/test/provenance.test.js +176 -0
- package/dist/test/provider-backends.test.d.ts +1 -0
- package/dist/test/provider-backends.test.js +699 -0
- package/dist/test/responses.test.d.ts +1 -0
- package/dist/test/responses.test.js +813 -0
- package/dist/test/routed-backend.test.d.ts +1 -0
- package/dist/test/routed-backend.test.js +39 -0
- package/dist/test/router.test.d.ts +1 -0
- package/dist/test/router.test.js +297 -0
- package/dist/test/server-resilience.test.d.ts +1 -0
- package/dist/test/server-resilience.test.js +169 -0
- package/dist/test/sse-codec.test.d.ts +1 -0
- package/dist/test/sse-codec.test.js +186 -0
- package/dist/test/web-search-loop.test.d.ts +1 -0
- package/dist/test/web-search-loop.test.js +469 -0
- package/dist/test/wire-validation.test.d.ts +1 -0
- package/dist/test/wire-validation.test.js +140 -0
- package/package.json +48 -0
|
@@ -0,0 +1,1195 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Anthropic Messages adapter. Claude Code speaks the Anthropic Messages API to
|
|
3
|
+
* whatever `ANTHROPIC_BASE_URL` points at, so to back it with a local model we
|
|
4
|
+
* translate `/v1/messages` (and `/v1/messages/count_tokens`, and the
|
|
5
|
+
* `/v1/models` discovery probe) to and from the gateway's OpenAI Chat
|
|
6
|
+
* Completions core. The pure translation functions are exported for testing;
|
|
7
|
+
* the request handler wires them to a `Backend` and returns a `Response` the
|
|
8
|
+
* server pipes straight to the client (JSON or SSE).
|
|
9
|
+
*/
|
|
10
|
+
import { estimateTokens, randomId } from "@velum-labs/routekit-runtime";
|
|
11
|
+
import { SseDecoder, SseParseError } from "../sse/parse.js";
|
|
12
|
+
import { ANTHROPIC_MESSAGE_CONTENT, ANTHROPIC_REQUEST_METADATA, attachReasoningSelection, attachReasoningSelectionError, anthropicReasoningDetailsOf } from "./openai-chat-wire.js";
|
|
13
|
+
import { droppedField } from "./dropped.js";
|
|
14
|
+
import { unwrapUpstreamError } from "./upstream-error.js";
|
|
15
|
+
import { composeServerToolStream, runBufferedServerToolLoop, serverToolMarkerOf } from "./server-tool-loop.js";
|
|
16
|
+
import { resolveWebSearchExecutor } from "./web-search.js";
|
|
17
|
+
const ENCODER = new TextEncoder();
|
|
18
|
+
/**
|
|
19
|
+
* Whether an Anthropic tool is *server-executed* (run by Anthropic's backend,
|
|
20
|
+
* e.g. `web_search_20250305` / `code_execution_*`). Nothing behind this gateway
|
|
21
|
+
* can execute those, so advertising them to the upstream model would only produce
|
|
22
|
+
* calls nobody answers. Everything else — plain client tools (no `type` /
|
|
23
|
+
* `custom`) and Anthropic-defined client tools (`bash_*`, `text_editor_*`,
|
|
24
|
+
* `computer_*`), all of which the caller executes via ordinary `tool_use`
|
|
25
|
+
* blocks — is projected through.
|
|
26
|
+
*/
|
|
27
|
+
function isAnthropicServerTool(tool) {
|
|
28
|
+
const type = tool.type ?? "";
|
|
29
|
+
return type.startsWith("web_search") || type.startsWith("code_execution");
|
|
30
|
+
}
|
|
31
|
+
/** A server web search tool declaration (`web_search_20250305` et al.) — the
|
|
32
|
+
* one server tool the gateway can honor via its own web-search executor. */
|
|
33
|
+
function isAnthropicWebSearchTool(tool) {
|
|
34
|
+
return (tool.type ?? "").startsWith("web_search");
|
|
35
|
+
}
|
|
36
|
+
/** The name the gateway-executed web search tool is projected under chat-side. */
|
|
37
|
+
const WEB_SEARCH_TOOL_NAME = "web_search";
|
|
38
|
+
const WEB_SEARCH_TOOL_DESCRIPTION = "Search the web for current, factual information. The search runs server-side and " +
|
|
39
|
+
"returns result text with source URLs. Use it when the answer depends on information " +
|
|
40
|
+
"that may have changed since your training data.";
|
|
41
|
+
const WEB_SEARCH_TOOL_PARAMETERS = {
|
|
42
|
+
type: "object",
|
|
43
|
+
properties: {
|
|
44
|
+
query: { type: "string", description: "The web search query." }
|
|
45
|
+
},
|
|
46
|
+
required: ["query"],
|
|
47
|
+
additionalProperties: false
|
|
48
|
+
};
|
|
49
|
+
/** Render an echoed `web_search_tool_result`'s content as a chat tool message.
|
|
50
|
+
* Bulky opaque fields (`encrypted_content`) are stripped; the upstream model
|
|
51
|
+
* only needs the urls/titles to remember what the search found. */
|
|
52
|
+
function webSearchResultText(content) {
|
|
53
|
+
if (!Array.isArray(content))
|
|
54
|
+
return JSON.stringify(content ?? null);
|
|
55
|
+
const results = content.map((entry) => {
|
|
56
|
+
if (entry === null || typeof entry !== "object")
|
|
57
|
+
return entry;
|
|
58
|
+
const { encrypted_content: _encrypted, ...rest } = entry;
|
|
59
|
+
return rest;
|
|
60
|
+
});
|
|
61
|
+
return JSON.stringify(results);
|
|
62
|
+
}
|
|
63
|
+
// ---- request translation ----
|
|
64
|
+
function systemText(system) {
|
|
65
|
+
if (system == null)
|
|
66
|
+
return "";
|
|
67
|
+
if (typeof system === "string")
|
|
68
|
+
return system;
|
|
69
|
+
return system
|
|
70
|
+
.map((block) => block !== null && typeof block === "object" && typeof block.text === "string"
|
|
71
|
+
? block.text
|
|
72
|
+
: "")
|
|
73
|
+
.join("\n");
|
|
74
|
+
}
|
|
75
|
+
function blockText(content) {
|
|
76
|
+
if (content == null)
|
|
77
|
+
return "";
|
|
78
|
+
if (typeof content === "string")
|
|
79
|
+
return content;
|
|
80
|
+
return content
|
|
81
|
+
.map((block) => block !== null && typeof block === "object" && block.type === "text"
|
|
82
|
+
? block.text
|
|
83
|
+
: "")
|
|
84
|
+
.join("");
|
|
85
|
+
}
|
|
86
|
+
function mapToolChoice(choice) {
|
|
87
|
+
switch (choice.type) {
|
|
88
|
+
case "auto":
|
|
89
|
+
return "auto";
|
|
90
|
+
case "any":
|
|
91
|
+
return "required";
|
|
92
|
+
case "none":
|
|
93
|
+
return "none";
|
|
94
|
+
case "tool":
|
|
95
|
+
return { type: "function", function: { name: choice.name ?? "" } };
|
|
96
|
+
default: {
|
|
97
|
+
const unreachable = choice.type;
|
|
98
|
+
return unreachable;
|
|
99
|
+
}
|
|
100
|
+
}
|
|
101
|
+
}
|
|
102
|
+
function mapThinking(thinking, outputConfig) {
|
|
103
|
+
if (thinking.type === "disabled")
|
|
104
|
+
return undefined;
|
|
105
|
+
const effort = outputConfig?.effort;
|
|
106
|
+
return typeof effort === "string" && effort.length > 0 ? effort : undefined;
|
|
107
|
+
}
|
|
108
|
+
function thinkingValidationError(body) {
|
|
109
|
+
const thinking = body.thinking;
|
|
110
|
+
if (thinking == null || thinking.type !== "enabled")
|
|
111
|
+
return undefined;
|
|
112
|
+
const budget = thinking.budget_tokens;
|
|
113
|
+
if (!Number.isInteger(budget) || budget < 1_024) {
|
|
114
|
+
return "thinking.budget_tokens must be an integer greater than or equal to 1024";
|
|
115
|
+
}
|
|
116
|
+
if (typeof body.max_tokens === "number" && budget >= body.max_tokens) {
|
|
117
|
+
return `thinking.budget_tokens must be less than max_tokens (${body.max_tokens})`;
|
|
118
|
+
}
|
|
119
|
+
return undefined;
|
|
120
|
+
}
|
|
121
|
+
function toolResultContent(result) {
|
|
122
|
+
const text = blockText(result.content);
|
|
123
|
+
return result.is_error === true ? `[tool_error]\n${text}` : text;
|
|
124
|
+
}
|
|
125
|
+
function attachAnthropicContent(message, content) {
|
|
126
|
+
Object.defineProperty(message, ANTHROPIC_MESSAGE_CONTENT, {
|
|
127
|
+
value: [...content],
|
|
128
|
+
enumerable: true
|
|
129
|
+
});
|
|
130
|
+
}
|
|
131
|
+
/**
|
|
132
|
+
* Translate an Anthropic Messages request to an OpenAI Chat Completions body.
|
|
133
|
+
* The upstream model is always the backend's own model (Claude Code sends a
|
|
134
|
+
* `claude-*` id the local server would not recognise); the requested id is
|
|
135
|
+
* only echoed back in the response.
|
|
136
|
+
*/
|
|
137
|
+
export function anthropicToChat(body, backendModel, options = {}) {
|
|
138
|
+
const messages = [];
|
|
139
|
+
const system = systemText(body.system);
|
|
140
|
+
if (system.length > 0)
|
|
141
|
+
messages.push({ role: "system", content: system });
|
|
142
|
+
for (const message of body.messages) {
|
|
143
|
+
if (typeof message.content === "string") {
|
|
144
|
+
messages.push({ role: message.role, content: message.content });
|
|
145
|
+
continue;
|
|
146
|
+
}
|
|
147
|
+
const textParts = [];
|
|
148
|
+
const imageParts = [];
|
|
149
|
+
const toolCalls = [];
|
|
150
|
+
const toolResults = [];
|
|
151
|
+
const nativeContent = [];
|
|
152
|
+
let hasReplayableThinking = false;
|
|
153
|
+
// Echoed gateway-executed (or genuinely provider-executed, in a resumed
|
|
154
|
+
// session) web searches: `server_tool_use` + `web_search_tool_result`
|
|
155
|
+
// blocks ride in the assistant message and round-trip losslessly.
|
|
156
|
+
const serverToolUses = [];
|
|
157
|
+
const serverToolResults = [];
|
|
158
|
+
for (const block of message.content) {
|
|
159
|
+
switch (block.type) {
|
|
160
|
+
case "text":
|
|
161
|
+
textParts.push(block.text);
|
|
162
|
+
nativeContent.push({ type: "text", text: block.text });
|
|
163
|
+
break;
|
|
164
|
+
case "image": {
|
|
165
|
+
const source = block.source;
|
|
166
|
+
imageParts.push({
|
|
167
|
+
type: "image_url",
|
|
168
|
+
image_url: { url: `data:${source.media_type};base64,${source.data}` }
|
|
169
|
+
});
|
|
170
|
+
break;
|
|
171
|
+
}
|
|
172
|
+
case "tool_use": {
|
|
173
|
+
const tool = block;
|
|
174
|
+
toolCalls.push({
|
|
175
|
+
id: tool.id,
|
|
176
|
+
type: "function",
|
|
177
|
+
function: { name: tool.name, arguments: JSON.stringify(tool.input ?? {}) }
|
|
178
|
+
});
|
|
179
|
+
nativeContent.push({
|
|
180
|
+
type: "tool_use",
|
|
181
|
+
id: tool.id,
|
|
182
|
+
name: tool.name,
|
|
183
|
+
input: tool.input ?? {}
|
|
184
|
+
});
|
|
185
|
+
break;
|
|
186
|
+
}
|
|
187
|
+
case "tool_result": {
|
|
188
|
+
const result = block;
|
|
189
|
+
toolResults.push({ id: result.tool_use_id, content: toolResultContent(result) });
|
|
190
|
+
break;
|
|
191
|
+
}
|
|
192
|
+
case "server_tool_use": {
|
|
193
|
+
const tool = block;
|
|
194
|
+
serverToolUses.push({
|
|
195
|
+
id: tool.id,
|
|
196
|
+
type: "function",
|
|
197
|
+
function: { name: tool.name, arguments: JSON.stringify(tool.input ?? {}) }
|
|
198
|
+
});
|
|
199
|
+
break;
|
|
200
|
+
}
|
|
201
|
+
case "web_search_tool_result": {
|
|
202
|
+
const result = block;
|
|
203
|
+
serverToolResults.push({
|
|
204
|
+
id: result.tool_use_id ?? "",
|
|
205
|
+
content: webSearchResultText(result.content)
|
|
206
|
+
});
|
|
207
|
+
break;
|
|
208
|
+
}
|
|
209
|
+
case "thinking": {
|
|
210
|
+
const thinking = block;
|
|
211
|
+
// Only provider-issued non-empty signatures are safe to replay to
|
|
212
|
+
// Anthropic. Synthetic thinking emitted for another provider uses an
|
|
213
|
+
// empty signature and remains display-only.
|
|
214
|
+
if (typeof thinking.thinking === "string" &&
|
|
215
|
+
typeof thinking.signature === "string" &&
|
|
216
|
+
thinking.signature.length > 0) {
|
|
217
|
+
nativeContent.push({
|
|
218
|
+
type: "thinking",
|
|
219
|
+
thinking: thinking.thinking,
|
|
220
|
+
signature: thinking.signature
|
|
221
|
+
});
|
|
222
|
+
hasReplayableThinking = true;
|
|
223
|
+
}
|
|
224
|
+
else {
|
|
225
|
+
droppedField("anthropic", "thinking", "message");
|
|
226
|
+
}
|
|
227
|
+
break;
|
|
228
|
+
}
|
|
229
|
+
case "redacted_thinking": {
|
|
230
|
+
const redacted = block;
|
|
231
|
+
if (typeof redacted.data === "string" && redacted.data.length > 0) {
|
|
232
|
+
nativeContent.push({ type: "redacted_thinking", data: redacted.data });
|
|
233
|
+
hasReplayableThinking = true;
|
|
234
|
+
}
|
|
235
|
+
else {
|
|
236
|
+
droppedField("anthropic", "redacted_thinking", "message");
|
|
237
|
+
}
|
|
238
|
+
break;
|
|
239
|
+
}
|
|
240
|
+
default:
|
|
241
|
+
droppedField("anthropic", block.type, "message");
|
|
242
|
+
break;
|
|
243
|
+
}
|
|
244
|
+
}
|
|
245
|
+
if (message.role === "assistant") {
|
|
246
|
+
const text = textParts.join("");
|
|
247
|
+
if (imageParts.length > 0) {
|
|
248
|
+
droppedField("anthropic", "image", "assistant_message");
|
|
249
|
+
}
|
|
250
|
+
// Replay echoed server web searches as a chat tool exchange preceding
|
|
251
|
+
// the assistant's answer, so the upstream model remembers what was
|
|
252
|
+
// searched and found rather than blindly repeating it.
|
|
253
|
+
if (serverToolUses.length > 0) {
|
|
254
|
+
messages.push({ role: "assistant", content: null, tool_calls: serverToolUses });
|
|
255
|
+
for (const use of serverToolUses) {
|
|
256
|
+
const result = serverToolResults.find((entry) => entry.id === use.id);
|
|
257
|
+
messages.push({
|
|
258
|
+
role: "tool",
|
|
259
|
+
tool_call_id: use.id ?? "",
|
|
260
|
+
content: result?.content ?? "[web search results not retained]"
|
|
261
|
+
});
|
|
262
|
+
}
|
|
263
|
+
}
|
|
264
|
+
if (text.length > 0 || toolCalls.length > 0 || serverToolUses.length === 0) {
|
|
265
|
+
const assistant = { role: "assistant", content: text.length > 0 ? text : null };
|
|
266
|
+
if (toolCalls.length > 0)
|
|
267
|
+
assistant.tool_calls = toolCalls;
|
|
268
|
+
if (hasReplayableThinking)
|
|
269
|
+
attachAnthropicContent(assistant, nativeContent);
|
|
270
|
+
messages.push(assistant);
|
|
271
|
+
}
|
|
272
|
+
continue;
|
|
273
|
+
}
|
|
274
|
+
// user turn: tool results become standalone tool messages; remaining
|
|
275
|
+
// text/images become a user message.
|
|
276
|
+
for (const result of toolResults) {
|
|
277
|
+
messages.push({ role: "tool", tool_call_id: result.id, content: result.content });
|
|
278
|
+
}
|
|
279
|
+
const text = textParts.join("");
|
|
280
|
+
if (imageParts.length > 0) {
|
|
281
|
+
const parts = [];
|
|
282
|
+
if (text.length > 0)
|
|
283
|
+
parts.push({ type: "text", text });
|
|
284
|
+
parts.push(...imageParts);
|
|
285
|
+
messages.push({ role: "user", content: parts });
|
|
286
|
+
}
|
|
287
|
+
else if (text.length > 0 || toolResults.length === 0) {
|
|
288
|
+
messages.push({ role: "user", content: text });
|
|
289
|
+
}
|
|
290
|
+
}
|
|
291
|
+
const chat = {
|
|
292
|
+
model: backendModel ?? body.model ?? "",
|
|
293
|
+
messages,
|
|
294
|
+
stream: body.stream === true
|
|
295
|
+
};
|
|
296
|
+
// `max_completion_tokens`, not legacy `max_tokens`: OpenAI reasoning models
|
|
297
|
+
// reject the latter, and the other dialect adapters already emit the modern
|
|
298
|
+
// field (Claude Code always sends `max_tokens`, so this path is always hit).
|
|
299
|
+
if (typeof body.max_tokens === "number")
|
|
300
|
+
chat.max_completion_tokens = body.max_tokens;
|
|
301
|
+
if (typeof body.temperature === "number")
|
|
302
|
+
chat.temperature = body.temperature;
|
|
303
|
+
if (typeof body.top_p === "number")
|
|
304
|
+
chat.top_p = body.top_p;
|
|
305
|
+
if (typeof body.top_k === "number")
|
|
306
|
+
chat.top_k = body.top_k;
|
|
307
|
+
// Explicit nulls mean "unset" (see AnthropicRequest).
|
|
308
|
+
if (body.metadata != null)
|
|
309
|
+
droppedField("anthropic", "metadata");
|
|
310
|
+
if (body.output_config != null &&
|
|
311
|
+
Object.hasOwn(body.output_config, "effort") &&
|
|
312
|
+
body.output_config.effort !== null &&
|
|
313
|
+
(typeof body.output_config.effort !== "string" ||
|
|
314
|
+
body.output_config.effort.length === 0)) {
|
|
315
|
+
attachReasoningSelectionError(chat, "output_config.effort must be a non-empty string");
|
|
316
|
+
}
|
|
317
|
+
// `thinking: null` means "no extended thinking" — skip, never dereference
|
|
318
|
+
// (same failure class as the Responses adapter's `reasoning: null`).
|
|
319
|
+
if (body.thinking != null) {
|
|
320
|
+
const reasoningEffort = mapThinking(body.thinking, body.output_config);
|
|
321
|
+
if (body.thinking.type === "disabled") {
|
|
322
|
+
attachReasoningSelection(chat, { mode: "disabled" });
|
|
323
|
+
}
|
|
324
|
+
else if (reasoningEffort !== undefined) {
|
|
325
|
+
chat.reasoning_effort = reasoningEffort;
|
|
326
|
+
attachReasoningSelection(chat, {
|
|
327
|
+
mode: "effort",
|
|
328
|
+
effort: reasoningEffort
|
|
329
|
+
});
|
|
330
|
+
}
|
|
331
|
+
else if (body.thinking.type === "adaptive") {
|
|
332
|
+
attachReasoningSelection(chat, { mode: "adaptive" });
|
|
333
|
+
}
|
|
334
|
+
else {
|
|
335
|
+
attachReasoningSelection(chat, {
|
|
336
|
+
mode: "budget",
|
|
337
|
+
budgetTokens: body.thinking.budget_tokens
|
|
338
|
+
});
|
|
339
|
+
}
|
|
340
|
+
}
|
|
341
|
+
const metadata = {
|
|
342
|
+
...(body.thinking != null ? { thinking: body.thinking } : {}),
|
|
343
|
+
...(body.output_config !== undefined ? { output_config: body.output_config } : {})
|
|
344
|
+
};
|
|
345
|
+
if (Object.keys(metadata).length > 0) {
|
|
346
|
+
Object.defineProperty(chat, ANTHROPIC_REQUEST_METADATA, {
|
|
347
|
+
value: metadata,
|
|
348
|
+
enumerable: true
|
|
349
|
+
});
|
|
350
|
+
}
|
|
351
|
+
if (Array.isArray(body.stop_sequences) && body.stop_sequences.length > 0) {
|
|
352
|
+
chat.stop = body.stop_sequences;
|
|
353
|
+
}
|
|
354
|
+
if (Array.isArray(body.tools) && body.tools.length > 0) {
|
|
355
|
+
// Web search is honorable when an executor exists (the server-tool loop
|
|
356
|
+
// runs it); other server tools (`code_execution_*`) stay excluded.
|
|
357
|
+
const honorWebSearch = options.serverTools === true;
|
|
358
|
+
const excluded = body.tools.filter((tool) => isAnthropicServerTool(tool) && !(honorWebSearch && isAnthropicWebSearchTool(tool)));
|
|
359
|
+
if (excluded.length > 0) {
|
|
360
|
+
for (const tool of excluded) {
|
|
361
|
+
droppedField("anthropic", tool.name ?? tool.type ?? "server_tool", "tools");
|
|
362
|
+
}
|
|
363
|
+
if (process.env.ROUTEKIT_DEBUG) {
|
|
364
|
+
process.stderr.write(`[routekit-debug] anthropic: excluding ${excluded.length} server-executed tool(s) ` +
|
|
365
|
+
`from the request: ${excluded.map((tool) => tool.name).join(", ")}\n`);
|
|
366
|
+
}
|
|
367
|
+
}
|
|
368
|
+
const tools = body.tools
|
|
369
|
+
.filter((tool) => !isAnthropicServerTool(tool) && typeof tool.name === "string" && tool.name.length > 0)
|
|
370
|
+
.map((tool) => ({
|
|
371
|
+
type: "function",
|
|
372
|
+
function: {
|
|
373
|
+
name: tool.name,
|
|
374
|
+
...(tool.description !== undefined ? { description: tool.description } : {}),
|
|
375
|
+
parameters: tool.input_schema ?? { type: "object", properties: {} }
|
|
376
|
+
}
|
|
377
|
+
}));
|
|
378
|
+
if (honorWebSearch &&
|
|
379
|
+
body.tools.some(isAnthropicWebSearchTool) &&
|
|
380
|
+
!tools.some((tool) => tool.function.name === WEB_SEARCH_TOOL_NAME)) {
|
|
381
|
+
tools.push({
|
|
382
|
+
type: "function",
|
|
383
|
+
function: {
|
|
384
|
+
name: WEB_SEARCH_TOOL_NAME,
|
|
385
|
+
description: WEB_SEARCH_TOOL_DESCRIPTION,
|
|
386
|
+
parameters: WEB_SEARCH_TOOL_PARAMETERS
|
|
387
|
+
}
|
|
388
|
+
});
|
|
389
|
+
}
|
|
390
|
+
if (tools.length > 0)
|
|
391
|
+
chat.tools = tools;
|
|
392
|
+
}
|
|
393
|
+
if (body.tool_choice != null) {
|
|
394
|
+
chat.tool_choice = mapToolChoice(body.tool_choice);
|
|
395
|
+
if (body.tool_choice.disable_parallel_tool_use === true)
|
|
396
|
+
chat.parallel_tool_calls = false;
|
|
397
|
+
}
|
|
398
|
+
if (body.stream === true)
|
|
399
|
+
chat.stream_options = { include_usage: true };
|
|
400
|
+
return chat;
|
|
401
|
+
}
|
|
402
|
+
// ---- response translation ----
|
|
403
|
+
export function mapStopReason(finishReason) {
|
|
404
|
+
switch (finishReason) {
|
|
405
|
+
case "length":
|
|
406
|
+
return "max_tokens";
|
|
407
|
+
case "tool_calls":
|
|
408
|
+
return "tool_use";
|
|
409
|
+
case "content_filter":
|
|
410
|
+
return "refusal";
|
|
411
|
+
case "stop":
|
|
412
|
+
case null:
|
|
413
|
+
case undefined:
|
|
414
|
+
return "end_turn";
|
|
415
|
+
default:
|
|
416
|
+
return "end_turn";
|
|
417
|
+
}
|
|
418
|
+
}
|
|
419
|
+
/** The native Anthropic blocks for one gateway-executed web search: the
|
|
420
|
+
* `server_tool_use` and its `web_search_tool_result`. Anthropic-executor
|
|
421
|
+
* results pass through verbatim; other executors build result blocks
|
|
422
|
+
* from their citations. */
|
|
423
|
+
function executedSearchBlocks(search) {
|
|
424
|
+
const resultContent = search.status !== "completed"
|
|
425
|
+
? { type: "web_search_tool_result_error", error_code: "unavailable" }
|
|
426
|
+
: (search.outcome?.anthropicResultBlocks ??
|
|
427
|
+
(search.outcome?.citations ?? []).map((citation) => ({
|
|
428
|
+
type: "web_search_result",
|
|
429
|
+
url: citation.url,
|
|
430
|
+
...(citation.title !== undefined ? { title: citation.title } : {})
|
|
431
|
+
})));
|
|
432
|
+
return [
|
|
433
|
+
{ type: "server_tool_use", id: search.itemId, name: WEB_SEARCH_TOOL_NAME, input: { query: search.query } },
|
|
434
|
+
{ type: "web_search_tool_result", tool_use_id: search.itemId, content: resultContent }
|
|
435
|
+
];
|
|
436
|
+
}
|
|
437
|
+
export function chatToAnthropicMessage(openai, model, searches = [], events) {
|
|
438
|
+
const choice = openai.choices?.[0];
|
|
439
|
+
const message = choice?.message;
|
|
440
|
+
const content = [];
|
|
441
|
+
const nativeReasoning = anthropicReasoningDetailsOf(message?.reasoning_details, "message").sort((a, b) => a.index - b.index);
|
|
442
|
+
const appendNativeReasoning = (details) => {
|
|
443
|
+
for (const detail of details) {
|
|
444
|
+
if (detail.type === "thinking") {
|
|
445
|
+
content.push({
|
|
446
|
+
type: "thinking",
|
|
447
|
+
thinking: detail.thinking ?? "",
|
|
448
|
+
signature: detail.signature ?? ""
|
|
449
|
+
});
|
|
450
|
+
}
|
|
451
|
+
else {
|
|
452
|
+
content.push({ type: "redacted_thinking", data: detail.data });
|
|
453
|
+
}
|
|
454
|
+
}
|
|
455
|
+
};
|
|
456
|
+
// Gateway-executed steps precede the terminal model step. Preserve their
|
|
457
|
+
// signed/redacted reasoning in exact step order around search blocks.
|
|
458
|
+
if (events !== undefined) {
|
|
459
|
+
for (const event of events) {
|
|
460
|
+
if (event.kind === "reasoning") {
|
|
461
|
+
appendNativeReasoning(anthropicReasoningDetailsOf(event.details, "message"));
|
|
462
|
+
}
|
|
463
|
+
else {
|
|
464
|
+
content.push(...executedSearchBlocks(event.search));
|
|
465
|
+
}
|
|
466
|
+
}
|
|
467
|
+
}
|
|
468
|
+
else {
|
|
469
|
+
for (const search of searches)
|
|
470
|
+
content.push(...executedSearchBlocks(search));
|
|
471
|
+
}
|
|
472
|
+
appendNativeReasoning(nativeReasoning);
|
|
473
|
+
const rawReasoning = typeof message?.reasoning === "string" && message.reasoning.length > 0
|
|
474
|
+
? message.reasoning
|
|
475
|
+
: "";
|
|
476
|
+
const narration = typeof message?.reasoning_content === "string" &&
|
|
477
|
+
message.reasoning_content.length > 0
|
|
478
|
+
? message.reasoning_content.replace(/\*\*/g, "")
|
|
479
|
+
: "";
|
|
480
|
+
if (nativeReasoning.length === 0 && rawReasoning.length > 0) {
|
|
481
|
+
// Generic providers cannot produce an Anthropic-verifiable signature.
|
|
482
|
+
// The empty marker makes the block displayable; ingress deliberately
|
|
483
|
+
// refuses to replay it as native signed history.
|
|
484
|
+
content.push({
|
|
485
|
+
type: "thinking",
|
|
486
|
+
thinking: rawReasoning,
|
|
487
|
+
signature: ""
|
|
488
|
+
});
|
|
489
|
+
}
|
|
490
|
+
if (narration.length > 0) {
|
|
491
|
+
content.push({ type: "thinking", thinking: narration, signature: "" });
|
|
492
|
+
}
|
|
493
|
+
const text = typeof message?.content === "string" ? message.content : "";
|
|
494
|
+
if (text.length > 0)
|
|
495
|
+
content.push({ type: "text", text });
|
|
496
|
+
if (Array.isArray(message?.tool_calls)) {
|
|
497
|
+
for (const call of message.tool_calls) {
|
|
498
|
+
let input = {};
|
|
499
|
+
const args = call.function?.arguments;
|
|
500
|
+
if (typeof args === "string" && args.length > 0) {
|
|
501
|
+
try {
|
|
502
|
+
input = JSON.parse(args);
|
|
503
|
+
}
|
|
504
|
+
catch {
|
|
505
|
+
input = {};
|
|
506
|
+
}
|
|
507
|
+
}
|
|
508
|
+
content.push({
|
|
509
|
+
type: "tool_use",
|
|
510
|
+
id: call.id ?? `toolu_${randomId()}`,
|
|
511
|
+
name: call.function?.name ?? "",
|
|
512
|
+
input
|
|
513
|
+
});
|
|
514
|
+
}
|
|
515
|
+
}
|
|
516
|
+
if (content.length === 0)
|
|
517
|
+
content.push({ type: "text", text: "" });
|
|
518
|
+
const response = {
|
|
519
|
+
id: openai.id !== undefined ? `msg_${openai.id}` : `msg_${randomId()}`,
|
|
520
|
+
type: "message",
|
|
521
|
+
role: "assistant",
|
|
522
|
+
model,
|
|
523
|
+
content,
|
|
524
|
+
stop_reason: typeof choice?.anthropic_stop_reason === "string"
|
|
525
|
+
? choice.anthropic_stop_reason
|
|
526
|
+
: mapStopReason(choice?.finish_reason),
|
|
527
|
+
stop_sequence: typeof choice?.anthropic_stop_sequence === "string"
|
|
528
|
+
? choice.anthropic_stop_sequence
|
|
529
|
+
: null
|
|
530
|
+
};
|
|
531
|
+
if (openai.usage !== undefined) {
|
|
532
|
+
response.usage = {
|
|
533
|
+
...(openai.usage.prompt_tokens !== undefined ? { input_tokens: openai.usage.prompt_tokens } : {}),
|
|
534
|
+
...(openai.usage.completion_tokens !== undefined ? { output_tokens: openai.usage.completion_tokens } : {})
|
|
535
|
+
};
|
|
536
|
+
}
|
|
537
|
+
return response;
|
|
538
|
+
}
|
|
539
|
+
// ---- streaming translation (OpenAI chat SSE -> Anthropic Messages SSE) ----
|
|
540
|
+
function sse(type, data) {
|
|
541
|
+
return ENCODER.encode(`event: ${type}\ndata: ${JSON.stringify(data)}\n\n`);
|
|
542
|
+
}
|
|
543
|
+
export function openAiSseToAnthropic(upstream, model) {
|
|
544
|
+
const reader = upstream.getReader();
|
|
545
|
+
const sseDecoder = new SseDecoder();
|
|
546
|
+
// OpenAI tool-call fragments map onto Anthropic `tool_use` content blocks.
|
|
547
|
+
// Fragments are keyed by `index` when present, else by `id`; an id/index-less
|
|
548
|
+
// fragment (Anthropic/Responses translations omit `index`) appends to the last
|
|
549
|
+
// open call. Keying everything to index 0 used to merge parallel index-less
|
|
550
|
+
// calls into one block — the same bug the shared assembler now avoids.
|
|
551
|
+
const toolBlockByIndex = new Map();
|
|
552
|
+
const toolBlockById = new Map();
|
|
553
|
+
const toolBlocks = [];
|
|
554
|
+
let lastToolBlock;
|
|
555
|
+
const messageId = `msg_${randomId()}`;
|
|
556
|
+
const state = {
|
|
557
|
+
started: false,
|
|
558
|
+
textOpen: false,
|
|
559
|
+
textIndex: -1,
|
|
560
|
+
thinkingOpen: false,
|
|
561
|
+
thinkingIndex: -1,
|
|
562
|
+
thinkingSourceIndex: undefined,
|
|
563
|
+
pendingNarration: [],
|
|
564
|
+
outputStarted: false,
|
|
565
|
+
nextIndex: 0,
|
|
566
|
+
finished: false,
|
|
567
|
+
inputTokens: undefined,
|
|
568
|
+
outputTokens: undefined,
|
|
569
|
+
keepaliveTimer: undefined
|
|
570
|
+
};
|
|
571
|
+
const ensureStarted = (controller) => {
|
|
572
|
+
if (state.started)
|
|
573
|
+
return;
|
|
574
|
+
state.started = true;
|
|
575
|
+
controller.enqueue(sse("message_start", {
|
|
576
|
+
type: "message_start",
|
|
577
|
+
message: {
|
|
578
|
+
id: messageId,
|
|
579
|
+
type: "message",
|
|
580
|
+
role: "assistant",
|
|
581
|
+
model,
|
|
582
|
+
content: [],
|
|
583
|
+
stop_reason: null,
|
|
584
|
+
stop_sequence: null,
|
|
585
|
+
...(state.inputTokens !== undefined ? { usage: { input_tokens: state.inputTokens } } : {})
|
|
586
|
+
}
|
|
587
|
+
}));
|
|
588
|
+
};
|
|
589
|
+
// Generic reasoning has no provider-verifiable signature. Native Anthropic
|
|
590
|
+
// metadata below carries its real block lifecycle and signature separately.
|
|
591
|
+
const ensureThinking = (controller) => {
|
|
592
|
+
ensureStarted(controller);
|
|
593
|
+
if (state.thinkingOpen || state.outputStarted)
|
|
594
|
+
return;
|
|
595
|
+
state.thinkingOpen = true;
|
|
596
|
+
state.thinkingSourceIndex = undefined;
|
|
597
|
+
state.thinkingIndex = state.nextIndex++;
|
|
598
|
+
controller.enqueue(sse("content_block_start", {
|
|
599
|
+
type: "content_block_start",
|
|
600
|
+
index: state.thinkingIndex,
|
|
601
|
+
content_block: { type: "thinking", thinking: "" }
|
|
602
|
+
}));
|
|
603
|
+
};
|
|
604
|
+
const closeThinking = (controller, sourceIndex) => {
|
|
605
|
+
if (!state.thinkingOpen)
|
|
606
|
+
return;
|
|
607
|
+
if (sourceIndex !== undefined &&
|
|
608
|
+
state.thinkingSourceIndex !== undefined &&
|
|
609
|
+
sourceIndex !== state.thinkingSourceIndex) {
|
|
610
|
+
return;
|
|
611
|
+
}
|
|
612
|
+
state.thinkingOpen = false;
|
|
613
|
+
controller.enqueue(sse("content_block_stop", { type: "content_block_stop", index: state.thinkingIndex }));
|
|
614
|
+
state.thinkingSourceIndex = undefined;
|
|
615
|
+
};
|
|
616
|
+
const emitNarration = (controller, text) => {
|
|
617
|
+
ensureThinking(controller);
|
|
618
|
+
if (!state.thinkingOpen || state.thinkingSourceIndex !== undefined)
|
|
619
|
+
return;
|
|
620
|
+
controller.enqueue(sse("content_block_delta", {
|
|
621
|
+
type: "content_block_delta",
|
|
622
|
+
index: state.thinkingIndex,
|
|
623
|
+
delta: {
|
|
624
|
+
type: "thinking_delta",
|
|
625
|
+
thinking: text.replace(/\*\*/g, "")
|
|
626
|
+
}
|
|
627
|
+
}));
|
|
628
|
+
};
|
|
629
|
+
const flushPendingNarration = (controller) => {
|
|
630
|
+
if (state.pendingNarration.length === 0)
|
|
631
|
+
return;
|
|
632
|
+
const pending = state.pendingNarration.join("");
|
|
633
|
+
state.pendingNarration = [];
|
|
634
|
+
emitNarration(controller, pending);
|
|
635
|
+
};
|
|
636
|
+
const ensureText = (controller) => {
|
|
637
|
+
ensureStarted(controller);
|
|
638
|
+
closeThinking(controller);
|
|
639
|
+
flushPendingNarration(controller);
|
|
640
|
+
closeThinking(controller);
|
|
641
|
+
state.outputStarted = true;
|
|
642
|
+
if (state.textOpen)
|
|
643
|
+
return;
|
|
644
|
+
state.textOpen = true;
|
|
645
|
+
state.textIndex = state.nextIndex++;
|
|
646
|
+
controller.enqueue(sse("content_block_start", {
|
|
647
|
+
type: "content_block_start",
|
|
648
|
+
index: state.textIndex,
|
|
649
|
+
content_block: { type: "text", text: "" }
|
|
650
|
+
}));
|
|
651
|
+
};
|
|
652
|
+
const closeOpenBlocks = (controller) => {
|
|
653
|
+
closeThinking(controller);
|
|
654
|
+
flushPendingNarration(controller);
|
|
655
|
+
closeThinking(controller);
|
|
656
|
+
if (state.textOpen) {
|
|
657
|
+
controller.enqueue(sse("content_block_stop", { type: "content_block_stop", index: state.textIndex }));
|
|
658
|
+
}
|
|
659
|
+
for (const index of toolBlocks) {
|
|
660
|
+
controller.enqueue(sse("content_block_stop", { type: "content_block_stop", index }));
|
|
661
|
+
}
|
|
662
|
+
};
|
|
663
|
+
const finalize = (controller, stopReason, stopSequence = null) => {
|
|
664
|
+
if (state.finished)
|
|
665
|
+
return;
|
|
666
|
+
state.finished = true;
|
|
667
|
+
if (state.keepaliveTimer !== undefined)
|
|
668
|
+
clearInterval(state.keepaliveTimer);
|
|
669
|
+
closeOpenBlocks(controller);
|
|
670
|
+
controller.enqueue(sse("message_delta", {
|
|
671
|
+
type: "message_delta",
|
|
672
|
+
delta: { stop_reason: stopReason, stop_sequence: stopSequence },
|
|
673
|
+
...(state.inputTokens !== undefined || state.outputTokens !== undefined
|
|
674
|
+
? {
|
|
675
|
+
usage: {
|
|
676
|
+
...(state.inputTokens !== undefined ? { input_tokens: state.inputTokens } : {}),
|
|
677
|
+
...(state.outputTokens !== undefined ? { output_tokens: state.outputTokens } : {})
|
|
678
|
+
}
|
|
679
|
+
}
|
|
680
|
+
: {})
|
|
681
|
+
}));
|
|
682
|
+
controller.enqueue(sse("message_stop", { type: "message_stop" }));
|
|
683
|
+
};
|
|
684
|
+
/**
|
|
685
|
+
* The upstream ended (reader closed or a `[DONE]` arrived) before any
|
|
686
|
+
* `finish_reason`. Truncation is an error, not a clean stop (WS5.2): emit an
|
|
687
|
+
* Anthropic `error` event rather than fabricating `stop_reason:"end_turn"`, so
|
|
688
|
+
* the caller sees a failed turn instead of silently accepting a partial answer.
|
|
689
|
+
*/
|
|
690
|
+
const finalizeTruncated = (controller, detail) => {
|
|
691
|
+
if (state.finished)
|
|
692
|
+
return;
|
|
693
|
+
state.finished = true;
|
|
694
|
+
if (state.keepaliveTimer !== undefined)
|
|
695
|
+
clearInterval(state.keepaliveTimer);
|
|
696
|
+
closeOpenBlocks(controller);
|
|
697
|
+
controller.enqueue(sse("error", {
|
|
698
|
+
type: "error",
|
|
699
|
+
error: { type: "incomplete_stream", message: detail }
|
|
700
|
+
}));
|
|
701
|
+
};
|
|
702
|
+
const finalizeUpstreamError = (controller, error) => {
|
|
703
|
+
if (state.finished)
|
|
704
|
+
return;
|
|
705
|
+
state.finished = true;
|
|
706
|
+
if (state.keepaliveTimer !== undefined)
|
|
707
|
+
clearInterval(state.keepaliveTimer);
|
|
708
|
+
closeOpenBlocks(controller);
|
|
709
|
+
controller.enqueue(sse("error", {
|
|
710
|
+
type: "error",
|
|
711
|
+
error: unwrapUpstreamError(JSON.stringify({ error }))
|
|
712
|
+
}));
|
|
713
|
+
};
|
|
714
|
+
// The server-tool loop injects marker chunks around each gateway-executed
|
|
715
|
+
// web search; render them as native `server_tool_use` /
|
|
716
|
+
// `web_search_tool_result` blocks (each opened and closed immediately —
|
|
717
|
+
// their content is complete when the marker arrives).
|
|
718
|
+
const handleServerToolMarker = (controller, marker) => {
|
|
719
|
+
ensureStarted(controller);
|
|
720
|
+
closeThinking(controller);
|
|
721
|
+
flushPendingNarration(controller);
|
|
722
|
+
closeThinking(controller);
|
|
723
|
+
state.outputStarted = true;
|
|
724
|
+
if (marker.phase === "start") {
|
|
725
|
+
const index = state.nextIndex++;
|
|
726
|
+
controller.enqueue(sse("content_block_start", {
|
|
727
|
+
type: "content_block_start",
|
|
728
|
+
index,
|
|
729
|
+
content_block: { type: "server_tool_use", id: marker.item_id, name: "web_search", input: {} }
|
|
730
|
+
}));
|
|
731
|
+
controller.enqueue(sse("content_block_delta", {
|
|
732
|
+
type: "content_block_delta",
|
|
733
|
+
index,
|
|
734
|
+
delta: { type: "input_json_delta", partial_json: JSON.stringify({ query: marker.query }) }
|
|
735
|
+
}));
|
|
736
|
+
controller.enqueue(sse("content_block_stop", { type: "content_block_stop", index }));
|
|
737
|
+
return;
|
|
738
|
+
}
|
|
739
|
+
const index = state.nextIndex++;
|
|
740
|
+
const content = marker.status === "failed"
|
|
741
|
+
? { type: "web_search_tool_result_error", error_code: "unavailable" }
|
|
742
|
+
: (marker.result_blocks ?? []);
|
|
743
|
+
controller.enqueue(sse("content_block_start", {
|
|
744
|
+
type: "content_block_start",
|
|
745
|
+
index,
|
|
746
|
+
content_block: { type: "web_search_tool_result", tool_use_id: marker.item_id, content }
|
|
747
|
+
}));
|
|
748
|
+
controller.enqueue(sse("content_block_stop", { type: "content_block_stop", index }));
|
|
749
|
+
// A completed server-side tool step is an internal model boundary, not the
|
|
750
|
+
// final answer. The continuation model may legitimately begin with another
|
|
751
|
+
// signed thinking/redacted block.
|
|
752
|
+
state.outputStarted = false;
|
|
753
|
+
};
|
|
754
|
+
const handleReasoningDetails = (controller, details) => {
|
|
755
|
+
let carriedText = false;
|
|
756
|
+
for (const detail of details) {
|
|
757
|
+
if (detail.type === "redacted_thinking") {
|
|
758
|
+
if (state.outputStarted)
|
|
759
|
+
continue;
|
|
760
|
+
ensureStarted(controller);
|
|
761
|
+
closeThinking(controller);
|
|
762
|
+
const index = state.nextIndex++;
|
|
763
|
+
controller.enqueue(sse("content_block_start", {
|
|
764
|
+
type: "content_block_start",
|
|
765
|
+
index,
|
|
766
|
+
content_block: { type: "redacted_thinking", data: detail.data }
|
|
767
|
+
}));
|
|
768
|
+
controller.enqueue(sse("content_block_stop", { type: "content_block_stop", index }));
|
|
769
|
+
continue;
|
|
770
|
+
}
|
|
771
|
+
if (detail.phase === "start") {
|
|
772
|
+
if (state.outputStarted)
|
|
773
|
+
continue;
|
|
774
|
+
ensureStarted(controller);
|
|
775
|
+
closeThinking(controller);
|
|
776
|
+
state.thinkingOpen = true;
|
|
777
|
+
state.thinkingSourceIndex = detail.index;
|
|
778
|
+
state.thinkingIndex = state.nextIndex++;
|
|
779
|
+
controller.enqueue(sse("content_block_start", {
|
|
780
|
+
type: "content_block_start",
|
|
781
|
+
index: state.thinkingIndex,
|
|
782
|
+
content_block: {
|
|
783
|
+
type: "thinking",
|
|
784
|
+
thinking: "",
|
|
785
|
+
signature: detail.signature ?? ""
|
|
786
|
+
}
|
|
787
|
+
}));
|
|
788
|
+
continue;
|
|
789
|
+
}
|
|
790
|
+
if (state.thinkingSourceIndex !== detail.index ||
|
|
791
|
+
!state.thinkingOpen ||
|
|
792
|
+
state.outputStarted) {
|
|
793
|
+
continue;
|
|
794
|
+
}
|
|
795
|
+
if (detail.phase === "delta" && typeof detail.thinking === "string") {
|
|
796
|
+
carriedText = true;
|
|
797
|
+
controller.enqueue(sse("content_block_delta", {
|
|
798
|
+
type: "content_block_delta",
|
|
799
|
+
index: state.thinkingIndex,
|
|
800
|
+
delta: { type: "thinking_delta", thinking: detail.thinking }
|
|
801
|
+
}));
|
|
802
|
+
}
|
|
803
|
+
else if (detail.phase === "signature" && typeof detail.signature === "string") {
|
|
804
|
+
controller.enqueue(sse("content_block_delta", {
|
|
805
|
+
type: "content_block_delta",
|
|
806
|
+
index: state.thinkingIndex,
|
|
807
|
+
delta: { type: "signature_delta", signature: detail.signature }
|
|
808
|
+
}));
|
|
809
|
+
}
|
|
810
|
+
else if (detail.phase === "stop") {
|
|
811
|
+
closeThinking(controller, detail.index);
|
|
812
|
+
}
|
|
813
|
+
}
|
|
814
|
+
return carriedText;
|
|
815
|
+
};
|
|
816
|
+
const process = (controller, chunk) => {
|
|
817
|
+
if (chunk.error !== undefined && chunk.error !== null) {
|
|
818
|
+
finalizeUpstreamError(controller, chunk.error);
|
|
819
|
+
return;
|
|
820
|
+
}
|
|
821
|
+
const choice = chunk.choices?.[0];
|
|
822
|
+
if (choice === undefined) {
|
|
823
|
+
if (chunk.usage?.prompt_tokens !== undefined)
|
|
824
|
+
state.inputTokens = chunk.usage.prompt_tokens;
|
|
825
|
+
if (chunk.usage?.completion_tokens !== undefined)
|
|
826
|
+
state.outputTokens = chunk.usage.completion_tokens;
|
|
827
|
+
return;
|
|
828
|
+
}
|
|
829
|
+
const delta = choice.delta ?? {};
|
|
830
|
+
const nativeDetails = anthropicReasoningDetailsOf(delta.reasoning_details, "stream");
|
|
831
|
+
const nativeCarriedText = nativeDetails.length > 0 &&
|
|
832
|
+
handleReasoningDetails(controller, nativeDetails);
|
|
833
|
+
if (state.pendingNarration.length > 0 &&
|
|
834
|
+
(!state.thinkingOpen || state.thinkingSourceIndex === undefined)) {
|
|
835
|
+
flushPendingNarration(controller);
|
|
836
|
+
}
|
|
837
|
+
if (typeof delta.reasoning_content === "string" &&
|
|
838
|
+
delta.reasoning_content.length > 0 &&
|
|
839
|
+
!state.outputStarted) {
|
|
840
|
+
if (state.thinkingOpen && state.thinkingSourceIndex !== undefined) {
|
|
841
|
+
// Never contaminate provider-signed thinking with gateway narration:
|
|
842
|
+
// the signature must continue to describe exactly the native text.
|
|
843
|
+
state.pendingNarration.push(delta.reasoning_content);
|
|
844
|
+
}
|
|
845
|
+
else {
|
|
846
|
+
emitNarration(controller, delta.reasoning_content);
|
|
847
|
+
}
|
|
848
|
+
}
|
|
849
|
+
if (!nativeCarriedText &&
|
|
850
|
+
typeof delta.reasoning === "string" &&
|
|
851
|
+
delta.reasoning.length > 0 &&
|
|
852
|
+
!state.outputStarted) {
|
|
853
|
+
// Raw model thinking tokens pass through verbatim: they are already
|
|
854
|
+
// plain text, and Anthropic thinking blocks stream token deltas natively.
|
|
855
|
+
ensureThinking(controller);
|
|
856
|
+
controller.enqueue(sse("content_block_delta", {
|
|
857
|
+
type: "content_block_delta",
|
|
858
|
+
index: state.thinkingIndex,
|
|
859
|
+
delta: { type: "thinking_delta", thinking: delta.reasoning }
|
|
860
|
+
}));
|
|
861
|
+
}
|
|
862
|
+
if (typeof delta.content === "string" && delta.content.length > 0) {
|
|
863
|
+
ensureText(controller);
|
|
864
|
+
controller.enqueue(sse("content_block_delta", {
|
|
865
|
+
type: "content_block_delta",
|
|
866
|
+
index: state.textIndex,
|
|
867
|
+
delta: { type: "text_delta", text: delta.content }
|
|
868
|
+
}));
|
|
869
|
+
}
|
|
870
|
+
if (Array.isArray(delta.tool_calls)) {
|
|
871
|
+
for (const call of delta.tool_calls) {
|
|
872
|
+
const indexKey = typeof call.index === "number" ? call.index : undefined;
|
|
873
|
+
const idKey = typeof call.id === "string" && call.id.length > 0 ? call.id : undefined;
|
|
874
|
+
let block = indexKey !== undefined
|
|
875
|
+
? toolBlockByIndex.get(indexKey)
|
|
876
|
+
: idKey !== undefined
|
|
877
|
+
? toolBlockById.get(idKey)
|
|
878
|
+
: lastToolBlock;
|
|
879
|
+
if (block === undefined) {
|
|
880
|
+
ensureStarted(controller);
|
|
881
|
+
closeThinking(controller);
|
|
882
|
+
flushPendingNarration(controller);
|
|
883
|
+
closeThinking(controller);
|
|
884
|
+
state.outputStarted = true;
|
|
885
|
+
block = state.nextIndex++;
|
|
886
|
+
toolBlocks.push(block);
|
|
887
|
+
controller.enqueue(sse("content_block_start", {
|
|
888
|
+
type: "content_block_start",
|
|
889
|
+
index: block,
|
|
890
|
+
content_block: {
|
|
891
|
+
type: "tool_use",
|
|
892
|
+
id: call.id ?? `toolu_${randomId()}`,
|
|
893
|
+
name: call.function?.name ?? "",
|
|
894
|
+
input: {}
|
|
895
|
+
}
|
|
896
|
+
}));
|
|
897
|
+
}
|
|
898
|
+
if (indexKey !== undefined && !toolBlockByIndex.has(indexKey))
|
|
899
|
+
toolBlockByIndex.set(indexKey, block);
|
|
900
|
+
if (idKey !== undefined && !toolBlockById.has(idKey))
|
|
901
|
+
toolBlockById.set(idKey, block);
|
|
902
|
+
lastToolBlock = block;
|
|
903
|
+
const args = call.function?.arguments;
|
|
904
|
+
if (typeof args === "string" && args.length > 0) {
|
|
905
|
+
controller.enqueue(sse("content_block_delta", {
|
|
906
|
+
type: "content_block_delta",
|
|
907
|
+
index: block,
|
|
908
|
+
delta: { type: "input_json_delta", partial_json: args }
|
|
909
|
+
}));
|
|
910
|
+
}
|
|
911
|
+
}
|
|
912
|
+
}
|
|
913
|
+
if (chunk.usage?.prompt_tokens !== undefined)
|
|
914
|
+
state.inputTokens = chunk.usage.prompt_tokens;
|
|
915
|
+
if (chunk.usage?.completion_tokens !== undefined)
|
|
916
|
+
state.outputTokens = chunk.usage.completion_tokens;
|
|
917
|
+
if (choice.finish_reason !== null && choice.finish_reason !== undefined) {
|
|
918
|
+
finalize(controller, typeof choice.anthropic_stop_reason === "string"
|
|
919
|
+
? choice.anthropic_stop_reason
|
|
920
|
+
: mapStopReason(choice.finish_reason), typeof choice.anthropic_stop_sequence === "string"
|
|
921
|
+
? choice.anthropic_stop_sequence
|
|
922
|
+
: null);
|
|
923
|
+
}
|
|
924
|
+
};
|
|
925
|
+
// Backpressure handshake: the pump awaits `resumePull` whenever the consumer's
|
|
926
|
+
// desired size drops to zero, and `pull` resolves it. This replaces the old
|
|
927
|
+
// "return when desiredSize changed" hack with an explicit pump that reads the
|
|
928
|
+
// upstream reader to completion while honoring backpressure.
|
|
929
|
+
let resumePull;
|
|
930
|
+
const awaitPull = () => new Promise((resolve) => {
|
|
931
|
+
resumePull = resolve;
|
|
932
|
+
});
|
|
933
|
+
const handleEvent = (controller, data) => {
|
|
934
|
+
if (data.length === 0)
|
|
935
|
+
return;
|
|
936
|
+
if (data === "[DONE]") {
|
|
937
|
+
// A `[DONE]` without a prior finish_reason is truncation, not a clean stop.
|
|
938
|
+
if (!state.finished)
|
|
939
|
+
finalizeTruncated(controller, "upstream sent [DONE] before a finish reason");
|
|
940
|
+
return;
|
|
941
|
+
}
|
|
942
|
+
let chunk;
|
|
943
|
+
try {
|
|
944
|
+
chunk = JSON.parse(data);
|
|
945
|
+
}
|
|
946
|
+
catch (error) {
|
|
947
|
+
// The live upstream stream is authoritative: a malformed payload is a
|
|
948
|
+
// stream error, never silently skipped (WS5). Surface it and stop.
|
|
949
|
+
const detail = error instanceof Error ? error.message : String(error);
|
|
950
|
+
throw new SseParseError(`malformed OpenAI SSE payload in Anthropic translation: ${detail}`, data.slice(0, 200));
|
|
951
|
+
}
|
|
952
|
+
const marker = serverToolMarkerOf(chunk);
|
|
953
|
+
if (marker !== undefined) {
|
|
954
|
+
handleServerToolMarker(controller, marker);
|
|
955
|
+
return;
|
|
956
|
+
}
|
|
957
|
+
process(controller, chunk);
|
|
958
|
+
};
|
|
959
|
+
const pump = async (controller) => {
|
|
960
|
+
try {
|
|
961
|
+
for (;;) {
|
|
962
|
+
if ((controller.desiredSize ?? 1) <= 0)
|
|
963
|
+
await awaitPull();
|
|
964
|
+
const { done, value } = await reader.read();
|
|
965
|
+
if (done) {
|
|
966
|
+
for (const event of sseDecoder.flush())
|
|
967
|
+
handleEvent(controller, event.data);
|
|
968
|
+
// Upstream closed with no finish_reason: incomplete, not `end_turn`.
|
|
969
|
+
if (!state.finished)
|
|
970
|
+
finalizeTruncated(controller, "upstream stream ended before a finish reason");
|
|
971
|
+
controller.close();
|
|
972
|
+
return;
|
|
973
|
+
}
|
|
974
|
+
if (value !== undefined) {
|
|
975
|
+
for (const event of sseDecoder.feed(value))
|
|
976
|
+
handleEvent(controller, event.data);
|
|
977
|
+
}
|
|
978
|
+
}
|
|
979
|
+
}
|
|
980
|
+
catch (error) {
|
|
981
|
+
if (state.keepaliveTimer !== undefined)
|
|
982
|
+
clearInterval(state.keepaliveTimer);
|
|
983
|
+
controller.error(error);
|
|
984
|
+
void reader.cancel(error).catch(() => undefined);
|
|
985
|
+
}
|
|
986
|
+
};
|
|
987
|
+
return new ReadableStream({
|
|
988
|
+
start(controller) {
|
|
989
|
+
// Start the message immediately and keep the connection alive with `ping`
|
|
990
|
+
// events while the upstream is still producing its first token. Claude
|
|
991
|
+
// Code times out if it sees nothing during a slow upstream phase (the
|
|
992
|
+
// chat-layer keepalive comments are dropped by this translator, so this
|
|
993
|
+
// ping is the single keepalive that reaches the client).
|
|
994
|
+
ensureStarted(controller);
|
|
995
|
+
state.keepaliveTimer = setInterval(() => {
|
|
996
|
+
if (state.finished)
|
|
997
|
+
return;
|
|
998
|
+
// Honor backpressure: skip the ping if the consumer's queue is full.
|
|
999
|
+
if ((controller.desiredSize ?? 1) <= 0)
|
|
1000
|
+
return;
|
|
1001
|
+
try {
|
|
1002
|
+
controller.enqueue(sse("ping", { type: "ping" }));
|
|
1003
|
+
}
|
|
1004
|
+
catch {
|
|
1005
|
+
// controller closed
|
|
1006
|
+
}
|
|
1007
|
+
}, 3000);
|
|
1008
|
+
void pump(controller);
|
|
1009
|
+
},
|
|
1010
|
+
pull() {
|
|
1011
|
+
resumePull?.();
|
|
1012
|
+
resumePull = undefined;
|
|
1013
|
+
},
|
|
1014
|
+
cancel(reason) {
|
|
1015
|
+
if (state.keepaliveTimer !== undefined)
|
|
1016
|
+
clearInterval(state.keepaliveTimer);
|
|
1017
|
+
resumePull?.();
|
|
1018
|
+
resumePull = undefined;
|
|
1019
|
+
return reader.cancel(reason);
|
|
1020
|
+
}
|
|
1021
|
+
});
|
|
1022
|
+
}
|
|
1023
|
+
// ---- token counting + discovery ----
|
|
1024
|
+
export function countTokensEstimate(body) {
|
|
1025
|
+
const parts = [systemText(body.system)];
|
|
1026
|
+
for (const message of body.messages)
|
|
1027
|
+
parts.push(blockText(message.content));
|
|
1028
|
+
return estimateTokens(...parts);
|
|
1029
|
+
}
|
|
1030
|
+
// ---- handlers (return a Response the server pipes) ----
|
|
1031
|
+
function jsonResponse(status, value) {
|
|
1032
|
+
return new Response(JSON.stringify(value), {
|
|
1033
|
+
status,
|
|
1034
|
+
headers: { "content-type": "application/json" }
|
|
1035
|
+
});
|
|
1036
|
+
}
|
|
1037
|
+
export async function handleAnthropicMessages(backend, body, modelCallId, signal, backendOptions = {}) {
|
|
1038
|
+
const invalidThinking = thinkingValidationError(body);
|
|
1039
|
+
if (invalidThinking !== undefined) {
|
|
1040
|
+
return jsonResponse(400, {
|
|
1041
|
+
type: "error",
|
|
1042
|
+
error: { type: "invalid_request_error", message: invalidThinking }
|
|
1043
|
+
});
|
|
1044
|
+
}
|
|
1045
|
+
const requestedModel = body.model ?? backend.defaultModel ?? "";
|
|
1046
|
+
const resolvedModel = backend.resolveModel?.(body.model);
|
|
1047
|
+
if (body.model !== undefined &&
|
|
1048
|
+
backend.resolveModel !== undefined &&
|
|
1049
|
+
resolvedModel === undefined) {
|
|
1050
|
+
return jsonResponse(400, {
|
|
1051
|
+
type: "error",
|
|
1052
|
+
error: {
|
|
1053
|
+
type: "invalid_request_error",
|
|
1054
|
+
message: `unknown model: ${body.model}`
|
|
1055
|
+
}
|
|
1056
|
+
});
|
|
1057
|
+
}
|
|
1058
|
+
const upstreamModel = resolvedModel ?? backend.defaultModel;
|
|
1059
|
+
if (upstreamModel === undefined) {
|
|
1060
|
+
return jsonResponse(503, {
|
|
1061
|
+
type: "error",
|
|
1062
|
+
error: {
|
|
1063
|
+
type: "unavailable",
|
|
1064
|
+
message: "no model is available; configure a provider"
|
|
1065
|
+
}
|
|
1066
|
+
});
|
|
1067
|
+
}
|
|
1068
|
+
// Server-executed web search is honored when the caller declared the server
|
|
1069
|
+
// tool, an executor is available, and no *client* tool already owns the
|
|
1070
|
+
// projected name (a client `web_search` must keep round-tripping untouched).
|
|
1071
|
+
const declaresWebSearch = body.tools?.some(isAnthropicWebSearchTool) === true;
|
|
1072
|
+
const clientNameCollision = body.tools?.some((tool) => !isAnthropicServerTool(tool) && tool.name === WEB_SEARCH_TOOL_NAME) === true;
|
|
1073
|
+
const executor = declaresWebSearch && !clientNameCollision ? resolveWebSearchExecutor("anthropic") : undefined;
|
|
1074
|
+
const serverTools = executor !== undefined;
|
|
1075
|
+
const chat = anthropicToChat(body, upstreamModel, { serverTools });
|
|
1076
|
+
const requestOptions = {
|
|
1077
|
+
...backendOptions,
|
|
1078
|
+
modelCallId,
|
|
1079
|
+
// The streamed response is translated to Anthropic SSE by
|
|
1080
|
+
// openAiSseToAnthropic, which emits its own `ping` keepalive.
|
|
1081
|
+
...(body.stream === true ? { translated: true } : {})
|
|
1082
|
+
};
|
|
1083
|
+
const upstream = await backend.chat(chat, signal, requestOptions);
|
|
1084
|
+
if (!upstream.ok) {
|
|
1085
|
+
const detail = await upstream.text();
|
|
1086
|
+
return jsonResponse(upstream.status, { type: "error", error: unwrapUpstreamError(detail) });
|
|
1087
|
+
}
|
|
1088
|
+
if (executor !== undefined) {
|
|
1089
|
+
const loopOptions = {
|
|
1090
|
+
chat,
|
|
1091
|
+
runStep: (stepChat) => backend.chat(stepChat, signal, requestOptions),
|
|
1092
|
+
serverToolNames: new Set([WEB_SEARCH_TOOL_NAME]),
|
|
1093
|
+
executor,
|
|
1094
|
+
...(signal !== undefined ? { signal } : {})
|
|
1095
|
+
};
|
|
1096
|
+
if (body.stream === true) {
|
|
1097
|
+
const source = upstream.body;
|
|
1098
|
+
if (source === null)
|
|
1099
|
+
return jsonResponse(502, { type: "error", error: { type: "api_error", message: "no upstream stream" } });
|
|
1100
|
+
const composed = composeServerToolStream({ ...loopOptions, firstStep: upstream });
|
|
1101
|
+
return new Response(openAiSseToAnthropic(composed, requestedModel), {
|
|
1102
|
+
status: 200,
|
|
1103
|
+
headers: { "content-type": "text/event-stream", "cache-control": "no-cache" }
|
|
1104
|
+
});
|
|
1105
|
+
}
|
|
1106
|
+
const outcome = await runBufferedServerToolLoop({ ...loopOptions, firstStep: upstream });
|
|
1107
|
+
if (outcome.kind === "upstream_error") {
|
|
1108
|
+
const detail = await outcome.response.text();
|
|
1109
|
+
return jsonResponse(outcome.response.status, {
|
|
1110
|
+
type: "error",
|
|
1111
|
+
error: { type: "api_error", message: detail.slice(0, 2000) }
|
|
1112
|
+
});
|
|
1113
|
+
}
|
|
1114
|
+
return jsonResponse(200, chatToAnthropicMessage(outcome.openai, requestedModel, outcome.searches, outcome.events));
|
|
1115
|
+
}
|
|
1116
|
+
if (body.stream === true) {
|
|
1117
|
+
const source = upstream.body;
|
|
1118
|
+
if (source === null)
|
|
1119
|
+
return jsonResponse(502, { type: "error", error: { type: "api_error", message: "no upstream stream" } });
|
|
1120
|
+
return new Response(openAiSseToAnthropic(source, requestedModel), {
|
|
1121
|
+
status: 200,
|
|
1122
|
+
headers: { "content-type": "text/event-stream", "cache-control": "no-cache" }
|
|
1123
|
+
});
|
|
1124
|
+
}
|
|
1125
|
+
const openai = (await upstream.json());
|
|
1126
|
+
return jsonResponse(200, chatToAnthropicMessage(openai, requestedModel));
|
|
1127
|
+
}
|
|
1128
|
+
export function handleCountTokens(body) {
|
|
1129
|
+
return jsonResponse(200, { input_tokens: countTokensEstimate(body) });
|
|
1130
|
+
}
|
|
1131
|
+
/** Claude Code only lists models whose id begins with `claude` or `anthropic`. */
|
|
1132
|
+
function isAnthropicFamilyId(id) {
|
|
1133
|
+
return id.startsWith("claude") || id.startsWith("anthropic");
|
|
1134
|
+
}
|
|
1135
|
+
/** The `claude-` prefix used to alias non-Anthropic models past Claude's filter. */
|
|
1136
|
+
export const CLAUDE_ALIAS_PREFIX = "claude-";
|
|
1137
|
+
/**
|
|
1138
|
+
* The id a model is advertised under in Claude Code's `/model` picker. Claude
|
|
1139
|
+
* only lists ids beginning with `claude`/`anthropic`, so non-Anthropic models
|
|
1140
|
+
* are aliased with a `claude-` prefix; the gateway maps the alias back when
|
|
1141
|
+
* routing (see `resolveAlias`), and the picker shows the real id via
|
|
1142
|
+
* `display_name`. This is the claude-code-router trick: the `model` field is an
|
|
1143
|
+
* identifier we control end-to-end, so any model can be made selectable.
|
|
1144
|
+
*/
|
|
1145
|
+
export function claudeModelAlias(id) {
|
|
1146
|
+
return isAnthropicFamilyId(id) ? id : `${CLAUDE_ALIAS_PREFIX}${id}`;
|
|
1147
|
+
}
|
|
1148
|
+
export function resolveClaudeModelAlias(requested, modelIds = []) {
|
|
1149
|
+
if (requested === undefined || modelIds.includes(requested))
|
|
1150
|
+
return requested;
|
|
1151
|
+
if (!requested.startsWith(CLAUDE_ALIAS_PREFIX))
|
|
1152
|
+
return requested;
|
|
1153
|
+
const candidate = requested.slice(CLAUDE_ALIAS_PREFIX.length);
|
|
1154
|
+
return modelIds.includes(candidate) && claudeModelAlias(candidate) === requested
|
|
1155
|
+
? candidate
|
|
1156
|
+
: requested;
|
|
1157
|
+
}
|
|
1158
|
+
/**
|
|
1159
|
+
* Anthropic-shaped `/v1/models` discovery response. Every advertised model is
|
|
1160
|
+
* listed so it appears in Claude Code's `/model` picker: Anthropic-family ids
|
|
1161
|
+
* as-is, others under a `claude-`prefixed alias with the real id as
|
|
1162
|
+
* `display_name`. `modelIds` is the full advertised set (default model first);
|
|
1163
|
+
* when absent we fall back to the single backend default.
|
|
1164
|
+
*/
|
|
1165
|
+
export function anthropicModelsResponse(backendModel, modelIds, modelRoutes = []) {
|
|
1166
|
+
const source = modelIds !== undefined && modelIds.length > 0
|
|
1167
|
+
? modelIds
|
|
1168
|
+
: backendModel !== undefined
|
|
1169
|
+
? [backendModel]
|
|
1170
|
+
: [];
|
|
1171
|
+
const seen = new Set();
|
|
1172
|
+
const routes = new Map(modelRoutes.map((route) => [route.publicId, route]));
|
|
1173
|
+
const models = [];
|
|
1174
|
+
for (const realId of source) {
|
|
1175
|
+
const route = routes.get(realId);
|
|
1176
|
+
const displayName = route?.provider === "claude-code" ? route.nativeId : realId;
|
|
1177
|
+
const id = claudeModelAlias(displayName);
|
|
1178
|
+
if (seen.has(id))
|
|
1179
|
+
continue;
|
|
1180
|
+
seen.add(id);
|
|
1181
|
+
models.push({
|
|
1182
|
+
type: "model",
|
|
1183
|
+
id,
|
|
1184
|
+
display_name: displayName,
|
|
1185
|
+
created_at: new Date(0).toISOString()
|
|
1186
|
+
});
|
|
1187
|
+
}
|
|
1188
|
+
const ids = models.map((model) => model.id);
|
|
1189
|
+
return new Response(JSON.stringify({
|
|
1190
|
+
data: models,
|
|
1191
|
+
has_more: false,
|
|
1192
|
+
first_id: ids[0],
|
|
1193
|
+
last_id: ids[ids.length - 1]
|
|
1194
|
+
}), { status: 200, headers: { "content-type": "application/json" } });
|
|
1195
|
+
}
|