@sriinnu/kosha-discovery 1.2.0 → 1.5.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +125 -52
- package/dist/aliases.d.ts +6 -1
- package/dist/aliases.d.ts.map +1 -1
- package/dist/aliases.js +38 -12
- package/dist/aliases.js.map +1 -1
- package/dist/cache.d.ts.map +1 -1
- package/dist/cache.js +40 -3
- package/dist/cache.js.map +1 -1
- package/dist/claude-generation.d.ts +59 -0
- package/dist/claude-generation.d.ts.map +1 -0
- package/dist/claude-generation.js +117 -0
- package/dist/claude-generation.js.map +1 -0
- package/dist/cli-cmd-doctor.d.ts +17 -0
- package/dist/cli-cmd-doctor.d.ts.map +1 -0
- package/dist/cli-cmd-doctor.js +199 -0
- package/dist/cli-cmd-doctor.js.map +1 -0
- package/dist/cli-cmd-model.js +2 -2
- package/dist/cli-cmd-model.js.map +1 -1
- package/dist/cli-cmd-spend.d.ts +10 -0
- package/dist/cli-cmd-spend.d.ts.map +1 -0
- package/dist/cli-cmd-spend.js +130 -0
- package/dist/cli-cmd-spend.js.map +1 -0
- package/dist/cli-commands.d.ts +4 -1
- package/dist/cli-commands.d.ts.map +1 -1
- package/dist/cli-commands.js +10 -2
- package/dist/cli-commands.js.map +1 -1
- package/dist/cli-format.js.map +1 -1
- package/dist/cli-help.d.ts.map +1 -1
- package/dist/cli-help.js +13 -2
- package/dist/cli-help.js.map +1 -1
- package/dist/cli.js +10 -0
- package/dist/cli.js.map +1 -1
- package/dist/cost.d.ts +152 -0
- package/dist/cost.d.ts.map +1 -0
- package/dist/cost.js +377 -0
- package/dist/cost.js.map +1 -0
- package/dist/credentials/resolver.d.ts +10 -0
- package/dist/credentials/resolver.d.ts.map +1 -1
- package/dist/credentials/resolver.js +64 -31
- package/dist/credentials/resolver.js.map +1 -1
- package/dist/discovery/anthropic.d.ts +19 -1
- package/dist/discovery/anthropic.d.ts.map +1 -1
- package/dist/discovery/anthropic.js +112 -8
- package/dist/discovery/anthropic.js.map +1 -1
- package/dist/discovery/base.d.ts +7 -0
- package/dist/discovery/base.d.ts.map +1 -1
- package/dist/discovery/base.js +32 -4
- package/dist/discovery/base.js.map +1 -1
- package/dist/discovery/bedrock.d.ts.map +1 -1
- package/dist/discovery/cerebras.d.ts.map +1 -1
- package/dist/discovery/cohere.d.ts.map +1 -1
- package/dist/discovery/deepinfra.d.ts.map +1 -1
- package/dist/discovery/deepseek.d.ts.map +1 -1
- package/dist/discovery/fireworks.d.ts.map +1 -1
- package/dist/discovery/glm.d.ts.map +1 -1
- package/dist/discovery/google.d.ts +3 -1
- package/dist/discovery/google.d.ts.map +1 -1
- package/dist/discovery/google.js +13 -6
- package/dist/discovery/google.js.map +1 -1
- package/dist/discovery/groq.d.ts.map +1 -1
- package/dist/discovery/index.d.ts +2 -0
- package/dist/discovery/index.d.ts.map +1 -1
- package/dist/discovery/index.js +6 -0
- package/dist/discovery/index.js.map +1 -1
- package/dist/discovery/llama-cpp.d.ts.map +1 -1
- package/dist/discovery/lmstudio.d.ts +26 -0
- package/dist/discovery/lmstudio.d.ts.map +1 -0
- package/dist/discovery/lmstudio.js +93 -0
- package/dist/discovery/lmstudio.js.map +1 -0
- package/dist/discovery/minimax.d.ts.map +1 -1
- package/dist/discovery/mistral.d.ts.map +1 -1
- package/dist/discovery/moonshot.d.ts.map +1 -1
- package/dist/discovery/nvidia.d.ts.map +1 -1
- package/dist/discovery/ollama.d.ts.map +1 -1
- package/dist/discovery/openai-compatible.d.ts.map +1 -1
- package/dist/discovery/openai.d.ts.map +1 -1
- package/dist/discovery/openrouter.d.ts +29 -5
- package/dist/discovery/openrouter.d.ts.map +1 -1
- package/dist/discovery/openrouter.js +56 -18
- package/dist/discovery/openrouter.js.map +1 -1
- package/dist/discovery/perplexity.d.ts.map +1 -1
- package/dist/discovery/promo-overrides.js.map +1 -1
- package/dist/discovery/static-direct.d.ts +9 -1
- package/dist/discovery/static-direct.d.ts.map +1 -1
- package/dist/discovery/static-direct.js +68 -7
- package/dist/discovery/static-direct.js.map +1 -1
- package/dist/discovery/together.d.ts.map +1 -1
- package/dist/discovery/vercel.d.ts.map +1 -1
- package/dist/discovery/vertex.d.ts.map +1 -1
- package/dist/discovery/vllm.d.ts +26 -0
- package/dist/discovery/vllm.d.ts.map +1 -0
- package/dist/discovery/vllm.js +93 -0
- package/dist/discovery/vllm.js.map +1 -0
- package/dist/discovery/zai.d.ts.map +1 -1
- package/dist/discovery-contract.d.ts +7 -0
- package/dist/discovery-contract.d.ts.map +1 -1
- package/dist/discovery-contract.js.map +1 -1
- package/dist/discovery-routes.d.ts +9 -0
- package/dist/discovery-routes.d.ts.map +1 -1
- package/dist/discovery-routes.js +106 -18
- package/dist/discovery-routes.js.map +1 -1
- package/dist/enrichment/litellm.d.ts +1 -1
- package/dist/enrichment/litellm.d.ts.map +1 -1
- package/dist/enrichment/litellm.js +3 -3
- package/dist/entry.d.ts +14 -0
- package/dist/entry.d.ts.map +1 -0
- package/dist/entry.js +26 -0
- package/dist/entry.js.map +1 -0
- package/dist/index.d.ts +10 -0
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +5 -0
- package/dist/index.js.map +1 -1
- package/dist/mcp-server.d.ts +207 -2
- package/dist/mcp-server.d.ts.map +1 -1
- package/dist/mcp-server.js +203 -51
- package/dist/mcp-server.js.map +1 -1
- package/dist/model-features.d.ts.map +1 -1
- package/dist/model-features.js +11 -2
- package/dist/model-features.js.map +1 -1
- package/dist/normalize.d.ts +7 -7
- package/dist/normalize.js +8 -8
- package/dist/provider-catalog.d.ts.map +1 -1
- package/dist/provider-catalog.js +26 -0
- package/dist/provider-catalog.js.map +1 -1
- package/dist/proxy.d.ts +49 -7
- package/dist/proxy.d.ts.map +1 -1
- package/dist/proxy.js +754 -89
- package/dist/proxy.js.map +1 -1
- package/dist/registry-query.js +2 -2
- package/dist/registry-query.js.map +1 -1
- package/dist/registry-routing.d.ts +65 -0
- package/dist/registry-routing.d.ts.map +1 -0
- package/dist/registry-routing.js +165 -0
- package/dist/registry-routing.js.map +1 -0
- package/dist/registry-runtime.d.ts +2 -0
- package/dist/registry-runtime.d.ts.map +1 -1
- package/dist/registry-runtime.js +177 -23
- package/dist/registry-runtime.js.map +1 -1
- package/dist/registry-selection.js +6 -0
- package/dist/registry-selection.js.map +1 -1
- package/dist/registry.d.ts +69 -0
- package/dist/registry.d.ts.map +1 -1
- package/dist/registry.js +133 -1
- package/dist/registry.js.map +1 -1
- package/dist/resilience.d.ts +15 -0
- package/dist/resilience.d.ts.map +1 -1
- package/dist/resilience.js +23 -1
- package/dist/resilience.js.map +1 -1
- package/dist/server.d.ts +44 -2
- package/dist/server.d.ts.map +1 -1
- package/dist/server.js +328 -33
- package/dist/server.js.map +1 -1
- package/dist/tally.d.ts +80 -0
- package/dist/tally.d.ts.map +1 -0
- package/dist/tally.js +176 -0
- package/dist/tally.js.map +1 -0
- package/dist/types.d.ts +22 -1
- package/dist/types.d.ts.map +1 -1
- package/dist/wire-anthropic.d.ts +264 -0
- package/dist/wire-anthropic.d.ts.map +1 -0
- package/dist/wire-anthropic.js +960 -0
- package/dist/wire-anthropic.js.map +1 -0
- package/package.json +14 -10
- package/logo.png +0 -0
|
@@ -0,0 +1,960 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* kosha-discovery — OpenAI ↔ Anthropic wire-format translator.
|
|
3
|
+
*
|
|
4
|
+
* The proxy accepts an OpenAI chat/completions request; Anthropic speaks
|
|
5
|
+
* `/v1/messages`. This module bridges the two in both directions:
|
|
6
|
+
*
|
|
7
|
+
* - **request** system → top-level `system`; `tools` / `tool_choice` →
|
|
8
|
+
* Anthropic tools; assistant `tool_calls` → `tool_use` blocks; `tool`
|
|
9
|
+
* role → `tool_result` blocks; `image_url` parts → `image` blocks;
|
|
10
|
+
* `response_format: json_schema` → `output_config.format`;
|
|
11
|
+
* `reasoning_effort` → `output_config.effort`; sampling parameters are
|
|
12
|
+
* dropped on Claude generations that reject them.
|
|
13
|
+
* - **response** text + `tool_use` blocks → `message.content` +
|
|
14
|
+
* `tool_calls`; Anthropic usage (cache fields included) → OpenAI usage.
|
|
15
|
+
* - **stream** Anthropic SSE events → OpenAI `chat.completion.chunk` SSE,
|
|
16
|
+
* with the final usage exposed to the caller for ledger reconciliation.
|
|
17
|
+
*
|
|
18
|
+
* Still unsupported — these throw {@link UnsupportedWireContentError} so the
|
|
19
|
+
* proxy fails over to a native OpenAI-compatible route instead of shipping a
|
|
20
|
+
* silently-mangled request: audio / file input parts, non-`function` tool
|
|
21
|
+
* types, and `response_format: json_schema` on Claude generations without
|
|
22
|
+
* native structured outputs. Fields with no Anthropic equivalent (`n`,
|
|
23
|
+
* `seed`, `logit_bias`, penalties, logprobs) are dropped and reported in the
|
|
24
|
+
* translation notes; `user` maps to `metadata.user_id`.
|
|
25
|
+
*
|
|
26
|
+
* Pure functions; no I/O.
|
|
27
|
+
* @module
|
|
28
|
+
*/
|
|
29
|
+
import { CLAUDE_TEMPERATURE_MAX, claudeEffortLadder, claudeSamplingSupport, claudeSupportsForcedToolChoice, claudeSupportsNativeJsonSchema, claudeSupportsPrefill, } from "./claude-generation.js";
|
|
30
|
+
/** Anthropic requires max_tokens; we cap at this when the caller didn't supply one. */
|
|
31
|
+
const DEFAULT_MAX_TOKENS = 4_096;
|
|
32
|
+
/**
|
|
33
|
+
* Anthropic rejects empty message content, so when the conversation has no
|
|
34
|
+
* user turn to open with (system-only request, assistant-first history) we
|
|
35
|
+
* synthesize a minimal one.
|
|
36
|
+
*/
|
|
37
|
+
const PLACEHOLDER_USER_TEXT = "Continue.";
|
|
38
|
+
/** Image media types Anthropic accepts as base64 / URL image blocks. */
|
|
39
|
+
const IMAGE_MEDIA_TYPES = new Set(["image/jpeg", "image/png", "image/gif", "image/webp"]);
|
|
40
|
+
/**
|
|
41
|
+
* Raised when an OpenAI chat-completions body carries wire features the
|
|
42
|
+
* Anthropic translator cannot faithfully carry. The proxy catches this by
|
|
43
|
+
* class and falls back to a native route rather than shipping a
|
|
44
|
+
* silently-mangled request. Throwing beats silent data loss on the path
|
|
45
|
+
* kosha:cheapest[…] resolves through here.
|
|
46
|
+
*/
|
|
47
|
+
export class UnsupportedWireContentError extends Error {
|
|
48
|
+
constructor(message) {
|
|
49
|
+
super(message);
|
|
50
|
+
this.name = "UnsupportedWireContentError";
|
|
51
|
+
}
|
|
52
|
+
}
|
|
53
|
+
// ---------------------------------------------------------------------------
|
|
54
|
+
// Per-generation Claude behaviour (shared with model-features.ts)
|
|
55
|
+
// ---------------------------------------------------------------------------
|
|
56
|
+
export { claudeSamplingSupport, claudeSupportsForcedToolChoice, claudeSupportsNativeJsonSchema };
|
|
57
|
+
/**
|
|
58
|
+
* Notes embed caller-supplied strings (a tool name, an effort value) and are
|
|
59
|
+
* reflected into a response header, and HTTP header values must be
|
|
60
|
+
* ISO-8859-1 — undici's `Headers.set` throws on anything above U+00FF, and
|
|
61
|
+
* Node rejects other control characters at write time. Keep printable ASCII
|
|
62
|
+
* only and bound the length so a note can never make the proxy 500 after
|
|
63
|
+
* the upstream call has already been paid for.
|
|
64
|
+
*/
|
|
65
|
+
function noteToken(value, max = 64) {
|
|
66
|
+
const cleaned = value.replace(/[^\x20-\x7e]/g, "?");
|
|
67
|
+
return cleaned.length > max ? `${cleaned.slice(0, max)}…`.replace("…", "...") : cleaned;
|
|
68
|
+
}
|
|
69
|
+
/** OpenAI `reasoning_effort` vocabulary → Anthropic effort levels. */
|
|
70
|
+
const EFFORT_MAP = {
|
|
71
|
+
minimal: "low",
|
|
72
|
+
low: "low",
|
|
73
|
+
medium: "medium",
|
|
74
|
+
high: "high",
|
|
75
|
+
xhigh: "xhigh",
|
|
76
|
+
max: "max",
|
|
77
|
+
};
|
|
78
|
+
// ---------------------------------------------------------------------------
|
|
79
|
+
// Request coercion (untyped JSON → OpenAIChatRequest)
|
|
80
|
+
// ---------------------------------------------------------------------------
|
|
81
|
+
/**
|
|
82
|
+
* Narrow a parsed OpenAI chat-completions body (`Record<string, unknown>`) into
|
|
83
|
+
* a typed {@link OpenAIChatRequest} via runtime guards. This is the single
|
|
84
|
+
* untyped-JSON → typed boundary for the translator.
|
|
85
|
+
*/
|
|
86
|
+
export function coerceOpenAIChatRequest(body) {
|
|
87
|
+
const messages = [];
|
|
88
|
+
const rawMessages = Array.isArray(body.messages) ? body.messages : [];
|
|
89
|
+
for (const entry of rawMessages) {
|
|
90
|
+
if (!entry || typeof entry !== "object")
|
|
91
|
+
continue;
|
|
92
|
+
const msg = entry;
|
|
93
|
+
const role = typeof msg.role === "string" ? msg.role : "user";
|
|
94
|
+
const content = Array.isArray(msg.content)
|
|
95
|
+
? msg.content
|
|
96
|
+
: typeof msg.content === "string"
|
|
97
|
+
? msg.content
|
|
98
|
+
: null;
|
|
99
|
+
const out = { role, content };
|
|
100
|
+
if (msg.tool_calls !== undefined)
|
|
101
|
+
out.tool_calls = msg.tool_calls;
|
|
102
|
+
if (typeof msg.tool_call_id === "string")
|
|
103
|
+
out.tool_call_id = msg.tool_call_id;
|
|
104
|
+
if (typeof msg.name === "string")
|
|
105
|
+
out.name = msg.name;
|
|
106
|
+
messages.push(out);
|
|
107
|
+
}
|
|
108
|
+
const req = {
|
|
109
|
+
model: typeof body.model === "string" ? body.model : "",
|
|
110
|
+
messages,
|
|
111
|
+
};
|
|
112
|
+
if (typeof body.max_tokens === "number")
|
|
113
|
+
req.max_tokens = body.max_tokens;
|
|
114
|
+
if (typeof body.max_completion_tokens === "number")
|
|
115
|
+
req.max_completion_tokens = body.max_completion_tokens;
|
|
116
|
+
if (typeof body.temperature === "number")
|
|
117
|
+
req.temperature = body.temperature;
|
|
118
|
+
if (typeof body.top_p === "number")
|
|
119
|
+
req.top_p = body.top_p;
|
|
120
|
+
if (typeof body.stop === "string")
|
|
121
|
+
req.stop = body.stop;
|
|
122
|
+
else if (Array.isArray(body.stop))
|
|
123
|
+
req.stop = body.stop.filter((s) => typeof s === "string");
|
|
124
|
+
if (body.stream === true)
|
|
125
|
+
req.stream = true;
|
|
126
|
+
if (body.stream_options && typeof body.stream_options === "object") {
|
|
127
|
+
req.stream_options = { include_usage: body.stream_options.include_usage === true };
|
|
128
|
+
}
|
|
129
|
+
if (body.tools !== undefined)
|
|
130
|
+
req.tools = body.tools;
|
|
131
|
+
if (body.tool_choice !== undefined)
|
|
132
|
+
req.tool_choice = body.tool_choice;
|
|
133
|
+
if (typeof body.parallel_tool_calls === "boolean")
|
|
134
|
+
req.parallel_tool_calls = body.parallel_tool_calls;
|
|
135
|
+
if (body.response_format !== undefined)
|
|
136
|
+
req.response_format = body.response_format;
|
|
137
|
+
if (typeof body.reasoning_effort === "string")
|
|
138
|
+
req.reasoning_effort = body.reasoning_effort;
|
|
139
|
+
if (typeof body.user === "string" && body.user.length > 0)
|
|
140
|
+
req.user = body.user;
|
|
141
|
+
const unsupported = [];
|
|
142
|
+
if (typeof body.n === "number" && body.n > 1)
|
|
143
|
+
unsupported.push("n");
|
|
144
|
+
for (const key of ["seed", "logit_bias", "presence_penalty", "frequency_penalty", "logprobs", "top_logprobs"]) {
|
|
145
|
+
if (body[key] !== undefined && body[key] !== null)
|
|
146
|
+
unsupported.push(key);
|
|
147
|
+
}
|
|
148
|
+
if (unsupported.length > 0)
|
|
149
|
+
req.unsupportedFields = unsupported;
|
|
150
|
+
return req;
|
|
151
|
+
}
|
|
152
|
+
// ---------------------------------------------------------------------------
|
|
153
|
+
// Request translation
|
|
154
|
+
// ---------------------------------------------------------------------------
|
|
155
|
+
/** Translate an OpenAI chat-completions request body into an Anthropic /v1/messages body. */
|
|
156
|
+
export function translateOpenAIToAnthropic(req) {
|
|
157
|
+
return translateOpenAIToAnthropicWithNotes(req).request;
|
|
158
|
+
}
|
|
159
|
+
/**
|
|
160
|
+
* Translate an OpenAI chat-completions request and report every lossy
|
|
161
|
+
* mapping made along the way (dropped sampling params, degraded forced tool
|
|
162
|
+
* choice, …). The proxy reflects the notes to the caller in a response
|
|
163
|
+
* header so nothing kosha changed is invisible.
|
|
164
|
+
*/
|
|
165
|
+
export function translateOpenAIToAnthropicWithNotes(req) {
|
|
166
|
+
const notes = [];
|
|
167
|
+
const systemParts = [];
|
|
168
|
+
// Tools first: response_format / tool_choice handling below depends on them.
|
|
169
|
+
const tools = req.tools !== undefined && req.tools !== null ? translateTools(req.tools) : undefined;
|
|
170
|
+
let toolChoice = translateToolChoice(req.tool_choice, req.model, tools !== undefined, notes, systemParts);
|
|
171
|
+
if (req.parallel_tool_calls === false && tools && tools.length > 0) {
|
|
172
|
+
if (!toolChoice)
|
|
173
|
+
toolChoice = { type: "auto" };
|
|
174
|
+
if (toolChoice.type !== "none")
|
|
175
|
+
toolChoice = { ...toolChoice, disable_parallel_tool_use: true };
|
|
176
|
+
}
|
|
177
|
+
const outputConfig = {};
|
|
178
|
+
const formatInstruction = applyResponseFormat(req.response_format, req.model, outputConfig, notes);
|
|
179
|
+
applyEffort(req.reasoning_effort, req.model, outputConfig, notes);
|
|
180
|
+
// JSON output cannot be combined with forced tool use; fall back to auto.
|
|
181
|
+
if (outputConfig.format && toolChoice && (toolChoice.type === "any" || toolChoice.type === "tool")) {
|
|
182
|
+
toolChoice = { type: "auto", ...(toolChoice.disable_parallel_tool_use ? { disable_parallel_tool_use: true } : {}) };
|
|
183
|
+
notes.push("tool_choice degraded to auto: forced tool use cannot be combined with response_format json_schema");
|
|
184
|
+
}
|
|
185
|
+
if (tools && !claudeSupportsNativeJsonSchema(req.model)) {
|
|
186
|
+
let stripped = 0;
|
|
187
|
+
for (const tool of tools) {
|
|
188
|
+
if (tool.strict) {
|
|
189
|
+
delete tool.strict;
|
|
190
|
+
stripped += 1;
|
|
191
|
+
}
|
|
192
|
+
}
|
|
193
|
+
if (stripped > 0)
|
|
194
|
+
notes.push(`dropped strict on ${stripped} tool(s): ${req.model} predates strict tool use`);
|
|
195
|
+
}
|
|
196
|
+
const messages = [];
|
|
197
|
+
for (const msg of req.messages ?? []) {
|
|
198
|
+
switch (msg.role) {
|
|
199
|
+
case "system":
|
|
200
|
+
case "developer": {
|
|
201
|
+
const text = flattenText(msg.content);
|
|
202
|
+
if (text)
|
|
203
|
+
systemParts.push(text);
|
|
204
|
+
break;
|
|
205
|
+
}
|
|
206
|
+
case "tool": {
|
|
207
|
+
if (!msg.tool_call_id) {
|
|
208
|
+
throw new UnsupportedWireContentError("tool message is missing tool_call_id");
|
|
209
|
+
}
|
|
210
|
+
const text = flattenText(msg.content);
|
|
211
|
+
const block = { type: "tool_result", tool_use_id: msg.tool_call_id };
|
|
212
|
+
if (text)
|
|
213
|
+
block.content = text;
|
|
214
|
+
messages.push({ role: "user", content: [block] });
|
|
215
|
+
break;
|
|
216
|
+
}
|
|
217
|
+
case "assistant": {
|
|
218
|
+
const blocks = assistantBlocks(msg);
|
|
219
|
+
if (blocks.length === 0)
|
|
220
|
+
break;
|
|
221
|
+
messages.push({ role: "assistant", content: collapseTextOnly(blocks) });
|
|
222
|
+
break;
|
|
223
|
+
}
|
|
224
|
+
default: {
|
|
225
|
+
const blocks = userBlocks(msg.content);
|
|
226
|
+
if (blocks.length === 0)
|
|
227
|
+
break;
|
|
228
|
+
messages.push({ role: "user", content: collapseTextOnly(blocks) });
|
|
229
|
+
}
|
|
230
|
+
}
|
|
231
|
+
}
|
|
232
|
+
// Anthropic's /v1/messages requires the conversation to start with a user
|
|
233
|
+
// message — if the caller only sent a system prompt we synthesize one.
|
|
234
|
+
if (messages.length === 0) {
|
|
235
|
+
messages.push({ role: "user", content: PLACEHOLDER_USER_TEXT });
|
|
236
|
+
}
|
|
237
|
+
else if (messages[0].role !== "user") {
|
|
238
|
+
messages.unshift({ role: "user", content: PLACEHOLDER_USER_TEXT });
|
|
239
|
+
}
|
|
240
|
+
// Anthropic forbids two messages with the same role in a row. The OpenAI
|
|
241
|
+
// side allows it (e.g. multiple tool-result messages), so consecutive
|
|
242
|
+
// same-role messages collapse into one.
|
|
243
|
+
const collapsed = finalizeTrailingAssistant(mergeConsecutiveRoles(messages), req.model, notes);
|
|
244
|
+
// The volatile JSON-mode instruction goes AFTER the caller's system prompt
|
|
245
|
+
// so a cached system-prompt prefix stays stable.
|
|
246
|
+
if (formatInstruction)
|
|
247
|
+
systemParts.push(formatInstruction);
|
|
248
|
+
const maxTokens = req.max_completion_tokens ?? req.max_tokens;
|
|
249
|
+
const out = {
|
|
250
|
+
model: req.model,
|
|
251
|
+
max_tokens: maxTokens && maxTokens > 0 ? Math.floor(maxTokens) : DEFAULT_MAX_TOKENS,
|
|
252
|
+
messages: collapsed,
|
|
253
|
+
};
|
|
254
|
+
if (systemParts.length > 0)
|
|
255
|
+
out.system = systemParts.join("\n\n");
|
|
256
|
+
applySampling(req, out, notes);
|
|
257
|
+
if (req.stop) {
|
|
258
|
+
const stops = (Array.isArray(req.stop) ? req.stop : [req.stop]).filter((stop) => stop.trim().length > 0);
|
|
259
|
+
if (stops.length > 0)
|
|
260
|
+
out.stop_sequences = stops;
|
|
261
|
+
else
|
|
262
|
+
notes.push("dropped stop: only empty / whitespace sequences were given");
|
|
263
|
+
}
|
|
264
|
+
if (req.stream)
|
|
265
|
+
out.stream = true;
|
|
266
|
+
if (tools && tools.length > 0)
|
|
267
|
+
out.tools = tools;
|
|
268
|
+
if (toolChoice)
|
|
269
|
+
out.tool_choice = toolChoice;
|
|
270
|
+
if (req.user)
|
|
271
|
+
out.metadata = { user_id: noteToken(req.user, 256) };
|
|
272
|
+
if (Object.keys(outputConfig).length > 0)
|
|
273
|
+
out.output_config = outputConfig;
|
|
274
|
+
if (req.unsupportedFields && req.unsupportedFields.length > 0) {
|
|
275
|
+
notes.push(`dropped ${req.unsupportedFields.join(", ")}: no Anthropic equivalent`);
|
|
276
|
+
}
|
|
277
|
+
return { request: out, notes };
|
|
278
|
+
}
|
|
279
|
+
/** Copy `temperature` / `top_p` subject to the target model's sampling rules. */
|
|
280
|
+
function applySampling(req, out, notes) {
|
|
281
|
+
const hasTemp = typeof req.temperature === "number";
|
|
282
|
+
const hasTopP = typeof req.top_p === "number";
|
|
283
|
+
if (!hasTemp && !hasTopP)
|
|
284
|
+
return;
|
|
285
|
+
const support = claudeSamplingSupport(req.model);
|
|
286
|
+
if (support === "none") {
|
|
287
|
+
const dropped = [hasTemp ? "temperature" : null, hasTopP ? "top_p" : null].filter(Boolean).join(" and ");
|
|
288
|
+
notes.push(`dropped ${dropped}: ${req.model} does not accept sampling parameters`);
|
|
289
|
+
return;
|
|
290
|
+
}
|
|
291
|
+
const temperature = hasTemp ? clampTemperature(req.temperature, notes) : undefined;
|
|
292
|
+
if (support === "one" && hasTemp && hasTopP) {
|
|
293
|
+
out.temperature = temperature;
|
|
294
|
+
notes.push(`dropped top_p: ${req.model} accepts only one of temperature / top_p; kept temperature`);
|
|
295
|
+
return;
|
|
296
|
+
}
|
|
297
|
+
if (hasTemp)
|
|
298
|
+
out.temperature = temperature;
|
|
299
|
+
if (hasTopP)
|
|
300
|
+
out.top_p = req.top_p;
|
|
301
|
+
}
|
|
302
|
+
/** OpenAI allows 0..2; Anthropic rejects anything above 1. */
|
|
303
|
+
function clampTemperature(value, notes) {
|
|
304
|
+
if (value > CLAUDE_TEMPERATURE_MAX) {
|
|
305
|
+
notes.push(`clamped temperature ${value} to ${CLAUDE_TEMPERATURE_MAX}: Anthropic's maximum`);
|
|
306
|
+
return CLAUDE_TEMPERATURE_MAX;
|
|
307
|
+
}
|
|
308
|
+
if (value < 0) {
|
|
309
|
+
notes.push(`clamped temperature ${value} to 0`);
|
|
310
|
+
return 0;
|
|
311
|
+
}
|
|
312
|
+
return value;
|
|
313
|
+
}
|
|
314
|
+
/** Map OpenAI `reasoning_effort` onto `output_config.effort`, clamped to the model's ladder. */
|
|
315
|
+
function applyEffort(effort, modelId, outputConfig, notes) {
|
|
316
|
+
if (!effort)
|
|
317
|
+
return;
|
|
318
|
+
const mapped = EFFORT_MAP[effort.toLowerCase()];
|
|
319
|
+
if (!mapped) {
|
|
320
|
+
notes.push(`dropped reasoning_effort '${noteToken(effort)}': unknown value`);
|
|
321
|
+
return;
|
|
322
|
+
}
|
|
323
|
+
const ladder = claudeEffortLadder(modelId);
|
|
324
|
+
if (ladder.length === 0) {
|
|
325
|
+
notes.push(`dropped reasoning_effort: ${modelId} does not support output_config.effort`);
|
|
326
|
+
return;
|
|
327
|
+
}
|
|
328
|
+
if (ladder.includes(mapped)) {
|
|
329
|
+
outputConfig.effort = mapped;
|
|
330
|
+
return;
|
|
331
|
+
}
|
|
332
|
+
// xhigh / max requested on a generation without that rung → nearest lower.
|
|
333
|
+
const fallback = mapped === "max" && ladder.includes("max") ? "max" : "high";
|
|
334
|
+
outputConfig.effort = fallback;
|
|
335
|
+
notes.push(`clamped reasoning_effort '${noteToken(effort)}' to '${fallback}': not available on ${modelId}`);
|
|
336
|
+
}
|
|
337
|
+
/**
|
|
338
|
+
* Translate `response_format`: json_schema → output_config.format;
|
|
339
|
+
* json_object → a system instruction, returned so the caller can append it
|
|
340
|
+
* after the user's own system prompt.
|
|
341
|
+
*/
|
|
342
|
+
function applyResponseFormat(fmt, modelId, outputConfig, notes) {
|
|
343
|
+
if (!fmt || typeof fmt !== "object")
|
|
344
|
+
return undefined;
|
|
345
|
+
const ftype = fmt.type;
|
|
346
|
+
if (ftype === "json_schema") {
|
|
347
|
+
if (!claudeSupportsNativeJsonSchema(modelId)) {
|
|
348
|
+
throw new UnsupportedWireContentError(`response_format 'json_schema' is not supported on ${modelId} (no native structured outputs)`);
|
|
349
|
+
}
|
|
350
|
+
const schema = fmt.json_schema?.schema;
|
|
351
|
+
if (!schema || typeof schema !== "object") {
|
|
352
|
+
throw new UnsupportedWireContentError("response_format json_schema is missing json_schema.schema");
|
|
353
|
+
}
|
|
354
|
+
outputConfig.format = { type: "json_schema", schema };
|
|
355
|
+
return undefined;
|
|
356
|
+
}
|
|
357
|
+
if (ftype === "json_object") {
|
|
358
|
+
// No native equivalent of OpenAI's schema-less JSON mode; the closest
|
|
359
|
+
// faithful mapping is an explicit instruction, which is how OpenAI's
|
|
360
|
+
// own docs recommend using json_object anyway.
|
|
361
|
+
notes.push("response_format json_object mapped to a system instruction (no native equivalent)");
|
|
362
|
+
return "Respond with a single valid JSON object and nothing else — no prose, no code fences.";
|
|
363
|
+
}
|
|
364
|
+
return undefined;
|
|
365
|
+
}
|
|
366
|
+
/** Translate OpenAI `tools` into Anthropic tool definitions. Only `function` tools are supported. */
|
|
367
|
+
function translateTools(tools) {
|
|
368
|
+
if (!Array.isArray(tools)) {
|
|
369
|
+
throw new UnsupportedWireContentError("tools must be an array");
|
|
370
|
+
}
|
|
371
|
+
return tools.map((raw, index) => {
|
|
372
|
+
const tool = raw;
|
|
373
|
+
if (!tool || tool.type !== "function" || !tool.function || typeof tool.function.name !== "string") {
|
|
374
|
+
throw new UnsupportedWireContentError(`tools[${index}]: only type "function" tools are supported by the Anthropic wire translator`);
|
|
375
|
+
}
|
|
376
|
+
const out = {
|
|
377
|
+
name: tool.function.name,
|
|
378
|
+
input_schema: tool.function.parameters && typeof tool.function.parameters === "object"
|
|
379
|
+
? tool.function.parameters
|
|
380
|
+
: { type: "object", properties: {} },
|
|
381
|
+
};
|
|
382
|
+
if (typeof tool.function.description === "string")
|
|
383
|
+
out.description = tool.function.description;
|
|
384
|
+
if (tool.function.strict === true)
|
|
385
|
+
out.strict = true;
|
|
386
|
+
return out;
|
|
387
|
+
});
|
|
388
|
+
}
|
|
389
|
+
/**
|
|
390
|
+
* Translate OpenAI `tool_choice`. Forced modes (`required`, a named function)
|
|
391
|
+
* degrade to `auto` plus a system instruction on models that reject forced
|
|
392
|
+
* tool use, and the degradation is recorded in `notes`.
|
|
393
|
+
*/
|
|
394
|
+
function translateToolChoice(choice, modelId, hasTools, notes, systemParts) {
|
|
395
|
+
if (choice === undefined || choice === null || choice === "auto")
|
|
396
|
+
return undefined;
|
|
397
|
+
if (!hasTools)
|
|
398
|
+
return undefined; // OpenAI rejects this combination upstream anyway
|
|
399
|
+
if (choice === "none")
|
|
400
|
+
return { type: "none" };
|
|
401
|
+
const forcedOk = claudeSupportsForcedToolChoice(modelId);
|
|
402
|
+
if (choice === "required") {
|
|
403
|
+
if (forcedOk)
|
|
404
|
+
return { type: "any" };
|
|
405
|
+
systemParts.push("You must respond by calling one of the provided tools.");
|
|
406
|
+
notes.push(`tool_choice 'required' degraded to auto + instruction: ${modelId} rejects forced tool use`);
|
|
407
|
+
return { type: "auto" };
|
|
408
|
+
}
|
|
409
|
+
if (typeof choice === "object") {
|
|
410
|
+
const named = choice.function?.name;
|
|
411
|
+
if (choice.type === "function" && typeof named === "string") {
|
|
412
|
+
if (forcedOk)
|
|
413
|
+
return { type: "tool", name: named };
|
|
414
|
+
systemParts.push(`You must respond by calling the tool \`${named}\`.`);
|
|
415
|
+
notes.push(`tool_choice '${noteToken(named)}' degraded to auto + instruction: ${modelId} rejects forced tool use`);
|
|
416
|
+
return { type: "auto" };
|
|
417
|
+
}
|
|
418
|
+
}
|
|
419
|
+
throw new UnsupportedWireContentError(`unsupported tool_choice value '${JSON.stringify(choice)}'`);
|
|
420
|
+
}
|
|
421
|
+
/** Assistant message → text + tool_use blocks. */
|
|
422
|
+
function assistantBlocks(msg) {
|
|
423
|
+
const blocks = [];
|
|
424
|
+
const text = flattenText(msg.content);
|
|
425
|
+
if (text)
|
|
426
|
+
blocks.push({ type: "text", text });
|
|
427
|
+
if (msg.tool_calls !== undefined && msg.tool_calls !== null) {
|
|
428
|
+
if (!Array.isArray(msg.tool_calls)) {
|
|
429
|
+
throw new UnsupportedWireContentError("assistant.tool_calls must be an array");
|
|
430
|
+
}
|
|
431
|
+
msg.tool_calls.forEach((raw, index) => {
|
|
432
|
+
const call = raw;
|
|
433
|
+
if (!call || typeof call.id !== "string" || !call.function || typeof call.function.name !== "string") {
|
|
434
|
+
throw new UnsupportedWireContentError(`tool_calls[${index}] is missing id or function.name`);
|
|
435
|
+
}
|
|
436
|
+
blocks.push({
|
|
437
|
+
type: "tool_use",
|
|
438
|
+
id: call.id,
|
|
439
|
+
name: call.function.name,
|
|
440
|
+
input: parseToolArguments(call.function.arguments, index),
|
|
441
|
+
});
|
|
442
|
+
});
|
|
443
|
+
}
|
|
444
|
+
return blocks;
|
|
445
|
+
}
|
|
446
|
+
/** OpenAI carries tool arguments as a JSON string; Anthropic wants the object. */
|
|
447
|
+
function parseToolArguments(raw, index) {
|
|
448
|
+
if (raw === undefined || raw === null || raw === "")
|
|
449
|
+
return {};
|
|
450
|
+
if (typeof raw === "object")
|
|
451
|
+
return raw;
|
|
452
|
+
if (typeof raw !== "string") {
|
|
453
|
+
throw new UnsupportedWireContentError(`tool_calls[${index}].function.arguments must be a JSON string`);
|
|
454
|
+
}
|
|
455
|
+
try {
|
|
456
|
+
const parsed = JSON.parse(raw);
|
|
457
|
+
return parsed && typeof parsed === "object" ? parsed : {};
|
|
458
|
+
}
|
|
459
|
+
catch {
|
|
460
|
+
throw new UnsupportedWireContentError(`tool_calls[${index}].function.arguments is not valid JSON`);
|
|
461
|
+
}
|
|
462
|
+
}
|
|
463
|
+
/** User message content → text / image blocks. Adjacent text parts are merged. */
|
|
464
|
+
function userBlocks(content) {
|
|
465
|
+
if (content === null || content === undefined)
|
|
466
|
+
return [];
|
|
467
|
+
if (typeof content === "string")
|
|
468
|
+
return content ? [{ type: "text", text: content }] : [];
|
|
469
|
+
const blocks = [];
|
|
470
|
+
for (const part of content) {
|
|
471
|
+
if (isTextPart(part)) {
|
|
472
|
+
const text = typeof part === "string" ? part : part.text;
|
|
473
|
+
if (!text)
|
|
474
|
+
continue;
|
|
475
|
+
const tail = blocks[blocks.length - 1];
|
|
476
|
+
if (tail && tail.type === "text")
|
|
477
|
+
tail.text += text;
|
|
478
|
+
else
|
|
479
|
+
blocks.push({ type: "text", text });
|
|
480
|
+
continue;
|
|
481
|
+
}
|
|
482
|
+
if (part && typeof part === "object" && part.type === "image_url") {
|
|
483
|
+
blocks.push(imageBlock(part));
|
|
484
|
+
continue;
|
|
485
|
+
}
|
|
486
|
+
throw new UnsupportedWireContentError(`unsupported message content block '${describePartType(part)}' — text and image_url are carried across the Anthropic wire translator`);
|
|
487
|
+
}
|
|
488
|
+
return blocks;
|
|
489
|
+
}
|
|
490
|
+
/** OpenAI `image_url` part → Anthropic `image` block (data: URLs become base64 sources). */
|
|
491
|
+
function imageBlock(part) {
|
|
492
|
+
const url = typeof part.image_url === "string"
|
|
493
|
+
? part.image_url
|
|
494
|
+
: part.image_url && typeof part.image_url === "object"
|
|
495
|
+
? part.image_url.url
|
|
496
|
+
: undefined;
|
|
497
|
+
if (typeof url !== "string" || url.length === 0) {
|
|
498
|
+
throw new UnsupportedWireContentError("image_url part is missing a url");
|
|
499
|
+
}
|
|
500
|
+
const dataUrl = /^data:([a-z0-9.+-]+\/[a-z0-9.+-]+)(?:;[^,]*)?;base64,(.+)$/is.exec(url);
|
|
501
|
+
if (dataUrl) {
|
|
502
|
+
const mediaType = dataUrl[1].toLowerCase();
|
|
503
|
+
if (!IMAGE_MEDIA_TYPES.has(mediaType)) {
|
|
504
|
+
throw new UnsupportedWireContentError(`image media type '${mediaType}' is not accepted by Anthropic`);
|
|
505
|
+
}
|
|
506
|
+
return { type: "image", source: { type: "base64", media_type: mediaType, data: dataUrl[2] } };
|
|
507
|
+
}
|
|
508
|
+
if (!/^https?:\/\//i.test(url)) {
|
|
509
|
+
throw new UnsupportedWireContentError("image_url must be an http(s) URL or a base64 data: URL");
|
|
510
|
+
}
|
|
511
|
+
return { type: "image", source: { type: "url", url } };
|
|
512
|
+
}
|
|
513
|
+
/** A block list that is a single text block collapses to a plain string (keeps request bodies small and readable). */
|
|
514
|
+
function collapseTextOnly(blocks) {
|
|
515
|
+
if (blocks.length === 1 && blocks[0].type === "text")
|
|
516
|
+
return blocks[0].text;
|
|
517
|
+
return blocks;
|
|
518
|
+
}
|
|
519
|
+
/**
|
|
520
|
+
* Collapse runs of consecutive same-role messages into one message per role
|
|
521
|
+
* boundary. Plain strings join with a blank line so the boundary survives in
|
|
522
|
+
* the rendered prompt; anything involving blocks becomes a block list.
|
|
523
|
+
*/
|
|
524
|
+
function mergeConsecutiveRoles(messages) {
|
|
525
|
+
const out = [];
|
|
526
|
+
for (const msg of messages) {
|
|
527
|
+
const tail = out[out.length - 1];
|
|
528
|
+
if (tail && tail.role === msg.role) {
|
|
529
|
+
tail.content = concatContent(tail.content, msg.content);
|
|
530
|
+
}
|
|
531
|
+
else {
|
|
532
|
+
out.push({ role: msg.role, content: msg.content });
|
|
533
|
+
}
|
|
534
|
+
}
|
|
535
|
+
return out;
|
|
536
|
+
}
|
|
537
|
+
function concatContent(a, b) {
|
|
538
|
+
if (typeof a === "string" && typeof b === "string") {
|
|
539
|
+
if (!a)
|
|
540
|
+
return b;
|
|
541
|
+
if (!b)
|
|
542
|
+
return a;
|
|
543
|
+
return `${a}\n\n${b}`;
|
|
544
|
+
}
|
|
545
|
+
return [...toBlocks(a), ...toBlocks(b)];
|
|
546
|
+
}
|
|
547
|
+
function toBlocks(content) {
|
|
548
|
+
if (typeof content === "string")
|
|
549
|
+
return content ? [{ type: "text", text: content }] : [];
|
|
550
|
+
return content;
|
|
551
|
+
}
|
|
552
|
+
/**
|
|
553
|
+
* Make a conversation that ends on an assistant turn acceptable to Anthropic.
|
|
554
|
+
*
|
|
555
|
+
* - A trailing assistant turn is a prefill. Prefill is rejected (400) on Opus
|
|
556
|
+
* 4.6 / Sonnet 4.6 and everything newer, and a trailing `tool_use` without
|
|
557
|
+
* its `tool_result` is rejected on every generation — in both cases a
|
|
558
|
+
* minimal user turn is appended so the request is valid, and a note records
|
|
559
|
+
* it.
|
|
560
|
+
* - Where prefill is still supported, trailing whitespace is trimmed (also a
|
|
561
|
+
* 400 otherwise). An assistant turn that becomes empty is dropped.
|
|
562
|
+
*/
|
|
563
|
+
function finalizeTrailingAssistant(messages, modelId, notes) {
|
|
564
|
+
const last = messages[messages.length - 1];
|
|
565
|
+
if (!last || last.role !== "assistant")
|
|
566
|
+
return messages;
|
|
567
|
+
// Trim first so the emptiness check below is accurate.
|
|
568
|
+
if (typeof last.content === "string") {
|
|
569
|
+
last.content = last.content.trimEnd();
|
|
570
|
+
}
|
|
571
|
+
else {
|
|
572
|
+
const tail = last.content[last.content.length - 1];
|
|
573
|
+
if (tail && tail.type === "text")
|
|
574
|
+
tail.text = tail.text.trimEnd();
|
|
575
|
+
last.content = last.content.filter((block) => block.type !== "text" || block.text.length > 0);
|
|
576
|
+
}
|
|
577
|
+
const isEmpty = typeof last.content === "string" ? last.content.length === 0 : last.content.length === 0;
|
|
578
|
+
if (isEmpty) {
|
|
579
|
+
messages.pop();
|
|
580
|
+
if (messages.length === 0)
|
|
581
|
+
messages.push({ role: "user", content: PLACEHOLDER_USER_TEXT });
|
|
582
|
+
return messages;
|
|
583
|
+
}
|
|
584
|
+
const endsInToolUse = typeof last.content !== "string" && last.content.some((block) => block.type === "tool_use");
|
|
585
|
+
if (endsInToolUse || !claudeSupportsPrefill(modelId)) {
|
|
586
|
+
messages.push({ role: "user", content: PLACEHOLDER_USER_TEXT });
|
|
587
|
+
notes.push(endsInToolUse
|
|
588
|
+
? "appended a user turn: conversation ended on tool_calls with no tool results"
|
|
589
|
+
: `appended a user turn: ${modelId} does not accept a trailing assistant (prefill) message`);
|
|
590
|
+
}
|
|
591
|
+
return messages;
|
|
592
|
+
}
|
|
593
|
+
/**
|
|
594
|
+
* Coerce content into a flat string. Only plain-text parts survive: a
|
|
595
|
+
* string, an explicit `{type:"text",text}` block, or an untyped `{text}`
|
|
596
|
+
* block. Used for system / tool / assistant text where images make no sense.
|
|
597
|
+
*/
|
|
598
|
+
function flattenText(content) {
|
|
599
|
+
if (content === null || content === undefined)
|
|
600
|
+
return "";
|
|
601
|
+
if (typeof content === "string")
|
|
602
|
+
return content;
|
|
603
|
+
const out = [];
|
|
604
|
+
for (const part of content) {
|
|
605
|
+
if (isTextPart(part)) {
|
|
606
|
+
out.push(typeof part === "string" ? part : part.text);
|
|
607
|
+
continue;
|
|
608
|
+
}
|
|
609
|
+
throw new UnsupportedWireContentError(`unsupported message content block '${describePartType(part)}' — only plain text is allowed in this position`);
|
|
610
|
+
}
|
|
611
|
+
return out.join("");
|
|
612
|
+
}
|
|
613
|
+
/** True for plain-text content parts: a string, {type:"text",text}, or untyped {text}. */
|
|
614
|
+
function isTextPart(part) {
|
|
615
|
+
if (typeof part === "string")
|
|
616
|
+
return true;
|
|
617
|
+
if (!part || typeof part !== "object")
|
|
618
|
+
return false;
|
|
619
|
+
const p = part;
|
|
620
|
+
if (p.type !== undefined && p.type !== "text")
|
|
621
|
+
return false;
|
|
622
|
+
return typeof p.text === "string";
|
|
623
|
+
}
|
|
624
|
+
/** Human-readable label for a content block, used in error messages. */
|
|
625
|
+
function describePartType(part) {
|
|
626
|
+
if (part && typeof part === "object" && "type" in part) {
|
|
627
|
+
return String(part.type ?? "unknown");
|
|
628
|
+
}
|
|
629
|
+
return typeof part;
|
|
630
|
+
}
|
|
631
|
+
// ---------------------------------------------------------------------------
|
|
632
|
+
// Response translation
|
|
633
|
+
// ---------------------------------------------------------------------------
|
|
634
|
+
/** Translate an Anthropic /v1/messages response back into OpenAI chat-completions shape. */
|
|
635
|
+
export function translateAnthropicToOpenAI(res, originalModel) {
|
|
636
|
+
const blocks = res.content ?? [];
|
|
637
|
+
const text = blocks
|
|
638
|
+
.filter((block) => block && block.type === "text" && typeof block.text === "string")
|
|
639
|
+
.map((block) => block.text)
|
|
640
|
+
.join("");
|
|
641
|
+
const toolCalls = blocks
|
|
642
|
+
.filter((block) => block && block.type === "tool_use" && typeof block.id === "string" && typeof block.name === "string")
|
|
643
|
+
.map((block) => ({
|
|
644
|
+
id: block.id,
|
|
645
|
+
type: "function",
|
|
646
|
+
function: { name: block.name, arguments: JSON.stringify(block.input ?? {}) },
|
|
647
|
+
}));
|
|
648
|
+
const message = {
|
|
649
|
+
role: "assistant",
|
|
650
|
+
// OpenAI SDKs expect null (not "") when the turn is tool calls only.
|
|
651
|
+
content: text.length > 0 ? text : toolCalls.length > 0 ? null : "",
|
|
652
|
+
};
|
|
653
|
+
if (toolCalls.length > 0)
|
|
654
|
+
message.tool_calls = toolCalls;
|
|
655
|
+
return {
|
|
656
|
+
id: res.id,
|
|
657
|
+
object: "chat.completion",
|
|
658
|
+
created: Math.floor(Date.now() / 1000),
|
|
659
|
+
model: originalModel,
|
|
660
|
+
choices: [{ index: 0, message, finish_reason: mapStopReason(res.stop_reason, toolCalls.length > 0) }],
|
|
661
|
+
usage: toOpenAIUsage(res.usage),
|
|
662
|
+
};
|
|
663
|
+
}
|
|
664
|
+
/**
|
|
665
|
+
* Anthropic usage → OpenAI usage. OpenAI's `prompt_tokens` counts every
|
|
666
|
+
* input token including cached ones, so cache reads/writes fold into it;
|
|
667
|
+
* the cached portion is echoed under `prompt_tokens_details.cached_tokens`
|
|
668
|
+
* (only when non-zero, so responses without caching stay byte-compatible).
|
|
669
|
+
*/
|
|
670
|
+
export function toOpenAIUsage(usage) {
|
|
671
|
+
const input = usage?.input_tokens ?? 0;
|
|
672
|
+
const output = usage?.output_tokens ?? 0;
|
|
673
|
+
const cacheRead = usage?.cache_read_input_tokens ?? 0;
|
|
674
|
+
const cacheWrite = usage?.cache_creation_input_tokens ?? 0;
|
|
675
|
+
const prompt = input + cacheRead + cacheWrite;
|
|
676
|
+
const out = { prompt_tokens: prompt, completion_tokens: output, total_tokens: prompt + output };
|
|
677
|
+
if (cacheRead > 0)
|
|
678
|
+
out.prompt_tokens_details = { cached_tokens: cacheRead };
|
|
679
|
+
return out;
|
|
680
|
+
}
|
|
681
|
+
/**
|
|
682
|
+
* Map Anthropic stop reasons onto the OpenAI vocabulary. `tool_use` becomes
|
|
683
|
+
* `tool_calls` only when a populated tool_calls array actually went out —
|
|
684
|
+
* a dangling `tool_calls` signal makes agent SDKs loop. `refusal` (safety
|
|
685
|
+
* classifier) maps to OpenAI's `content_filter`.
|
|
686
|
+
*/
|
|
687
|
+
function mapStopReason(reason, hasToolCalls = false) {
|
|
688
|
+
switch (reason) {
|
|
689
|
+
case "end_turn":
|
|
690
|
+
case "stop_sequence":
|
|
691
|
+
return "stop";
|
|
692
|
+
case "max_tokens":
|
|
693
|
+
return "length";
|
|
694
|
+
case "tool_use":
|
|
695
|
+
return hasToolCalls ? "tool_calls" : "stop";
|
|
696
|
+
case "refusal":
|
|
697
|
+
return "content_filter";
|
|
698
|
+
default:
|
|
699
|
+
return "stop";
|
|
700
|
+
}
|
|
701
|
+
}
|
|
702
|
+
/**
|
|
703
|
+
* Upper bound on SSE bytes retained while waiting for an event delimiter. A
|
|
704
|
+
* well-formed Anthropic event is a few KB; a misbehaving upstream that never
|
|
705
|
+
* sends a blank line must not grow memory (or the rescans) without limit.
|
|
706
|
+
*/
|
|
707
|
+
const MAX_SSE_BUFFER_BYTES = 1_048_576;
|
|
708
|
+
/**
|
|
709
|
+
* Translate an Anthropic `/v1/messages` SSE stream into OpenAI
|
|
710
|
+
* `chat.completion.chunk` SSE.
|
|
711
|
+
*
|
|
712
|
+
* Event mapping:
|
|
713
|
+
* - `message_start` → first chunk with `delta.role = "assistant"`; captures id + input usage
|
|
714
|
+
* - `content_block_start` → for `tool_use` blocks, a chunk announcing `tool_calls[i].id/name`
|
|
715
|
+
* - `content_block_delta` → `text_delta` → `delta.content`; `input_json_delta` → `tool_calls[i].function.arguments`
|
|
716
|
+
* - `message_delta` → chunk with `finish_reason`; captures output usage
|
|
717
|
+
* - `message_stop` → optional usage chunk (when `include_usage`), then `data: [DONE]`
|
|
718
|
+
* - `error` → `data: {"error": …}` then `[DONE]`
|
|
719
|
+
* - `ping`, `content_block_stop`, thinking deltas → ignored
|
|
720
|
+
*
|
|
721
|
+
* The stream is fault-tolerant: if the upstream closes without
|
|
722
|
+
* `message_stop`, the finish chunk and `[DONE]` are still emitted so the
|
|
723
|
+
* caller's SDK doesn't hang.
|
|
724
|
+
*/
|
|
725
|
+
export function translateAnthropicStreamToOpenAI(upstream, originalModel, options = {}) {
|
|
726
|
+
const encoder = new TextEncoder();
|
|
727
|
+
const decoder = new TextDecoder();
|
|
728
|
+
let buffer = "";
|
|
729
|
+
/** Index into `buffer` from which the next delimiter search starts, so each chunk is scanned once. */
|
|
730
|
+
let scanFrom = 0;
|
|
731
|
+
let id = `chatcmpl-${Date.now().toString(36)}`;
|
|
732
|
+
const created = Math.floor(Date.now() / 1000);
|
|
733
|
+
const usage = {};
|
|
734
|
+
let sawUsage = false;
|
|
735
|
+
let sawStop = false;
|
|
736
|
+
let stopReason = null;
|
|
737
|
+
let roleSent = false;
|
|
738
|
+
let finishSent = false;
|
|
739
|
+
let done = false;
|
|
740
|
+
const toolIndexByBlock = new Map();
|
|
741
|
+
/** Argument bytes streamed so far per tool index; zero at stop → emit "{}". */
|
|
742
|
+
const toolArgBytes = new Map();
|
|
743
|
+
let toolCount = 0;
|
|
744
|
+
let resolveUsage;
|
|
745
|
+
const usagePromise = new Promise((resolve) => {
|
|
746
|
+
resolveUsage = resolve;
|
|
747
|
+
});
|
|
748
|
+
/** Reconciliation is only trustworthy once Anthropic reported the final output count. */
|
|
749
|
+
const settleUsage = () => resolveUsage(sawUsage && sawStop ? { ...usage } : null);
|
|
750
|
+
const encodeChunk = (delta, finishReason = null) => encoder.encode(`data: ${JSON.stringify({
|
|
751
|
+
id,
|
|
752
|
+
object: "chat.completion.chunk",
|
|
753
|
+
created,
|
|
754
|
+
model: originalModel,
|
|
755
|
+
choices: [{ index: 0, delta, finish_reason: finishReason }],
|
|
756
|
+
})}\n\n`);
|
|
757
|
+
const mergeUsage = (src) => {
|
|
758
|
+
if (!src || typeof src !== "object")
|
|
759
|
+
return;
|
|
760
|
+
const u = src;
|
|
761
|
+
for (const key of ["input_tokens", "output_tokens", "cache_creation_input_tokens", "cache_read_input_tokens"]) {
|
|
762
|
+
if (typeof u[key] === "number") {
|
|
763
|
+
usage[key] = u[key];
|
|
764
|
+
sawUsage = true;
|
|
765
|
+
}
|
|
766
|
+
}
|
|
767
|
+
};
|
|
768
|
+
/**
|
|
769
|
+
* OpenAI SDKs `JSON.parse` the accumulated arguments; real OpenAI always
|
|
770
|
+
* emits at least "{}" for a no-argument call, so a tool block that streamed
|
|
771
|
+
* zero argument bytes gets "{}" before the stream finishes.
|
|
772
|
+
*/
|
|
773
|
+
const closeEmptyToolArgs = (controller, only) => {
|
|
774
|
+
for (const [toolIndex, bytes] of toolArgBytes) {
|
|
775
|
+
if (only !== undefined && toolIndex !== only)
|
|
776
|
+
continue;
|
|
777
|
+
if (bytes > 0)
|
|
778
|
+
continue;
|
|
779
|
+
controller.enqueue(encodeChunk({ tool_calls: [{ index: toolIndex, function: { arguments: "{}" } }] }));
|
|
780
|
+
toolArgBytes.set(toolIndex, 2);
|
|
781
|
+
}
|
|
782
|
+
};
|
|
783
|
+
const emitFinish = (controller) => {
|
|
784
|
+
if (finishSent)
|
|
785
|
+
return;
|
|
786
|
+
finishSent = true;
|
|
787
|
+
if (!roleSent) {
|
|
788
|
+
controller.enqueue(encodeChunk({ role: "assistant", content: "" }));
|
|
789
|
+
roleSent = true;
|
|
790
|
+
}
|
|
791
|
+
closeEmptyToolArgs(controller);
|
|
792
|
+
controller.enqueue(encodeChunk({}, mapStopReason(stopReason, toolCount > 0)));
|
|
793
|
+
};
|
|
794
|
+
const emitDone = (controller) => {
|
|
795
|
+
if (done)
|
|
796
|
+
return;
|
|
797
|
+
done = true;
|
|
798
|
+
emitFinish(controller);
|
|
799
|
+
if (options.includeUsage) {
|
|
800
|
+
controller.enqueue(encoder.encode(`data: ${JSON.stringify({
|
|
801
|
+
id,
|
|
802
|
+
object: "chat.completion.chunk",
|
|
803
|
+
created,
|
|
804
|
+
model: originalModel,
|
|
805
|
+
choices: [],
|
|
806
|
+
usage: toOpenAIUsage(usage),
|
|
807
|
+
})}\n\n`));
|
|
808
|
+
}
|
|
809
|
+
controller.enqueue(encoder.encode("data: [DONE]\n\n"));
|
|
810
|
+
settleUsage();
|
|
811
|
+
};
|
|
812
|
+
const handleEvent = (evt, controller) => {
|
|
813
|
+
switch (evt.type) {
|
|
814
|
+
case "message_start": {
|
|
815
|
+
const message = evt.message;
|
|
816
|
+
if (message && typeof message.id === "string")
|
|
817
|
+
id = message.id;
|
|
818
|
+
mergeUsage(message?.usage);
|
|
819
|
+
if (!roleSent) {
|
|
820
|
+
controller.enqueue(encodeChunk({ role: "assistant", content: "" }));
|
|
821
|
+
roleSent = true;
|
|
822
|
+
}
|
|
823
|
+
break;
|
|
824
|
+
}
|
|
825
|
+
case "content_block_start": {
|
|
826
|
+
const block = evt.content_block;
|
|
827
|
+
if (block?.type === "tool_use" && typeof evt.index === "number") {
|
|
828
|
+
const toolIndex = toolCount++;
|
|
829
|
+
toolIndexByBlock.set(evt.index, toolIndex);
|
|
830
|
+
toolArgBytes.set(toolIndex, 0);
|
|
831
|
+
controller.enqueue(encodeChunk({
|
|
832
|
+
tool_calls: [
|
|
833
|
+
{
|
|
834
|
+
index: toolIndex,
|
|
835
|
+
id: typeof block.id === "string" ? block.id : `call_${toolIndex}`,
|
|
836
|
+
type: "function",
|
|
837
|
+
function: { name: typeof block.name === "string" ? block.name : "", arguments: "" },
|
|
838
|
+
},
|
|
839
|
+
],
|
|
840
|
+
}));
|
|
841
|
+
}
|
|
842
|
+
break;
|
|
843
|
+
}
|
|
844
|
+
case "content_block_delta": {
|
|
845
|
+
const delta = evt.delta;
|
|
846
|
+
if (delta?.type === "text_delta" && typeof delta.text === "string") {
|
|
847
|
+
controller.enqueue(encodeChunk({ content: delta.text }));
|
|
848
|
+
}
|
|
849
|
+
else if (delta?.type === "input_json_delta" && typeof delta.partial_json === "string") {
|
|
850
|
+
const toolIndex = typeof evt.index === "number" ? toolIndexByBlock.get(evt.index) : undefined;
|
|
851
|
+
if (toolIndex !== undefined && delta.partial_json.length > 0) {
|
|
852
|
+
toolArgBytes.set(toolIndex, (toolArgBytes.get(toolIndex) ?? 0) + delta.partial_json.length);
|
|
853
|
+
controller.enqueue(encodeChunk({ tool_calls: [{ index: toolIndex, function: { arguments: delta.partial_json } }] }));
|
|
854
|
+
}
|
|
855
|
+
}
|
|
856
|
+
break;
|
|
857
|
+
}
|
|
858
|
+
case "content_block_stop": {
|
|
859
|
+
const toolIndex = typeof evt.index === "number" ? toolIndexByBlock.get(evt.index) : undefined;
|
|
860
|
+
if (toolIndex !== undefined)
|
|
861
|
+
closeEmptyToolArgs(controller, toolIndex);
|
|
862
|
+
break;
|
|
863
|
+
}
|
|
864
|
+
case "message_delta": {
|
|
865
|
+
const delta = evt.delta;
|
|
866
|
+
if (delta && typeof delta.stop_reason === "string")
|
|
867
|
+
stopReason = delta.stop_reason;
|
|
868
|
+
mergeUsage(evt.usage);
|
|
869
|
+
sawStop = true;
|
|
870
|
+
emitFinish(controller);
|
|
871
|
+
break;
|
|
872
|
+
}
|
|
873
|
+
case "message_stop":
|
|
874
|
+
emitDone(controller);
|
|
875
|
+
break;
|
|
876
|
+
case "error": {
|
|
877
|
+
const err = evt.error;
|
|
878
|
+
controller.enqueue(encoder.encode(`data: ${JSON.stringify({
|
|
879
|
+
error: {
|
|
880
|
+
message: typeof err?.message === "string" ? err.message : "anthropic stream error",
|
|
881
|
+
type: typeof err?.type === "string" ? err.type : "upstream_error",
|
|
882
|
+
},
|
|
883
|
+
})}\n\n`));
|
|
884
|
+
emitDone(controller);
|
|
885
|
+
break;
|
|
886
|
+
}
|
|
887
|
+
default:
|
|
888
|
+
break; // ping, thinking / signature deltas
|
|
889
|
+
}
|
|
890
|
+
};
|
|
891
|
+
const handleEventText = (part, controller) => {
|
|
892
|
+
const data = part
|
|
893
|
+
.split(/\r?\n/)
|
|
894
|
+
.filter((line) => line.startsWith("data:"))
|
|
895
|
+
.map((line) => line.slice(5).trimStart())
|
|
896
|
+
.join("\n");
|
|
897
|
+
if (!data)
|
|
898
|
+
return;
|
|
899
|
+
let evt;
|
|
900
|
+
try {
|
|
901
|
+
evt = JSON.parse(data);
|
|
902
|
+
}
|
|
903
|
+
catch {
|
|
904
|
+
return; // partial or malformed event — skip, never crash the stream
|
|
905
|
+
}
|
|
906
|
+
if (evt && typeof evt === "object" && !done)
|
|
907
|
+
handleEvent(evt, controller);
|
|
908
|
+
};
|
|
909
|
+
/**
|
|
910
|
+
* Consume complete events (delimited by a blank line) from `buffer`. Only
|
|
911
|
+
* bytes appended since the last call are searched, so total work is linear
|
|
912
|
+
* in the stream size even when events arrive in tiny chunks.
|
|
913
|
+
*/
|
|
914
|
+
const drain = (controller, flushAll) => {
|
|
915
|
+
const delimiter = /\r?\n\r?\n/g;
|
|
916
|
+
// Back up 3 chars so a delimiter split across chunk boundaries is still found.
|
|
917
|
+
delimiter.lastIndex = Math.max(0, scanFrom - 3);
|
|
918
|
+
let consumed = 0;
|
|
919
|
+
let match = delimiter.exec(buffer);
|
|
920
|
+
while (match !== null) {
|
|
921
|
+
handleEventText(buffer.slice(consumed, match.index), controller);
|
|
922
|
+
consumed = match.index + match[0].length;
|
|
923
|
+
match = delimiter.exec(buffer);
|
|
924
|
+
}
|
|
925
|
+
if (flushAll) {
|
|
926
|
+
handleEventText(buffer.slice(consumed), controller);
|
|
927
|
+
buffer = "";
|
|
928
|
+
scanFrom = 0;
|
|
929
|
+
return;
|
|
930
|
+
}
|
|
931
|
+
buffer = buffer.slice(consumed);
|
|
932
|
+
scanFrom = buffer.length;
|
|
933
|
+
};
|
|
934
|
+
const transform = new TransformStream({
|
|
935
|
+
transform(bytes, controller) {
|
|
936
|
+
buffer += decoder.decode(bytes, { stream: true });
|
|
937
|
+
drain(controller, false);
|
|
938
|
+
if (buffer.length > MAX_SSE_BUFFER_BYTES) {
|
|
939
|
+
// Upstream is not sending well-formed SSE. Fail the stream the
|
|
940
|
+
// OpenAI way, settle usage, and stop pulling from upstream.
|
|
941
|
+
controller.enqueue(encoder.encode(`data: ${JSON.stringify({ error: { message: "anthropic stream exceeded the event buffer limit", type: "upstream_error" } })}\n\n`));
|
|
942
|
+
emitDone(controller);
|
|
943
|
+
controller.terminate();
|
|
944
|
+
}
|
|
945
|
+
},
|
|
946
|
+
flush(controller) {
|
|
947
|
+
buffer += decoder.decode();
|
|
948
|
+
drain(controller, true);
|
|
949
|
+
emitDone(controller);
|
|
950
|
+
},
|
|
951
|
+
// Runs when the client cancels the response or the upstream errors —
|
|
952
|
+
// neither path reaches flush(), and the proxy's ledger write is
|
|
953
|
+
// waiting on the usage promise.
|
|
954
|
+
cancel() {
|
|
955
|
+
settleUsage();
|
|
956
|
+
},
|
|
957
|
+
});
|
|
958
|
+
return { stream: upstream.pipeThrough(transform), usage: usagePromise };
|
|
959
|
+
}
|
|
960
|
+
//# sourceMappingURL=wire-anthropic.js.map
|