llm-switcher 1.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. package/.gitattributes +16 -0
  2. package/LICENSE +21 -0
  3. package/README.md +587 -0
  4. package/README.vi.md +585 -0
  5. package/blindfold/blindfold.mjs +633 -0
  6. package/blindfold/make-certs.sh +88 -0
  7. package/blindfold/wsframe.mjs +176 -0
  8. package/codex-catalog-template.json +1 -0
  9. package/config.example.json +84 -0
  10. package/contract-exclusions.json +41 -0
  11. package/contract.mjs +561 -0
  12. package/docs/LLM-RESPONSE-MATRIX.md +165 -0
  13. package/docs/TOKEN-OPTIMIZER-INTEROP.md +110 -0
  14. package/docs/codex-blindfold.md +214 -0
  15. package/docs/cross-platform.md +136 -0
  16. package/docs/diagrams/blindfold-request-routing.html +14972 -0
  17. package/docs/diagrams/blindfold-request-routing.sequence.json +175 -0
  18. package/docs/diagrams/blindfold-switch-lifecycle.html +14958 -0
  19. package/docs/diagrams/blindfold-switch-lifecycle.lifecycle.json +159 -0
  20. package/docs/diagrams/codex-model-name-resolution.html +15005 -0
  21. package/docs/diagrams/codex-model-name-resolution.workflow.json +71 -0
  22. package/docs/response-matrix.json +1131 -0
  23. package/formats.mjs +2308 -0
  24. package/mcp.mjs +340 -0
  25. package/package.json +36 -0
  26. package/proxy.mjs +1743 -0
  27. package/service.mjs +132 -0
  28. package/shim.mjs +292 -0
  29. package/skills/llm-switcher/SKILL.md +88 -0
  30. package/state.mjs +978 -0
  31. package/switch +5 -0
  32. package/switch.cmd +2 -0
  33. package/switch.mjs +930 -0
  34. package/tests/blindfold.test.mjs +307 -0
  35. package/tests/blindfold.wire.test.mjs +170 -0
  36. package/tests/contract/run.test.mjs +214 -0
  37. package/tests/contract-check.test.mjs +458 -0
  38. package/tests/contract-lab.test.mjs +755 -0
  39. package/tests/datadir.test.mjs +37 -0
  40. package/tests/formats.test.mjs +794 -0
  41. package/tests/gateway.e2e.test.mjs +999 -0
  42. package/tests/helpers.mjs +24 -0
  43. package/tests/lifecycle.test.mjs +416 -0
  44. package/tests/live-optimizer-interop.mjs +205 -0
  45. package/tests/mcp.test.mjs +91 -0
  46. package/tests/service.test.mjs +69 -0
  47. package/tests/shim.test.mjs +228 -0
  48. package/tests/state.test.mjs +675 -0
  49. package/tests/switch.test.mjs +156 -0
  50. package/tests/wsframe.test.mjs +154 -0
  51. package/ui.html +2234 -0
package/formats.mjs ADDED
@@ -0,0 +1,2308 @@
1
+ // ============================================================
2
+ // formats.mjs — LLM Switcher protocol adapters (zero dependency)
3
+ //
4
+ // Two-way conversion flow through IR (Intermediate Representation):
5
+ //
6
+ // client --parse--> IR --emit--> upstream --events--> client
7
+ //
8
+ // Client (input) : anthropic | openai-chat | responses (Codex) | vertex
9
+ // Upstream (out) : openai-chat | anthropic | vertex
10
+ //
11
+ // IR shape:
12
+ // {
13
+ // model, system,
14
+ // messages: [{ role:'system'|'user'|'assistant'|'tool',
15
+ // content: string | [{type:'text',text}|{type:'image_url',image_url:{url}}],
16
+ // toolCalls?: [{id,name,args}], toolCallId?, name? }],
17
+ // tools: [{ name, description, parameters }],
18
+ // toolChoice: 'auto'|'required'|'none'|{name}|null,
19
+ // params: { maxTokens, temperature, topP, topK, stop[],
20
+ // presencePenalty?, frequencyPenalty? },
21
+ // thinking: { type:'enabled'|'disabled'|'adaptive', budget?, effort? } | null,
22
+ // stream: bool
23
+ // }
24
+ // ============================================================
25
+
26
+ export const IN_FORMATS = ['anthropic', 'openai-chat', 'responses', 'vertex'];
27
+ export const OUT_FORMATS = ['openai-chat', 'anthropic', 'vertex'];
28
+
29
+ // ---------------- stop reasons ----------------
30
+
31
+ // Canonical finish used internally for the event stream.
32
+ function canonFinish(raw) {
33
+ const r = String(raw || '').toLowerCase();
34
+ if (!r) return null;
35
+ if (['stop', 'end_turn', 'stop_sequence', 'eos', 'end'].includes(r)) return 'stop';
36
+ if (['length', 'max_tokens', 'max_output_tokens'].includes(r)) return 'length';
37
+ if (['tool_calls', 'tool_call', 'function_call', 'function_calls', 'tool_use'].includes(r)) return 'tool_calls';
38
+ if (['content_filter', 'safety', 'recitation', 'language', 'blocklist', 'prohibited_content', 'spii'].includes(r)) return 'content_filter';
39
+ return 'stop';
40
+ }
41
+
42
+ // `length` wins over tool calls: a cut-off tool call must reach the client as truncated, not as done.
43
+ function chatFinish(canonical, hasTools) {
44
+ if (canonical === 'length') return 'length';
45
+ if (hasTools) return 'tool_calls';
46
+ switch (canonical) {
47
+ case 'length': return 'length';
48
+ case 'content_filter': return 'content_filter';
49
+ default: return 'stop';
50
+ }
51
+ }
52
+
53
+ // ---------------- smart field detection (moved from proxy) ----------------
54
+
55
+ const REASONING_KEYS = ['reasoning_content', 'reasoning', 'thinking', 'thought', 'thinking_content', 'thinking_text', 'reasoning_text', 'chain_of_thought', 'cot', 'rationale'];
56
+ const TEXT_KEYS = ['text', 'answer', 'output', 'response', 'completion', 'generated_text', 'result', 'message_text', 'reply'];
57
+
58
+ function asText(v) {
59
+ if (typeof v === 'string') return v;
60
+ if (typeof v === 'number' || typeof v === 'boolean') return String(v);
61
+ if (Array.isArray(v)) {
62
+ return v
63
+ .map(x => (typeof x === 'string' ? x : (x && typeof x === 'object' ? (typeof x.text === 'string' ? x.text : '') : '')))
64
+ .filter(Boolean)
65
+ .join('\n');
66
+ }
67
+ return '';
68
+ }
69
+
70
+ function splitParts(parts) {
71
+ const out = { thinking: [], text: [], tools: [], signature: null };
72
+ if (!Array.isArray(parts)) return out;
73
+ for (const p of parts) {
74
+ if (typeof p === 'string') {
75
+ if (p) out.text.push(p);
76
+ continue;
77
+ }
78
+ if (!p || typeof p !== 'object') continue;
79
+ const t = typeof p.text === 'string' ? p.text : '';
80
+ const partSig = typeof (p.thoughtSignature ?? p.thought_signature) === 'string' ? (p.thoughtSignature ?? p.thought_signature) : null;
81
+ const fc = p.functionCall || p.function_call;
82
+ if (p.thought === true) {
83
+ if (t) out.thinking.push(t);
84
+ } else if (t) {
85
+ out.text.push(t);
86
+ }
87
+ // Gemini 3 may attach the signature to an empty text part in the last chunk -> still must pick it up.
88
+ if (!out.signature && partSig && !fc) out.signature = partSig;
89
+ if (fc && typeof fc === 'object') {
90
+ const args = fc.args !== undefined ? fc.args : fc.arguments;
91
+ out.tools.push({
92
+ index: out.tools.length,
93
+ id: p.id || fc.id || null,
94
+ name: fc.name || '',
95
+ args: typeof args === 'string' ? args : JSON.stringify(args ?? {}),
96
+ sig: partSig
97
+ });
98
+ }
99
+ }
100
+ return out;
101
+ }
102
+
103
+ function smartReasoning(node) {
104
+ if (!node || typeof node !== 'object') return null;
105
+ const texts = [];
106
+ let signature = null;
107
+ const take = (s) => { if (typeof s === 'string' && s) texts.push(s); };
108
+ for (const k of REASONING_KEYS) {
109
+ const v = node[k];
110
+ if (typeof v === 'string' && v) { take(v); break; }
111
+ if (v && typeof v === 'object' && !Array.isArray(v)) {
112
+ take(v.text ?? v.content ?? '');
113
+ if (texts.length) break;
114
+ } else if (Array.isArray(v) && v.length) {
115
+ for (const it of v) {
116
+ if (typeof it === 'string') take(it);
117
+ else if (it && typeof it === 'object') take(it.text ?? it.content ?? it.summary ?? '');
118
+ }
119
+ if (texts.length) break;
120
+ }
121
+ }
122
+ const det = node.reasoning_details || node.reasoningDetails;
123
+ if (!texts.length && Array.isArray(det)) {
124
+ for (const it of det) {
125
+ if (typeof it === 'string') { take(it); continue; }
126
+ if (!it || typeof it !== 'object') continue;
127
+ take(it.text ?? it.content ?? it.summary ?? '');
128
+ if (!signature && typeof it.signature === 'string' && it.signature) signature = it.signature;
129
+ }
130
+ }
131
+ const parts = node.parts || node.content?.parts;
132
+ if (Array.isArray(parts)) {
133
+ const sp = splitParts(parts);
134
+ // The same thought can arrive in both places; take the parts only when nothing else carried it.
135
+ if (sp.thinking.length && !texts.length) texts.push(sp.thinking.join(''));
136
+ if (!signature) signature = sp.signature;
137
+ }
138
+ if (!signature) {
139
+ const sig = node.thoughtSignature ?? node.thought_signature ?? node.signature;
140
+ if (typeof sig === 'string' && sig) signature = sig;
141
+ }
142
+ const text = texts.join('');
143
+ if (!text) return null;
144
+ return { text, signature };
145
+ }
146
+
147
+ function smartText(node) {
148
+ if (!node || typeof node !== 'object') return '';
149
+ const c = node.content;
150
+ if (c !== undefined && c !== null) {
151
+ if (typeof c === 'string') return c;
152
+ if (Array.isArray(c)) return splitParts(c).text.join('\n');
153
+ if (typeof c === 'object') {
154
+ if (Array.isArray(c.parts)) return splitParts(c.parts).text.join('');
155
+ if (typeof c.text === 'string' && c.text) return c.text;
156
+ }
157
+ return asText(c);
158
+ }
159
+ for (const k of TEXT_KEYS) {
160
+ const v = node[k];
161
+ if (typeof v === 'string' && v) return v;
162
+ }
163
+ return '';
164
+ }
165
+
166
+ function smartToolCalls(node) {
167
+ if (!node || typeof node !== 'object') return [];
168
+ const norm = (tc, idx) => {
169
+ if (!tc || typeof tc !== 'object') return null;
170
+ const fn = (tc.function && typeof tc.function === 'object') ? tc.function : {};
171
+ const name = fn.name || tc.name || '';
172
+ const rawArgs = fn.arguments ?? fn.args ?? tc.args ?? tc.arguments ?? tc.input;
173
+ // Follow-up OpenAI stream chunks only carry {index, function:{arguments}} (no name/id):
174
+ // they must still be kept, otherwise all args after the first chunk are lost.
175
+ if (!name && (rawArgs === undefined || rawArgs === null || rawArgs === '')) return null;
176
+ return {
177
+ index: (typeof tc.index === 'number' ? tc.index : idx),
178
+ id: tc.id || fn.id || null,
179
+ name: name || null,
180
+ args: typeof rawArgs === 'string' ? rawArgs : (name ? JSON.stringify(rawArgs ?? {}) : JSON.stringify(rawArgs)),
181
+ sig: tc.extra_content?.google?.thought_signature || null
182
+ };
183
+ };
184
+ for (const k of ['tool_calls', 'toolCalls', 'tools_called', 'function_calls']) {
185
+ if (Array.isArray(node[k]) && node[k].length) {
186
+ return node[k].map(norm).filter(Boolean);
187
+ }
188
+ }
189
+ const single = node.function_call || node.functionCall;
190
+ if (single && typeof single === 'object') {
191
+ const one = norm({ function: single, id: node.id }, 0);
192
+ return one ? [one] : [];
193
+ }
194
+ const parts = node.parts || node.content?.parts;
195
+ if (Array.isArray(parts)) {
196
+ return splitParts(parts).tools.filter(t => t.name);
197
+ }
198
+ return [];
199
+ }
200
+
201
+ function smartUsage(u) {
202
+ const o = (u && typeof u === 'object') ? u : {};
203
+ const det = (o.prompt_tokens_details && typeof o.prompt_tokens_details === 'object') ? o.prompt_tokens_details : {};
204
+ const num = (...vals) => {
205
+ for (const v of vals) {
206
+ const n = Number(v);
207
+ if (Number.isFinite(n) && n > 0) return n;
208
+ }
209
+ return 0;
210
+ };
211
+ // Reasoning tokens are billed output the client never sees, and each upstream
212
+ // puts them somewhere else: OpenAI chat in completion_tokens_details, the
213
+ // Responses API in output_tokens_details, Anthropic as thinking_tokens, Vertex
214
+ // as thoughtsTokenCount. Measured 2026-09-20 through 9Router: 95 of 96 output
215
+ // tokens were reasoning, so losing this field misreports almost the whole cost.
216
+ const outDet = (o.completion_tokens_details && typeof o.completion_tokens_details === 'object')
217
+ ? o.completion_tokens_details
218
+ : (o.output_tokens_details && typeof o.output_tokens_details === 'object') ? o.output_tokens_details : {};
219
+ // Gemini counts thoughts apart from candidates. Every other API counts them inside the output.
220
+ const geminiOutput = (Number(o.candidatesTokenCount) || 0) + (Number(o.thoughtsTokenCount) || 0);
221
+ return {
222
+ prompt: num(o.prompt_tokens, o.input_tokens, o.promptTokenCount, o.inputTokens),
223
+ completion: num(o.completion_tokens, o.output_tokens, geminiOutput, o.outputTokens),
224
+ cached: num(det.cached_tokens, o.cached_tokens, o.cachedContentTokenCount, o.cached_content_token_count, o.cache_read_input_tokens),
225
+ reasoning: num(outDet.reasoning_tokens, outDet.thinking_tokens, o.thoughtsTokenCount, o.reasoning_tokens)
226
+ };
227
+ }
228
+
229
+ function smartFinish(root, choice) {
230
+ const raw = choice?.finish_reason ?? choice?.finishReason
231
+ ?? root?.finishReason ?? root?.finish_reason
232
+ ?? root?.stop_reason ?? choice?.stop_reason ?? null;
233
+ if (typeof raw === 'string' && raw) return raw;
234
+ if (root?.done === true || choice?.done === true) return 'stop';
235
+ return null;
236
+ }
237
+
238
+ function firstChoice(root) {
239
+ if (!root || typeof root !== 'object') return null;
240
+ for (const k of ['choices', 'candidates', 'outputs', 'results', 'messages']) {
241
+ if (Array.isArray(root[k]) && root[k][0] && typeof root[k][0] === 'object') return root[k][0];
242
+ }
243
+ return null;
244
+ }
245
+
246
+ function smartDelta(choice) {
247
+ if (!choice || typeof choice !== 'object') return null;
248
+ const d = choice.delta ?? choice.message ?? null;
249
+ if (d && typeof d === 'object') return d;
250
+ return choice;
251
+ }
252
+
253
+ const SCHEMA_MAPS = new Set(['properties', 'patternProperties', '$defs', 'definitions']);
254
+
255
+ function sanitizeJsonSchema(schema) {
256
+ if (!schema) return { type: 'object', properties: {} };
257
+ if (typeof schema === 'string') {
258
+ return schema;
259
+ }
260
+ if (typeof schema !== 'object') return schema;
261
+ if (Array.isArray(schema)) return schema.map(sanitizeJsonSchema);
262
+ const clean = {};
263
+ for (const [k, v] of Object.entries(schema)) {
264
+ if (k === '$schema' || k === 'cache_control' || k === 'encrypted') continue;
265
+ if (k === 'format' && ['uri', 'uri-reference'].includes(v)) continue;
266
+ // A name map: each value is a schema, the map itself is not. A parameter may be named "properties".
267
+ if (SCHEMA_MAPS.has(k) && v && typeof v === 'object' && !Array.isArray(v)) {
268
+ clean[k] = Object.fromEntries(Object.entries(v).map(([name, sub]) => [name, sanitizeJsonSchema(sub)]));
269
+ continue;
270
+ }
271
+ clean[k] = sanitizeJsonSchema(v);
272
+ }
273
+ if (!clean.type && clean.properties) {
274
+ clean.type = 'object';
275
+ }
276
+ if (clean.type === 'object' && !clean.properties) {
277
+ clean.properties = {};
278
+ }
279
+ return clean;
280
+ }
281
+
282
+ // ---------------- thinking helpers ----------------
283
+
284
+ function budgetToEffort(b) {
285
+ const n = Number(b) || 0;
286
+ return n >= 8000 ? 'high' : n >= 2000 ? 'medium' : 'low';
287
+ }
288
+
289
+ function effortToBudget(e) {
290
+ switch (String(e || '').toLowerCase()) {
291
+ case 'max':
292
+ case 'xhigh':
293
+ case 'high': return 8000;
294
+ case 'medium': return 4000;
295
+ case 'low':
296
+ case 'minimal': return 1024;
297
+ default: return 2000;
298
+ }
299
+ }
300
+
301
+ function clampBudget(b, fallback = 2048) {
302
+ const n = Number(b) || fallback;
303
+ return Math.max(1024, n);
304
+ }
305
+
306
+ function parseArgs(v) {
307
+ if (v === undefined || v === null) return {};
308
+ if (typeof v === 'object') return v;
309
+ if (typeof v === 'string') {
310
+ try { return JSON.parse(v); } catch { return { raw: v }; }
311
+ }
312
+ return {};
313
+ }
314
+
315
+ function stringifyArgs(v) {
316
+ if (typeof v === 'string') return v;
317
+ try { return JSON.stringify(v ?? {}); } catch { return '{}'; }
318
+ }
319
+
320
+ // ---------------- input parsers: client format -> IR ----------------
321
+
322
+ function baseIR() {
323
+ return {
324
+ model: '', system: '', messages: [], tools: [], toolChoice: null,
325
+ params: { maxTokens: null, temperature: null, topP: null, topK: null, stop: [] },
326
+ thinking: null, stream: false
327
+ };
328
+ }
329
+
330
+ // Anthropic Messages API -> IR (logic ported from the old transformAnthropicToOpenAI).
331
+ function anthropicToIR(payload) {
332
+ const ir = baseIR();
333
+ ir.model = payload.model || '';
334
+ // Anthropic API defaults to non-stream when the `stream` field is absent (the SDK leaves it empty for messages.create).
335
+ ir.stream = payload.stream === true;
336
+
337
+ if (payload.system) {
338
+ if (typeof payload.system === 'string') ir.system = payload.system;
339
+ else if (Array.isArray(payload.system)) {
340
+ ir.system = payload.system
341
+ .map(s => (typeof s === 'string' ? s : s.text || s.content || ''))
342
+ .filter(Boolean)
343
+ .join('\n\n');
344
+ }
345
+ }
346
+
347
+ if (Array.isArray(payload.messages)) {
348
+ for (const msg of payload.messages) {
349
+ if (!msg) continue;
350
+ if (typeof msg.content === 'string') {
351
+ ir.messages.push({ role: msg.role, content: msg.content });
352
+ continue;
353
+ }
354
+ if (Array.isArray(msg.content)) {
355
+ const parts = [];
356
+ const toolCalls = [];
357
+ for (const part of msg.content) {
358
+ if (!part) continue;
359
+ if (part.type === 'text') {
360
+ if (part.text) parts.push({ type: 'text', text: part.text });
361
+ } else if (part.type === 'image' && part.source?.type === 'base64') {
362
+ const mimeType = part.source.media_type || 'image/jpeg';
363
+ parts.push({ type: 'image_url', image_url: { url: `data:${mimeType};base64,${part.source.data}` } });
364
+ } else if (part.type === 'image' && part.source?.type === 'url' && part.source.url) {
365
+ parts.push({ type: 'image_url', image_url: { url: part.source.url } });
366
+ } else if (part.type === 'tool_use') {
367
+ toolCalls.push({ id: part.id, name: part.name, args: part.input ?? {} });
368
+ } else if (part.type === 'tool_result') {
369
+ let resultText = '';
370
+ if (typeof part.content === 'string') resultText = part.content;
371
+ else if (Array.isArray(part.content)) {
372
+ // Don't stuff image base64 into text (token blowup); keep only a placeholder.
373
+ resultText = part.content.map(c => {
374
+ if (typeof c === 'string') return c;
375
+ if (c?.type === 'text') return c.text || '';
376
+ if (c?.type === 'image') return '[image omitted]';
377
+ return JSON.stringify(c);
378
+ }).filter(Boolean).join('\n');
379
+ } else if (part.content) resultText = JSON.stringify(part.content);
380
+ if (part.is_error && !resultText.toLowerCase().startsWith('error')) {
381
+ resultText = `[Tool Error] ${resultText}`;
382
+ }
383
+ ir.messages.push({ role: 'tool', toolCallId: part.tool_use_id, content: resultText || '(empty tool output)' });
384
+ }
385
+ }
386
+ const out = { role: msg.role };
387
+ if (parts.length) {
388
+ out.content = parts.every(p => p.type === 'text') ? parts.map(p => p.text).join('\n') : parts;
389
+ }
390
+ if (toolCalls.length) out.toolCalls = toolCalls;
391
+ if (out.content !== undefined || out.toolCalls) ir.messages.push(out);
392
+ } else if (msg.content) {
393
+ // Fallback for payloads that preprocessing/compression tools mangled into an object
394
+ ir.messages.push({ role: msg.role, content: asText(msg.content) });
395
+ }
396
+ }
397
+ }
398
+
399
+ if (Array.isArray(payload.tools) && payload.tools.length) {
400
+ ir.tools = payload.tools.map(t => ({
401
+ name: t.name, description: t.description || '',
402
+ parameters: sanitizeJsonSchema(t.input_schema || {})
403
+ }));
404
+ }
405
+
406
+ if (payload.tool_choice) {
407
+ const tc = payload.tool_choice;
408
+ if (tc.type === 'auto') ir.toolChoice = 'auto';
409
+ else if (tc.type === 'any') ir.toolChoice = 'required';
410
+ else if (tc.type === 'none') ir.toolChoice = 'none';
411
+ else if (tc.type === 'tool' && tc.name) ir.toolChoice = { name: tc.name };
412
+ if (tc.disable_parallel_tool_use === true) ir.params.parallelToolCalls = false;
413
+ }
414
+
415
+ if (typeof payload.max_tokens === 'number') ir.params.maxTokens = payload.max_tokens;
416
+ if (typeof payload.temperature === 'number') ir.params.temperature = payload.temperature;
417
+ if (typeof payload.top_p === 'number') ir.params.topP = payload.top_p;
418
+ if (typeof payload.top_k === 'number') ir.params.topK = payload.top_k;
419
+ if (Array.isArray(payload.stop_sequences)) ir.params.stop = stopList(payload.stop_sequences);
420
+
421
+ ir.thinking = thinkingFromAnthropicParam(payload.thinking, payload.output_config?.effort);
422
+ return ir;
423
+ }
424
+
425
+ // Anthropic `thinking` param ({type, budget_tokens}) -> IR thinking ({type, budget, effort}).
426
+ function thinkingFromAnthropicParam(th, effort) {
427
+ if (!th || typeof th !== 'object') return null;
428
+ // Client explicitly disabled thinking -> honor that intent, don't "restore" thinking.
429
+ if (th.type === 'disabled') return { type: 'disabled' };
430
+ const out = { type: th.type === 'adaptive' ? 'adaptive' : 'enabled' };
431
+ const budget = Number(th.budget_tokens ?? th.budget);
432
+ if (Number.isFinite(budget) && budget > 0) out.budget = budget;
433
+ const eff = th.effort || effort;
434
+ if (typeof eff === 'string' && eff) out.effort = eff;
435
+ return out;
436
+ }
437
+
438
+ function isNoReasoningEffort(e) {
439
+ return String(e || '').toLowerCase() === 'none';
440
+ }
441
+
442
+ // OpenAI Chat Completions -> IR (light normalization).
443
+ // Anthropic rejects an empty stop string, and no upstream can match one.
444
+ function stopList(stop) {
445
+ return (Array.isArray(stop) ? stop : [stop]).filter(x => typeof x === 'string' && x);
446
+ }
447
+
448
+ // allowed_tools restricts the call to a subset of the declared tools. Hosted tools are not
449
+ // forwarded, so they never count as allowed.
450
+ function applyAllowedTools(ir, mode, list, nameOf) {
451
+ const names = new Set((Array.isArray(list) ? list : []).map(nameOf).filter(Boolean));
452
+ ir.tools = (ir.tools || []).filter(t => names.has(t.name));
453
+ ir.toolChoice = ir.tools.length ? (mode === 'required' ? 'required' : 'auto') : null;
454
+ }
455
+
456
+ function chatToIR(payload) {
457
+ const ir = baseIR();
458
+ ir.model = payload.model || '';
459
+ ir.stream = payload.stream === true;
460
+
461
+ if (Array.isArray(payload.messages)) {
462
+ for (const m of payload.messages) {
463
+ const role = m.role;
464
+ if (role === 'system' || role === 'developer') {
465
+ const t = typeof m.content === 'string' ? m.content : asText(m.content);
466
+ if (t) ir.system = ir.system ? `${ir.system}\n\n${t}` : t;
467
+ continue;
468
+ }
469
+ if (role === 'tool' || role === 'function') {
470
+ ir.messages.push({
471
+ role: 'tool', toolCallId: m.tool_call_id || m.id,
472
+ name: m.name, content: typeof m.content === 'string' ? m.content : asText(m.content)
473
+ });
474
+ continue;
475
+ }
476
+ const out = { role: role === 'assistant' ? 'assistant' : 'user' };
477
+ if (typeof m.content === 'string' || Array.isArray(m.content)) out.content = m.content;
478
+ else if (m.content != null) out.content = asText(m.content);
479
+ if (Array.isArray(m.tool_calls)) {
480
+ out.toolCalls = m.tool_calls.map((tc, i) => ({
481
+ id: tc.id || null,
482
+ name: tc.function?.name || tc.name || '',
483
+ args: parseArgs(tc.function?.arguments ?? tc.function?.args ?? tc.args)
484
+ })).filter(t => t.name);
485
+ }
486
+ if (out.content !== undefined || out.toolCalls) ir.messages.push(out);
487
+ }
488
+ }
489
+
490
+ const toolDefs = payload.tools;
491
+ if (Array.isArray(toolDefs) && toolDefs.length) {
492
+ ir.tools = toolDefs.map(t => {
493
+ const fn = t.function || t;
494
+ return { name: fn.name, description: fn.description || '', parameters: sanitizeJsonSchema(fn.parameters || fn.input_schema || {}) };
495
+ }).filter(t => t.name);
496
+ }
497
+
498
+ const tc = payload.tool_choice;
499
+ if (typeof tc === 'string') ir.toolChoice = tc;
500
+ else if (tc?.type === 'function' && tc.function?.name) ir.toolChoice = { name: tc.function.name };
501
+ else if (tc?.type === 'allowed_tools') applyAllowedTools(ir, tc.allowed_tools?.mode, tc.allowed_tools?.tools, t => t?.function?.name);
502
+
503
+ ir.params.maxTokens = payload.max_tokens ?? payload.max_completion_tokens ?? null;
504
+ if (typeof payload.temperature === 'number') ir.params.temperature = payload.temperature;
505
+ if (typeof payload.top_p === 'number') ir.params.topP = payload.top_p;
506
+ if (typeof payload.presence_penalty === 'number') ir.params.presencePenalty = payload.presence_penalty;
507
+ if (typeof payload.frequency_penalty === 'number') ir.params.frequencyPenalty = payload.frequency_penalty;
508
+ if (payload.stop !== undefined && payload.stop !== null) ir.params.stop = stopList(payload.stop);
509
+ if (payload.parallel_tool_calls === false) ir.params.parallelToolCalls = false;
510
+
511
+ if (payload.thinking && typeof payload.thinking === 'object') {
512
+ ir.thinking = thinkingFromAnthropicParam(payload.thinking, payload.reasoning_effort);
513
+ } else if (typeof payload.reasoning_effort === 'string') {
514
+ ir.thinking = isNoReasoningEffort(payload.reasoning_effort)
515
+ ? { type: 'disabled' }
516
+ : { type: 'enabled', budget: effortToBudget(payload.reasoning_effort), effort: payload.reasoning_effort };
517
+ } else if (payload.reasoning && typeof payload.reasoning === 'object') {
518
+ ir.thinking = thinkingFromReasoningParam(payload.reasoning);
519
+ }
520
+ return ir;
521
+ }
522
+
523
+ // OpenAI/OpenRouter `reasoning` object -> IR thinking.
524
+ function thinkingFromReasoningParam(r) {
525
+ if (!r || typeof r !== 'object') return null;
526
+ if (r.exclude === true || r.enabled === false || isNoReasoningEffort(r.effort)) return { type: 'disabled' };
527
+ return {
528
+ type: 'enabled',
529
+ budget: r.max_tokens ? clampBudget(r.max_tokens) : effortToBudget(r.effort),
530
+ effort: r.effort
531
+ };
532
+ }
533
+
534
+ // ---- Codex tool kinds ----
535
+ // Codex declares tools as: function | custom (freeform, e.g. apply_patch with Lark grammar) | namespace
536
+ // (wrapping function/custom) | local_shell (legacy) | hosted (web_search, tool_search...).
537
+ // Non-OpenAI upstreams only understand function tools, so:
538
+ // - custom -> function with a single string `input` param (grammar goes into the description), returns custom_tool_call;
539
+ // - namespace -> flat name "<namespace>__<name>", returns function_call with `namespace`;
540
+ // - local_shell -> function {command[], workdir, timeout_ms}, returns local_shell_call;
541
+ // - hosted tools run server-side by OpenAI -> can't be forwarded, skipped.
542
+ const CUSTOM_TOOL_PARAMS = {
543
+ type: 'object',
544
+ properties: { input: { type: 'string', description: 'The raw freeform tool input (not JSON-encoded).' } },
545
+ required: ['input'],
546
+ additionalProperties: false
547
+ };
548
+ const LOCAL_SHELL_PARAMS = {
549
+ type: 'object',
550
+ properties: {
551
+ command: { type: 'array', items: { type: 'string' }, description: 'Command and arguments, e.g. ["bash", "-lc", "ls -la"].' },
552
+ workdir: { type: 'string', description: 'Working directory for the command.' },
553
+ timeout_ms: { type: 'number', description: 'Timeout in milliseconds.' }
554
+ },
555
+ required: ['command']
556
+ };
557
+
558
+ function responsesToolName(namespace, name) {
559
+ const raw = namespace ? `${namespace}__${name}` : String(name || '');
560
+ return raw.replace(/[^A-Za-z0-9_-]/g, '_').slice(0, 64);
561
+ }
562
+
563
+ function customToolDescription(t) {
564
+ let d = t.description || '';
565
+ d += `${d ? '\n\n' : ''}This is a FREEFORM tool: pass the raw input text in the "input" string argument, exactly as the tool expects it (do not wrap it in JSON).`;
566
+ if (t.format?.type === 'grammar' && t.format.definition) {
567
+ d += `\nThe input must match this ${t.format.syntax || ''} grammar:\n${t.format.definition}`;
568
+ }
569
+ return d;
570
+ }
571
+
572
+ // function_call_output / custom_tool_call_output: `output` is a string or an array of content items.
573
+ function responsesOutputToText(out) {
574
+ if (typeof out === 'string') return out;
575
+ if (Array.isArray(out)) {
576
+ return out.map(x => {
577
+ if (typeof x === 'string') return x;
578
+ if (typeof x?.text === 'string') return x.text;
579
+ if (x?.type === 'input_image') return '[image]';
580
+ return JSON.stringify(x);
581
+ }).join('\n');
582
+ }
583
+ return out == null ? '' : JSON.stringify(out);
584
+ }
585
+
586
+ // OpenAI Responses API (Codex) -> IR.
587
+ function responsesToIR(payload) {
588
+ const ir = baseIR();
589
+ ir.model = payload.model || '';
590
+ ir.stream = payload.stream === true;
591
+ if (typeof payload.instructions === 'string' && payload.instructions) ir.system = payload.instructions;
592
+
593
+ const pushText = (role, text) => {
594
+ if (!text) return;
595
+ const last = ir.messages[ir.messages.length - 1];
596
+ if (last && last.role === role && typeof last.content === 'string' && !last.toolCalls) {
597
+ last.content += '\n' + text;
598
+ } else {
599
+ ir.messages.push({ role, content: text });
600
+ }
601
+ };
602
+
603
+ const appendSystem = (text) => {
604
+ if (text) ir.system = ir.system ? `${ir.system}\n\n${text}` : text;
605
+ };
606
+
607
+ // Parallel tool calls arrive as consecutive items -> merge into 1 assistant turn,
608
+ // otherwise OpenAI Chat/Anthropic will error that tool_calls aren't answered adjacently.
609
+ const pushCall = (call) => {
610
+ const last = ir.messages[ir.messages.length - 1];
611
+ if (last && last.role === 'assistant') last.toolCalls = [...(last.toolCalls || []), call];
612
+ else ir.messages.push({ role: 'assistant', toolCalls: [call] });
613
+ };
614
+
615
+ const input = payload.input;
616
+ if (typeof input === 'string' && input) {
617
+ ir.messages.push({ role: 'user', content: input });
618
+ } else if (Array.isArray(input)) {
619
+ for (const item of input) {
620
+ if (typeof item === 'string') { pushText('user', item); continue; }
621
+ if (!item || typeof item !== 'object') continue;
622
+ // EasyInputMessage ({role, content}) doesn't require `type`.
623
+ if (item.type === 'message' || (!item.type && item.role)) {
624
+ const c = item.content;
625
+ const parts = [];
626
+ if (typeof c === 'string') {
627
+ if (c) parts.push({ type: 'text', text: c });
628
+ } else if (Array.isArray(c)) {
629
+ for (const p of c) {
630
+ if (typeof p === 'string') { if (p) parts.push({ type: 'text', text: p }); continue; }
631
+ if (!p || typeof p !== 'object') continue;
632
+ if (p.type === 'input_image' && (p.image_url || p.url)) {
633
+ parts.push({ type: 'image_url', image_url: { url: typeof p.image_url === 'string' ? p.image_url : (p.image_url?.url || p.url) } });
634
+ } else if (typeof p.text === 'string' && p.text) {
635
+ parts.push({ type: 'text', text: p.text });
636
+ }
637
+ }
638
+ }
639
+ if (item.role === 'system' || item.role === 'developer') {
640
+ appendSystem(parts.filter(p => p.type === 'text').map(p => p.text).join('\n'));
641
+ continue;
642
+ }
643
+ const role = item.role === 'assistant' ? 'assistant' : 'user';
644
+ if (parts.every(p => p.type === 'text')) {
645
+ pushText(role, parts.map(p => p.text).join('\n'));
646
+ } else {
647
+ ir.messages.push({ role, content: parts });
648
+ }
649
+ } else if (item.type === 'function_call') {
650
+ pushCall({ id: item.call_id || item.id || null, name: responsesToolName(item.namespace, item.name), args: parseArgs(item.arguments) });
651
+ } else if (item.type === 'custom_tool_call') {
652
+ pushCall({ id: item.call_id || item.id || null, name: responsesToolName(item.namespace, item.name), args: { input: typeof item.input === 'string' ? item.input : responsesOutputToText(item.input) } });
653
+ } else if (item.type === 'local_shell_call') {
654
+ const a = item.action || {};
655
+ pushCall({ id: item.call_id || item.id || null, name: 'local_shell', args: { command: a.command || [], workdir: a.working_directory ?? undefined, timeout_ms: a.timeout_ms ?? undefined } });
656
+ } else if (item.type === 'function_call_output' || item.type === 'custom_tool_call_output' || item.type === 'local_shell_call_output') {
657
+ ir.messages.push({ role: 'tool', toolCallId: item.call_id || item.id, content: responsesOutputToText(item.output) });
658
+ }
659
+ }
660
+ }
661
+
662
+ if (Array.isArray(payload.tools) && payload.tools.length) {
663
+ ir.toolMeta = {};
664
+ const addTool = (t, namespace) => {
665
+ if (!t || typeof t !== 'object') return;
666
+ if (t.type === 'namespace' && Array.isArray(t.tools)) {
667
+ for (const inner of t.tools) addTool(inner, t.name);
668
+ return;
669
+ }
670
+ if (t.type === 'function' && t.name) {
671
+ const name = responsesToolName(namespace, t.name);
672
+ ir.tools.push({ name, description: t.description || '', parameters: sanitizeJsonSchema(t.parameters || {}) });
673
+ ir.toolMeta[name] = { kind: 'function', name: t.name, namespace: namespace || null };
674
+ } else if (t.type === 'custom' && t.name) {
675
+ const name = responsesToolName(namespace, t.name);
676
+ ir.tools.push({ name, description: customToolDescription(t), parameters: CUSTOM_TOOL_PARAMS });
677
+ ir.toolMeta[name] = { kind: 'custom', name: t.name, namespace: namespace || null };
678
+ } else if (t.type === 'local_shell') {
679
+ ir.tools.push({ name: 'local_shell', description: 'Run a shell command on the user\'s machine and return its output.', parameters: LOCAL_SHELL_PARAMS });
680
+ ir.toolMeta.local_shell = { kind: 'local_shell', name: 'local_shell', namespace: null };
681
+ }
682
+ // web_search / file_search / tool_search / image_generation...: hosted tools, skipped.
683
+ };
684
+ for (const t of payload.tools) addTool(t, null);
685
+ }
686
+
687
+ const tc = payload.tool_choice;
688
+ if (typeof tc === 'string') ir.toolChoice = tc;
689
+ else if ((tc?.type === 'function' || tc?.type === 'custom') && tc.name) ir.toolChoice = { name: responsesToolName(tc.namespace, tc.name) };
690
+ else if (tc?.type === 'allowed_tools') applyAllowedTools(ir, tc.mode, tc.tools, t => (t?.name ? responsesToolName(t.namespace, t.name) : null));
691
+ else if (['none', 'auto', 'required'].includes(tc?.type)) ir.toolChoice = tc.type;
692
+ // A hosted tool (web search, file search) is not forwarded, so it cannot be forced.
693
+ else if (tc && typeof tc === 'object') ir.toolChoice = 'auto';
694
+
695
+ if (typeof payload.max_output_tokens === 'number') ir.params.maxTokens = payload.max_output_tokens;
696
+ if (typeof payload.temperature === 'number') ir.params.temperature = payload.temperature;
697
+ if (typeof payload.top_p === 'number') ir.params.topP = payload.top_p;
698
+ if (payload.parallel_tool_calls === false) ir.params.parallelToolCalls = false;
699
+
700
+ if (payload.reasoning && typeof payload.reasoning === 'object') {
701
+ ir.thinking = thinkingFromReasoningParam(payload.reasoning);
702
+ }
703
+ return ir;
704
+ }
705
+
706
+ // Vertex generateContent -> IR.
707
+ function vertexToIR(payload) {
708
+ const ir = baseIR();
709
+ ir.model = payload.model || '';
710
+ ir.stream = false; // proxy overrides per the :streamGenerateContent endpoint
711
+
712
+ const sys = payload.systemInstruction?.parts;
713
+ if (Array.isArray(sys)) {
714
+ ir.system = sys.map(p => p.text || '').filter(Boolean).join('\n\n');
715
+ } else if (typeof payload.system_instruction === 'string') {
716
+ ir.system = payload.system_instruction;
717
+ }
718
+
719
+ // Vertex has no tool call ids: auto-generate ids for functionCalls then match functionResponses by name (FIFO),
720
+ // so upstream OpenAI/Anthropic gets the correct tool_call <-> tool result pairs instead of "healing" them into text.
721
+ const pendingByName = new Map();
722
+ let callSeq = 0;
723
+
724
+ if (Array.isArray(payload.contents)) {
725
+ for (const c of payload.contents) {
726
+ const parts = Array.isArray(c?.parts) ? c.parts : [];
727
+ const texts = [];
728
+ const toolCalls = [];
729
+ for (const p of parts) {
730
+ if (!p || typeof p !== 'object') continue;
731
+ if (typeof p.text === 'string' && p.text && p.thought !== true) texts.push(p.text);
732
+ if (p.functionCall) {
733
+ const name = p.functionCall.name || '';
734
+ const id = p.functionCall.id || `call_vtx_${callSeq++}_${name}`;
735
+ if (!pendingByName.has(name)) pendingByName.set(name, []);
736
+ pendingByName.get(name).push(id);
737
+ const sig = p.thoughtSignature || p.thought_signature || null;
738
+ rememberToolSignature(id, sig);
739
+ toolCalls.push({ id, name, args: p.functionCall.args ?? {}, sig });
740
+ }
741
+ if (p.functionResponse) {
742
+ const fr = p.functionResponse;
743
+ const name = fr.name || '';
744
+ const queue = pendingByName.get(name);
745
+ const id = fr.id || (queue && queue.length ? queue.shift() : `call_vtx_orphan_${name}`);
746
+ ir.messages.push({
747
+ role: 'tool', toolCallId: id, name: name || null,
748
+ content: typeof fr.response === 'string' ? fr.response : JSON.stringify(fr.response ?? '')
749
+ });
750
+ }
751
+ }
752
+ if (c.role === 'model') {
753
+ const out = { role: 'assistant' };
754
+ if (texts.length) out.content = texts.join('');
755
+ if (toolCalls.length) out.toolCalls = toolCalls;
756
+ if (out.content !== undefined || out.toolCalls) ir.messages.push(out);
757
+ } else if (c.role === 'function') {
758
+ // functionResponse already pushed above
759
+ } else {
760
+ if (texts.length) ir.messages.push({ role: 'user', content: texts.join('') });
761
+ }
762
+ }
763
+ }
764
+
765
+ const fns = payload.tools?.flatMap(t => t.functionDeclarations || t.function_declarations || []);
766
+ if (Array.isArray(fns) && fns.length) {
767
+ ir.tools = fns.filter(f => f.name).map(f => ({
768
+ name: f.name, description: f.description || '',
769
+ parameters: sanitizeJsonSchema(f.parameters || {})
770
+ }));
771
+ }
772
+
773
+ const gc = payload.generationConfig || payload.generation_config || {};
774
+ if (typeof gc.maxOutputTokens === 'number') ir.params.maxTokens = gc.maxOutputTokens;
775
+ if (typeof gc.temperature === 'number') ir.params.temperature = gc.temperature;
776
+ if (typeof gc.topP === 'number') ir.params.topP = gc.topP;
777
+ if (Array.isArray(gc.stopSequences)) ir.params.stop = stopList(gc.stopSequences);
778
+ const th = gc.thinkingConfig || gc.thinking_config;
779
+ if (th && typeof th === 'object') {
780
+ if (th.thinkingBudget === 0) ir.thinking = { type: 'disabled' };
781
+ else if (th.thinkingBudget > 0) ir.thinking = { type: 'enabled', budget: clampBudget(th.thinkingBudget) };
782
+ else if (th.thinkingLevel) ir.thinking = { type: 'enabled', budget: effortToBudget(th.thinkingLevel), effort: th.thinkingLevel };
783
+ else if (th.includeThoughts) ir.thinking = { type: 'enabled', budget: 2048 };
784
+ }
785
+ return ir;
786
+ }
787
+
788
+ function parseToIR(clientFormat, payload) {
789
+ switch (clientFormat) {
790
+ case 'anthropic': return anthropicToIR(payload);
791
+ case 'openai-chat': return chatToIR(payload);
792
+ case 'responses': return responsesToIR(payload);
793
+ case 'vertex': return vertexToIR(payload);
794
+ default: throw new Error(`Unknown client format: ${clientFormat}`);
795
+ }
796
+ }
797
+
798
+ // ---------------- emitters: IR -> upstream body ----------------
799
+
800
+ function hasNativeReasoning(model) {
801
+ // Detect capability families instead of pinning exact versions. Claude Opus
802
+ // model IDs change often, but all current Opus variants support reasoning.
803
+ const m = String(model || '').toLowerCase();
804
+ return m.includes('thinking') || m.includes('reasoning') || m.includes('reasoner') || m.includes('opus')
805
+ || /(^|[/_.:-])(o[134]|r1|qwq)([/_.:-]|$)/.test(m);
806
+ }
807
+
808
+ // Healer Engine: normalize tool call <-> tool result pairs before emitting to any upstream.
809
+ //
810
+ // Token-compressing tools (RTK, Headroom, Ponytail) or history trimming break tool pairs:
811
+ // - orphaned tool result (assistant turn holding the tool call was deleted)
812
+ // -> Anthropic: "tool_use_id does not correspond to any tool_use"
813
+ // -> OpenAI: "messages with role 'tool' must be a response to a preceding message with 'tool_calls'"
814
+ // - tool call with no immediately following result (result deleted or another message inserted in between)
815
+ // -> Anthropic: "tool_use ids were found without tool_result blocks immediately after"
816
+ // -> OpenAI: "assistant message with 'tool_calls' must be followed by tool messages"
817
+ //
818
+ // Rule: a tool result is only valid when placed right after the assistant turn that declared it. Orphaned results are
819
+ // converted to user text (preserving context); tool calls missing a result get a placeholder result.
820
+ const MISSING_TOOL_RESULT = '[Tool result unavailable: it was removed from the conversation history]';
821
+
822
+ function healToolPairs(messages) {
823
+ const out = [];
824
+ let pending = null; // Map<callId, name> of the most recent assistant turn
825
+ let deferred = []; // orphaned results seen while awaiting results -> flushed later to preserve adjacency
826
+ let seq = 0;
827
+
828
+ const orphanText = (m) => ({
829
+ role: 'user',
830
+ content: `[Tool Result${m.toolCallId ? ` (${m.toolCallId})` : ''}]: ${m.content ?? ''}`
831
+ });
832
+ const flush = () => {
833
+ if (pending) {
834
+ for (const [id, name] of pending) {
835
+ out.push({ role: 'tool', toolCallId: id, name, content: MISSING_TOOL_RESULT });
836
+ }
837
+ pending = null;
838
+ }
839
+ for (const m of deferred) out.push(orphanText(m));
840
+ deferred = [];
841
+ };
842
+
843
+ let unnamed = []; // ids given to calls that arrived with no id, in order
844
+
845
+ for (const m of messages) {
846
+ if (!m) continue;
847
+ if (m.role === 'tool') {
848
+ // A result with no id answers the next call that also had none.
849
+ const id = m.toolCallId || (pending && unnamed.find(u => pending.has(u)));
850
+ if (pending && id && pending.has(id)) {
851
+ const name = pending.get(id);
852
+ pending.delete(id);
853
+ out.push({ ...m, toolCallId: id, name: m.name || name });
854
+ } else if (pending) {
855
+ deferred.push(m);
856
+ } else {
857
+ out.push(orphanText(m));
858
+ }
859
+ continue;
860
+ }
861
+ flush();
862
+ if (m.role === 'assistant' && Array.isArray(m.toolCalls) && m.toolCalls.length) {
863
+ unnamed = [];
864
+ const toolCalls = m.toolCalls
865
+ .filter(tc => tc && tc.name)
866
+ .map(tc => {
867
+ if (tc.id) return tc;
868
+ const id = `call_heal_${seq++}`;
869
+ unnamed.push(id);
870
+ return { ...tc, id };
871
+ });
872
+ const msg = { ...m, toolCalls };
873
+ if (!toolCalls.length) delete msg.toolCalls;
874
+ out.push(msg);
875
+ if (toolCalls.length) pending = new Map(toolCalls.map(tc => [tc.id, tc.name]));
876
+ continue;
877
+ }
878
+ out.push(m);
879
+ }
880
+ flush();
881
+ return out;
882
+ }
883
+
884
+ export const THINKING_MODES = ['auto', 'native', 'off'];
885
+
886
+ function normThinkingMode(mode) {
887
+ return THINKING_MODES.includes(mode) ? mode : 'auto';
888
+ }
889
+
890
+ // IR -> OpenAI Chat Completions body.
891
+ //
892
+ // opts.thinkingMode (per profile):
893
+ // - 'auto' (default, for gateways like 9Router): restore thinking for reasoning models when compression tools
894
+ // drop the param, inject a <think> guide for models without native reasoning, send both
895
+ // `thinking` and `reasoning_effort`.
896
+ // - 'native' (genuine OpenAI / strict OpenAI-compatible servers): only send `reasoning_effort` when
897
+ // the client requests thinking, don't touch the system prompt, use `max_completion_tokens`.
898
+ // - 'off' : never send reasoning params, never inject prompts.
899
+ function irToChatBody(ir, model, opts = {}) {
900
+ const messages = [];
901
+ const mode = normThinkingMode(opts.thinkingMode);
902
+ const clientWantsThinking = Boolean(ir.thinking && ir.thinking.type !== 'disabled');
903
+ const modelIsThinking = hasNativeReasoning(model);
904
+ const wantsThinking = mode === 'off' ? false
905
+ : mode === 'native' ? clientWantsThinking
906
+ : Boolean(clientWantsThinking || (modelIsThinking && (!ir.thinking || ir.thinking.type !== 'disabled')));
907
+
908
+ let systemText = ir.system || '';
909
+ if (mode === 'auto' && wantsThinking && !modelIsThinking && !String(model).toLowerCase().includes('gemini')) {
910
+ const thinkGuide = 'You must provide your internal reasoning and step-by-step thinking inside <think> and </think> tags before your final response.';
911
+ systemText = systemText ? `${thinkGuide}\n\n${systemText}` : thinkGuide;
912
+ }
913
+ if (systemText.trim()) messages.push({ role: 'system', content: systemText });
914
+
915
+ for (const m of healToolPairs(ir.messages)) {
916
+ if (m.role === 'tool') {
917
+ messages.push({ role: 'tool', tool_call_id: m.toolCallId, content: m.content ?? '' });
918
+ continue;
919
+ }
920
+ const out = { role: m.role === 'assistant' ? 'assistant' : 'user' };
921
+ if (m.content !== undefined) out.content = m.content;
922
+ if (m.role === 'assistant' && Array.isArray(m.toolCalls) && m.toolCalls.length) {
923
+ out.tool_calls = m.toolCalls.map(tc => {
924
+ const call = { id: tc.id, type: 'function', function: { name: tc.name, arguments: stringifyArgs(tc.args) } };
925
+ const sig = tc.sig || lookupToolSignature(tc.id);
926
+ if (sig) call.extra_content = { google: { thought_signature: sig } };
927
+ return call;
928
+ });
929
+ if (out.content === undefined || out.content === '') out.content = null;
930
+ }
931
+ if (out.content === undefined) out.content = '';
932
+ messages.push(out);
933
+ }
934
+
935
+ const body = { model, messages, stream: ir.stream === true };
936
+ if (body.stream) body.stream_options = { include_usage: true };
937
+ if (ir.tools.length) {
938
+ body.tools = ir.tools.map(t => ({
939
+ type: 'function',
940
+ function: { name: t.name, description: t.description || '', parameters: sanitizeJsonSchema(t.parameters || {}) }
941
+ }));
942
+ }
943
+ if (ir.toolChoice) {
944
+ if (typeof ir.toolChoice === 'string') {
945
+ body.tool_choice = ir.toolChoice === 'required' ? 'required' : ir.toolChoice;
946
+ } else if (ir.toolChoice.name) {
947
+ body.tool_choice = { type: 'function', function: { name: ir.toolChoice.name } };
948
+ }
949
+ }
950
+ if (typeof ir.params.maxTokens === 'number') {
951
+ if (mode === 'native') body.max_completion_tokens = ir.params.maxTokens;
952
+ else body.max_tokens = ir.params.maxTokens;
953
+ }
954
+ if (typeof ir.params.temperature === 'number') body.temperature = ir.params.temperature;
955
+ if (typeof ir.params.topP === 'number') body.top_p = ir.params.topP;
956
+ if (typeof ir.params.presencePenalty === 'number') body.presence_penalty = ir.params.presencePenalty;
957
+ if (typeof ir.params.frequencyPenalty === 'number') body.frequency_penalty = ir.params.frequencyPenalty;
958
+ if (ir.params.stop.length) body.stop = ir.params.stop;
959
+ if (ir.params.parallelToolCalls === false && body.tools) body.parallel_tool_calls = false;
960
+
961
+ if (wantsThinking && mode === 'native') {
962
+ body.reasoning_effort = ir.thinking?.effort && !['max', 'xhigh'].includes(ir.thinking.effort)
963
+ ? ir.thinking.effort
964
+ : (ir.thinking?.type === 'adaptive' ? 'high' : (ir.thinking?.budget ? budgetToEffort(ir.thinking.budget) : 'medium'));
965
+ } else if (wantsThinking) {
966
+ if (ir.thinking?.type === 'adaptive') {
967
+ body.thinking = { type: 'adaptive' };
968
+ body.reasoning_effort = ir.thinking.effort || 'high';
969
+ } else {
970
+ // Restore thinking: if an external compression tool (RTK/Headroom) dropped the thinking object,
971
+ // but the target model is a reasoning model, the gateway auto re-enables thinking with a safe budget.
972
+ const rawBudget = ir.thinking?.budget ?? (ir.thinking?.effort ? effortToBudget(ir.thinking.effort) : 2048);
973
+ const safeBudget = clampBudget(rawBudget);
974
+ body.thinking = { type: 'enabled', budget_tokens: safeBudget };
975
+ body.reasoning_effort = ir.thinking?.effort || budgetToEffort(safeBudget);
976
+ }
977
+ }
978
+ return body;
979
+ }
980
+
981
+ function urlToAnthropicImage(url) {
982
+ const s = String(url || '');
983
+ const m = s.match(/^data:([^;]+);base64,(.+)$/s);
984
+ if (m) return { type: 'image', source: { type: 'base64', media_type: m[1], data: m[2] } };
985
+ if (/^https?:\/\//i.test(s)) return { type: 'image', source: { type: 'url', url: s } };
986
+ return null;
987
+ }
988
+
989
+ function contentToAnthropicBlocks(c) {
990
+ if (typeof c === 'string') return c ? [{ type: 'text', text: c }] : [];
991
+ if (!Array.isArray(c)) return [];
992
+ const blocks = [];
993
+ for (const p of c) {
994
+ if (!p) continue;
995
+ // Anthropic rejects empty text blocks ("text content blocks must be non-empty").
996
+ if (p.type === 'text' && p.text) blocks.push({ type: 'text', text: p.text });
997
+ else if (p.type === 'image_url') {
998
+ const img = urlToAnthropicImage(typeof p.image_url === 'string' ? p.image_url : p.image_url?.url);
999
+ if (img) blocks.push(img);
1000
+ }
1001
+ }
1002
+ return blocks;
1003
+ }
1004
+
1005
+ // IR -> Anthropic Messages body.
1006
+ function irToAnthropicBody(ir, model) {
1007
+ const messages = [];
1008
+
1009
+ for (const m of healToolPairs(ir.messages)) {
1010
+ if (m.role === 'tool') {
1011
+ messages.push({
1012
+ role: 'user',
1013
+ content: [{ type: 'tool_result', tool_use_id: m.toolCallId, content: m.content ?? '' }]
1014
+ });
1015
+ continue;
1016
+ }
1017
+ if (m.role === 'assistant') {
1018
+ const blocks = contentToAnthropicBlocks(m.content).filter(b => b.type === 'text');
1019
+ for (const tc of (m.toolCalls || [])) {
1020
+ blocks.push({ type: 'tool_use', id: tc.id, name: tc.name, input: parseArgs(tc.args) });
1021
+ }
1022
+ if (blocks.length) messages.push({ role: 'assistant', content: blocks });
1023
+ continue;
1024
+ }
1025
+ const blocks = contentToAnthropicBlocks(m.content);
1026
+ if (blocks.length) messages.push({ role: 'user', content: blocks });
1027
+ }
1028
+
1029
+ // Normalize messages for the Anthropic API:
1030
+ // 1. Merge consecutive same-role messages (consecutive user-user or assistant-assistant).
1031
+ // 2. Ensure the first message is always 'user'.
1032
+ // 3. Ensure there is at least 1 message.
1033
+ const normalizedMessages = [];
1034
+ const toBlocks = (c) => typeof c === 'string' ? (c ? [{ type: 'text', text: c }] : []) : Array.isArray(c) ? c : [];
1035
+
1036
+ for (const msg of messages) {
1037
+ const blocks = toBlocks(msg.content);
1038
+ if (!blocks.length) continue;
1039
+ const last = normalizedMessages[normalizedMessages.length - 1];
1040
+ if (last && last.role === msg.role) {
1041
+ last.content = [...toBlocks(last.content), ...blocks];
1042
+ } else {
1043
+ normalizedMessages.push({ role: msg.role, content: blocks });
1044
+ }
1045
+ }
1046
+
1047
+ if (normalizedMessages.length > 0 && normalizedMessages[0].role === 'assistant') {
1048
+ normalizedMessages.unshift({ role: 'user', content: [{ type: 'text', text: 'Hello' }] });
1049
+ }
1050
+ if (normalizedMessages.length === 0) {
1051
+ normalizedMessages.push({ role: 'user', content: [{ type: 'text', text: 'Hello' }] });
1052
+ }
1053
+
1054
+ const body = {
1055
+ model,
1056
+ max_tokens: (typeof ir.params.maxTokens === 'number' ? ir.params.maxTokens : 4096),
1057
+ messages: normalizedMessages,
1058
+ stream: ir.stream === true
1059
+ };
1060
+ if (ir.system && ir.system.trim()) body.system = ir.system;
1061
+ if (ir.tools.length) {
1062
+ body.tools = ir.tools.map(t => ({
1063
+ name: t.name, description: t.description || '',
1064
+ input_schema: sanitizeJsonSchema(t.parameters || {})
1065
+ }));
1066
+ }
1067
+ if (ir.toolChoice) {
1068
+ if (typeof ir.toolChoice === 'string') {
1069
+ body.tool_choice = ir.toolChoice === 'required' ? { type: 'any' } : { type: ir.toolChoice };
1070
+ } else if (ir.toolChoice.name) {
1071
+ body.tool_choice = { type: 'tool', name: ir.toolChoice.name };
1072
+ }
1073
+ }
1074
+ if (ir.params.parallelToolCalls === false && body.tool_choice) body.tool_choice.disable_parallel_tool_use = true;
1075
+ if (typeof ir.params.temperature === 'number') body.temperature = ir.params.temperature;
1076
+ if (typeof ir.params.topP === 'number') body.top_p = ir.params.topP;
1077
+ if (typeof ir.params.topK === 'number') body.top_k = ir.params.topK;
1078
+ if (ir.params.stop.length) body.stop_sequences = ir.params.stop;
1079
+ if (ir.thinking && ir.thinking.type === 'adaptive') {
1080
+ body.thinking = { type: 'adaptive' };
1081
+ } else if (ir.thinking && ir.thinking.type !== 'disabled' && body.max_tokens > 1024) {
1082
+ // Anthropic requires 1024 <= budget_tokens < max_tokens; drop thinking when max_tokens is too small.
1083
+ const budget = clampBudget(ir.thinking.budget ?? (ir.thinking.effort ? effortToBudget(ir.thinking.effort) : 4096));
1084
+ body.thinking = { type: 'enabled', budget_tokens: Math.min(budget, body.max_tokens - 1) };
1085
+ }
1086
+ if (body.thinking) {
1087
+ // When thinking is on, Anthropic rejects temperature != 1, top_k, or top_p < 0.95.
1088
+ if (body.temperature !== undefined && body.temperature !== 1) delete body.temperature;
1089
+ delete body.top_k;
1090
+ if (body.top_p !== undefined && body.top_p < 0.95) delete body.top_p;
1091
+ // tool_choice any/tool is incompatible with extended thinking.
1092
+ if (body.tool_choice && (body.tool_choice.type === 'any' || body.tool_choice.type === 'tool')) {
1093
+ body.tool_choice = { type: 'auto', ...(body.tool_choice.disable_parallel_tool_use ? { disable_parallel_tool_use: true } : {}) };
1094
+ }
1095
+ }
1096
+ return body;
1097
+ }
1098
+
1099
+ function dataUrlToInlineData(url) {
1100
+ const m = String(url || '').match(/^data:([^;]+);base64,(.+)$/s);
1101
+ if (!m) return null;
1102
+ return { inlineData: { mimeType: m[1], data: m[2] } };
1103
+ }
1104
+
1105
+ // IR -> Vertex generateContent body.
1106
+ function irToVertexBody(ir, model) {
1107
+ const contents = [];
1108
+ const textPartsOf = (c) => {
1109
+ if (typeof c === 'string') return c ? [{ text: c }] : [];
1110
+ if (Array.isArray(c)) {
1111
+ const parts = [];
1112
+ for (const p of c) {
1113
+ if (p.type === 'text' && p.text) parts.push({ text: p.text });
1114
+ else if (p.type === 'image_url') {
1115
+ const inline = dataUrlToInlineData(p.image_url?.url);
1116
+ if (inline) parts.push(inline);
1117
+ }
1118
+ }
1119
+ return parts;
1120
+ }
1121
+ return [];
1122
+ };
1123
+
1124
+ // Merge adjacent contents of the same kind: Gemini requires all functionResponses of one parallel-call round
1125
+ // to share a single content (functionResponse count must match the previous functionCall count).
1126
+ const push = (role, parts) => {
1127
+ if (!parts.length) return;
1128
+ const last = contents[contents.length - 1];
1129
+ const isFnResp = parts.some(p => p.functionResponse);
1130
+ const lastIsFnResp = last?.parts.some(p => p.functionResponse);
1131
+ if (last && last.role === role && isFnResp === lastIsFnResp) last.parts.push(...parts);
1132
+ else contents.push({ role, parts });
1133
+ };
1134
+
1135
+ const healed = healToolPairs(ir.messages);
1136
+ // The "current turn" starts at the last user message with real content (not a functionResponse).
1137
+ let turnStart = -1;
1138
+ healed.forEach((m, i) => { if (m.role === 'user') turnStart = i; });
1139
+ const needSig = geminiRequiresSignatures(model);
1140
+
1141
+ for (const [i, m] of healed.entries()) {
1142
+ if (m.role === 'tool') {
1143
+ let resp;
1144
+ try {
1145
+ const parsed = typeof m.content === 'string' ? JSON.parse(m.content) : m.content;
1146
+ // functionResponse.response must be an object (Struct), not an array/primitive.
1147
+ resp = (parsed && typeof parsed === 'object' && !Array.isArray(parsed)) ? parsed : { result: parsed ?? '' };
1148
+ } catch { resp = { result: m.content ?? '' }; }
1149
+ // Gemini API & Vertex only accept roles 'user' | 'model' (functionResponse lives in the 'user' role).
1150
+ push('user', [{ functionResponse: { name: m.name || 'tool', response: resp } }]);
1151
+ continue;
1152
+ }
1153
+ if (m.role === 'assistant') {
1154
+ const parts = textPartsOf(m.content).filter(p => p.text);
1155
+ (m.toolCalls || []).forEach((tc, j) => {
1156
+ const part = { functionCall: { name: tc.name, args: parseArgs(tc.args) } };
1157
+ const sig = tc.sig || lookupToolSignature(tc.id);
1158
+ if (sig) part.thoughtSignature = sig;
1159
+ // Only the first functionCall of each step in the current turn is checked.
1160
+ else if (needSig && j === 0 && i > turnStart) part.thoughtSignature = GEMINI_DUMMY_SIGNATURE;
1161
+ parts.push(part);
1162
+ });
1163
+ push('model', parts);
1164
+ continue;
1165
+ }
1166
+ push('user', textPartsOf(m.content));
1167
+ }
1168
+
1169
+ const body = { contents };
1170
+ if (ir.system && ir.system.trim()) {
1171
+ body.systemInstruction = { parts: [{ text: ir.system }] };
1172
+ }
1173
+ if (ir.tools.length) {
1174
+ body.tools = [{
1175
+ functionDeclarations: ir.tools.map(t => ({
1176
+ name: t.name, description: t.description || '',
1177
+ parameters: toGeminiSchema(t.parameters || { type: 'object', properties: {} })
1178
+ }))
1179
+ }];
1180
+ }
1181
+ if (ir.toolChoice) {
1182
+ const mode = ir.toolChoice === 'none' ? 'NONE' : ir.toolChoice === 'auto' ? 'AUTO' : 'ANY';
1183
+ const fcc = { mode };
1184
+ if (typeof ir.toolChoice === 'object' && ir.toolChoice.name) fcc.allowedFunctionNames = [ir.toolChoice.name];
1185
+ if (body.tools) body.toolConfig = { functionCallingConfig: fcc };
1186
+ }
1187
+ const gc = {};
1188
+ if (typeof ir.params.temperature === 'number') gc.temperature = ir.params.temperature;
1189
+ if (typeof ir.params.topP === 'number') gc.topP = ir.params.topP;
1190
+ if (typeof ir.params.topK === 'number') gc.topK = ir.params.topK;
1191
+ if (typeof ir.params.maxTokens === 'number') gc.maxOutputTokens = ir.params.maxTokens;
1192
+ if (ir.params.stop.length) gc.stopSequences = ir.params.stop;
1193
+ if (ir.thinking && ir.thinking.type !== 'disabled') {
1194
+ // includeThoughts: without this flag Gemini won't return thought parts -> the client loses thinking.
1195
+ gc.thinkingConfig = { includeThoughts: true };
1196
+ if (ir.thinking.type === 'enabled') {
1197
+ gc.thinkingConfig.thinkingBudget = clampBudget(ir.thinking.budget ?? (ir.thinking.effort ? effortToBudget(ir.thinking.effort) : 2048));
1198
+ }
1199
+ }
1200
+ if (Object.keys(gc).length) body.generationConfig = gc;
1201
+ return body;
1202
+ }
1203
+
1204
+ // JSON Schema -> Gemini/Vertex Schema (OpenAPI subset). Gemini rejects keys like
1205
+ // additionalProperties, $schema, $ref, const, exclusiveMinimum, array-form type...
1206
+ const GEMINI_SCHEMA_KEYS = new Set([
1207
+ 'type', 'format', 'title', 'description', 'nullable', 'enum', 'items', 'properties', 'required',
1208
+ 'minItems', 'maxItems', 'minProperties', 'maxProperties', 'minLength', 'maxLength', 'pattern',
1209
+ 'minimum', 'maximum', 'anyOf', 'propertyOrdering', 'default', 'example'
1210
+ ]);
1211
+
1212
+ // Resolve a local JSON-Schema $ref ('#/$defs/X', '#/definitions/X', '#/properties/...')
1213
+ // against the root parameters object. Returns the target node or null.
1214
+ function resolveLocalRef(root, ref) {
1215
+ if (typeof ref !== 'string' || !ref.startsWith('#/')) return null;
1216
+ const parts = ref.slice(2).split('/').map(p => p.replace(/~1/g, '/').replace(/~0/g, '~'));
1217
+ let node = root;
1218
+ for (const p of parts) {
1219
+ if (!node || typeof node !== 'object') return null;
1220
+ node = node[p];
1221
+ }
1222
+ return node && typeof node === 'object' && !Array.isArray(node) ? node : null;
1223
+ }
1224
+
1225
+ // A union of exactly one real schema plus an optional null branch collapses to that
1226
+ // schema (+nullable). Anything else (multi-branch anyOf) is left for the caller.
1227
+ function collapseNullableUnion(branches) {
1228
+ if (!Array.isArray(branches)) return null;
1229
+ const isNullBranch = (s) => {
1230
+ if (!s || typeof s !== 'object' || Array.isArray(s)) return false;
1231
+ if (s.type === 'null') return true;
1232
+ if (Array.isArray(s.type)) return s.type.length === 1 && s.type[0] === 'null';
1233
+ if (Array.isArray(s.enum) && s.enum.length === 1 && s.enum[0] === null) return true;
1234
+ if (Object.hasOwn(s, 'const') && s.const === null) return true;
1235
+ return false;
1236
+ };
1237
+ const rest = branches.filter(s => !isNullBranch(s));
1238
+ if (rest.length !== 1) return null;
1239
+ return { schema: rest[0], nullable: rest.length !== branches.length };
1240
+ }
1241
+
1242
+ function toGeminiSchema(schema, root, seen) {
1243
+ // Shorthand left by some MCP servers: a bare "object"/"string" where a Schema belongs.
1244
+ // Vertex rejects the raw string with INVALID_ARGUMENT, so expand it.
1245
+ if (typeof schema === 'string') {
1246
+ return schema === 'object' ? { type: 'object', properties: {} } : { type: schema };
1247
+ }
1248
+ if (!schema || typeof schema !== 'object' || Array.isArray(schema)) return schema;
1249
+ root = root || schema;
1250
+ seen = seen || new Set();
1251
+ if (seen.has(schema)) return {};
1252
+ // Inline local $refs: Vertex/GenAI function declarations do not support $ref/$defs.
1253
+ if (typeof schema.$ref === 'string') {
1254
+ const target = resolveLocalRef(root, schema.$ref);
1255
+ if (!target) return {};
1256
+ seen.add(schema);
1257
+ const merged = toGeminiSchema(target, root, seen);
1258
+ seen.delete(schema);
1259
+ const extra = {};
1260
+ for (const [k, v] of Object.entries(schema)) {
1261
+ if (k !== '$ref' && (k === 'description' || k === 'title' || k === 'default' || k === 'example')) extra[k] = v;
1262
+ }
1263
+ return { ...merged, ...extra };
1264
+ }
1265
+ const out = {};
1266
+ for (const [k, v] of Object.entries(schema)) {
1267
+ if (k === '$ref' || k === '$defs' || k === 'definitions') continue;
1268
+ if (k === 'const') { out.enum = [v]; continue; }
1269
+ if ((k === 'oneOf' || k === 'anyOf') && Array.isArray(v)) {
1270
+ const collapsed = collapseNullableUnion(v);
1271
+ if (collapsed) {
1272
+ const inner = toGeminiSchema(collapsed.schema, root, seen);
1273
+ if (inner && typeof inner === 'object' && !Array.isArray(inner)) Object.assign(out, inner);
1274
+ if (collapsed.nullable) out.nullable = true;
1275
+ continue;
1276
+ }
1277
+ if (k === 'oneOf') { out.anyOf = v.map(s => toGeminiSchema(s, root, seen)); continue; }
1278
+ }
1279
+ if (!GEMINI_SCHEMA_KEYS.has(k)) continue;
1280
+ if (k === 'properties' && v && typeof v === 'object') {
1281
+ out.properties = Object.fromEntries(Object.entries(v).map(([name, s]) => [name, toGeminiSchema(s, root, seen)]));
1282
+ } else if (k === 'items') {
1283
+ out.items = Array.isArray(v) ? toGeminiSchema(v[0] || {}, root, seen) : toGeminiSchema(v, root, seen);
1284
+ } else if (k === 'anyOf' && Array.isArray(v)) {
1285
+ out.anyOf = v.map(s => toGeminiSchema(s, root, seen));
1286
+ } else if (k === 'type' && Array.isArray(v)) {
1287
+ const types = v.filter(t => t !== 'null');
1288
+ out.type = types[0] || 'string';
1289
+ if (types.length !== v.length) out.nullable = true;
1290
+ } else if (k === 'format' && !['enum', 'date-time', 'int32', 'int64', 'float', 'double'].includes(v)) {
1291
+ continue;
1292
+ } else if (k === 'enum' && Array.isArray(v)) {
1293
+ out.enum = v.filter(x => x !== null).map(String);
1294
+ } else {
1295
+ out[k] = v;
1296
+ }
1297
+ }
1298
+ if (!out.type && out.properties) out.type = 'object';
1299
+ if (out.type === 'object' && !out.properties) out.properties = {};
1300
+ if (out.enum && !out.type) out.type = 'string';
1301
+ if (out.enum && out.type !== 'string') {
1302
+ // Gemini supports enum for strings only. Keep the constraint as text for the model.
1303
+ const values = (schema.enum || (Object.hasOwn(schema, 'const') ? [schema.const] : [])).filter(x => x !== null);
1304
+ if (values.length) out.description = `${out.description ? `${out.description} ` : ''}Allowed values: ${values.join(', ')}.`;
1305
+ delete out.enum;
1306
+ }
1307
+ if (Array.isArray(out.required)) {
1308
+ out.required = out.type === 'object' && out.properties ? out.required.filter(r => Object.hasOwn(out.properties, r)) : [];
1309
+ if (!out.required.length) delete out.required;
1310
+ }
1311
+ return out;
1312
+ }
1313
+
1314
+ // Claude Code injects an `x-anthropic-billing-header: ...` line at the top of system. Google Antigravity
1315
+ // answers that line with a bogus 429 RESOURCE_EXHAUSTED even when quota remains, so strip it only when the target model is antigravity
1316
+ // (`ag/...`). 9Router's Claude provider still needs the header, so keep it for all other models.
1317
+ const BILLING_HEADER_RE = /^x-anthropic-billing-header:[^\n]*(?:\r?\n)*/i;
1318
+
1319
+ function isAntigravityModel(model) {
1320
+ return /^(ag|antigravity)\//i.test(String(model || ''));
1321
+ }
1322
+
1323
+ function emitUpstreamBody(outFormat, ir, model, opts = {}) {
1324
+ // thinkingMode 'off' applies to all upstreams: drop the client's thinking entirely.
1325
+ let src = normThinkingMode(opts.thinkingMode) === 'off' ? { ...ir, thinking: { type: 'disabled' } } : ir;
1326
+ if (isAntigravityModel(model) && src.system) src = { ...src, system: src.system.replace(BILLING_HEADER_RE, '') };
1327
+ switch (outFormat) {
1328
+ case 'anthropic': return irToAnthropicBody(src, model);
1329
+ case 'vertex': return irToVertexBody(src, model);
1330
+ case 'openai-chat':
1331
+ default: return irToChatBody(src, model, opts);
1332
+ }
1333
+ }
1334
+
1335
+ // ---------------- native Anthropic healer (direct passthrough) ----------------
1336
+ // The anthropic -> anthropic branch skips IR (to preserve thinking signatures, cache_control, documents...),
1337
+ // so it patches the Anthropic payload directly, touching only what the API will definitely reject:
1338
+ // 1. orphaned tool_result -> text block;
1339
+ // 2. tool_use missing tool_result in the next user turn -> add a placeholder tool_result;
1340
+ // 3. tool_result must lead the user turn -> move it to the front;
1341
+ // 4. thinking block with a fake signature (generated by the gateway during conversion) -> drop, since Anthropic verifies signatures;
1342
+ // 5. thinking on but the last assistant turn of the tool loop doesn't start with thinking -> disable thinking
1343
+ // for this request (Anthropic: "a final assistant message must start with a thinking block").
1344
+
1345
+ export const PLACEHOLDER_SIGNATURE = 'reasoning-sig';
1346
+
1347
+ // Signatures the gateway issues to Anthropic clients are always "not from Anthropic": a placeholder, or
1348
+ // another provider's signature wrapped with the `lsw1.` prefix (so it can be sent back to that provider later).
1349
+ export const FOREIGN_SIG_PREFIX = 'lsw1.';
1350
+
1351
+ function isGatewaySignature(sig) {
1352
+ return !sig || sig === PLACEHOLDER_SIGNATURE || String(sig).startsWith(FOREIGN_SIG_PREFIX);
1353
+ }
1354
+
1355
+ function toAnthropicBlocks(content) {
1356
+ if (typeof content === 'string') return content ? [{ type: 'text', text: content }] : [];
1357
+ return Array.isArray(content) ? content.filter(Boolean) : [];
1358
+ }
1359
+
1360
+ function toolResultText(block) {
1361
+ const c = block.content;
1362
+ const text = typeof c === 'string' ? c
1363
+ : Array.isArray(c) ? c.map(x => (x?.type === 'text' ? x.text : x?.type === 'image' ? '[image omitted]' : JSON.stringify(x))).join('\n')
1364
+ : c ? JSON.stringify(c) : '';
1365
+ return `[Tool Result${block.tool_use_id ? ` (${block.tool_use_id})` : ''}]: ${text}`;
1366
+ }
1367
+
1368
+ function healAnthropicPayload(payload) {
1369
+ const notes = [];
1370
+ if (!payload || !Array.isArray(payload.messages)) return { payload, changed: false, notes };
1371
+ const src = payload.messages.filter(m => m && (m.role === 'user' || m.role === 'assistant'));
1372
+ let changed = src.length !== payload.messages.length;
1373
+
1374
+ // Step 0: drop thinking blocks with fake signatures; merge adjacent user turns (so a tool_result split
1375
+ // into a later turn can still be matched with its tool_use).
1376
+ const msgs = [];
1377
+ for (const m of src) {
1378
+ let content = m.content;
1379
+ if (m.role === 'assistant' && Array.isArray(content)) {
1380
+ const kept = content.filter(b => !(b && b.type === 'thinking' && isGatewaySignature(b.signature)));
1381
+ if (kept.length !== content.length) {
1382
+ changed = true;
1383
+ notes.push('stripped placeholder-signed thinking blocks');
1384
+ content = kept;
1385
+ }
1386
+ if (!content.length) {
1387
+ changed = true;
1388
+ continue;
1389
+ }
1390
+ }
1391
+ const last = msgs[msgs.length - 1];
1392
+ if (m.role === 'user' && last && last.role === 'user') {
1393
+ last.content = [...toAnthropicBlocks(last.content), ...toAnthropicBlocks(content)];
1394
+ changed = true;
1395
+ continue;
1396
+ }
1397
+ msgs.push({ ...m, content });
1398
+ }
1399
+
1400
+ // Steps 1-3: pair up tool_use / tool_result.
1401
+ const out = [];
1402
+ for (let i = 0; i < msgs.length; i++) {
1403
+ const m = msgs[i];
1404
+ if (m.role === 'assistant') {
1405
+ out.push(m);
1406
+ const toolIds = toAnthropicBlocks(m.content).filter(b => b.type === 'tool_use' && b.id).map(b => b.id);
1407
+ if (toolIds.length && msgs[i + 1]?.role !== 'user') {
1408
+ out.push({ role: 'user', content: toolIds.map(id => ({ type: 'tool_result', tool_use_id: id, content: MISSING_TOOL_RESULT })) });
1409
+ changed = true;
1410
+ notes.push('added placeholder tool_result');
1411
+ }
1412
+ continue;
1413
+ }
1414
+ const prev = out[out.length - 1];
1415
+ const pending = prev?.role === 'assistant'
1416
+ ? toAnthropicBlocks(prev.content).filter(b => b.type === 'tool_use' && b.id).map(b => b.id)
1417
+ : [];
1418
+ const blocks = toAnthropicBlocks(m.content);
1419
+ const results = new Map();
1420
+ const rest = [];
1421
+ for (const b of blocks) {
1422
+ if (b.type === 'tool_result' && pending.includes(b.tool_use_id) && !results.has(b.tool_use_id)) {
1423
+ results.set(b.tool_use_id, b);
1424
+ } else if (b.type === 'tool_result') {
1425
+ rest.push({ type: 'text', text: toolResultText(b) });
1426
+ changed = true;
1427
+ notes.push('converted orphaned tool_result to text');
1428
+ } else {
1429
+ rest.push(b);
1430
+ }
1431
+ }
1432
+ if (!blocks.some(b => b.type === 'tool_result') && !pending.length) {
1433
+ out.push(m);
1434
+ continue;
1435
+ }
1436
+ const head = pending.map(id => {
1437
+ if (results.has(id)) return results.get(id);
1438
+ changed = true;
1439
+ notes.push('added placeholder tool_result');
1440
+ return { type: 'tool_result', tool_use_id: id, content: MISSING_TOOL_RESULT };
1441
+ });
1442
+ const newContent = [...head, ...rest];
1443
+ // Blocks are reused by reference, so an unchanged turn has the same objects in the same order.
1444
+ if (newContent.length === blocks.length && newContent.every((b, i) => b === blocks[i])) {
1445
+ out.push(m);
1446
+ continue;
1447
+ }
1448
+ changed = true;
1449
+ notes.push('normalized tool_result placement');
1450
+ out.push({ ...m, content: newContent.length ? newContent : [{ type: 'text', text: '(empty)' }] });
1451
+ }
1452
+
1453
+ let body = { ...payload, messages: out };
1454
+
1455
+ // Step 5: thinking + an in-progress tool loop whose last assistant turn doesn't open with thinking.
1456
+ const thinkingOn = payload.thinking && payload.thinking.type && payload.thinking.type !== 'disabled';
1457
+ if (thinkingOn) {
1458
+ const lastUser = out[out.length - 1];
1459
+ const lastAssistant = out[out.length - 2];
1460
+ const inToolLoop = lastUser?.role === 'user' && toAnthropicBlocks(lastUser.content).some(b => b.type === 'tool_result');
1461
+ const first = lastAssistant?.role === 'assistant' ? toAnthropicBlocks(lastAssistant.content)[0] : null;
1462
+ if (inToolLoop && first && first.type !== 'thinking' && first.type !== 'redacted_thinking') {
1463
+ body = { ...body };
1464
+ delete body.thinking;
1465
+ changed = true;
1466
+ notes.push('disabled thinking: last assistant turn has no valid thinking block');
1467
+ }
1468
+ }
1469
+
1470
+ if (!changed) return { payload, changed: false, notes };
1471
+ return { payload: body, changed: true, notes: [...new Set(notes)] };
1472
+ }
1473
+
1474
+ // Estimate tokens for /count_tokens when the upstream has no real counting endpoint.
1475
+ function estimateTokens(payload) {
1476
+ let chars = 0;
1477
+ let images = 0;
1478
+ const walk = (v) => {
1479
+ if (typeof v === 'string') { chars += v.length; return; }
1480
+ if (Array.isArray(v)) { v.forEach(walk); return; }
1481
+ if (v && typeof v === 'object') {
1482
+ if (v.type === 'image' || v.type === 'image_url' || v.type === 'document') { images++; return; }
1483
+ for (const [k, x] of Object.entries(v)) {
1484
+ if (k === 'data' || k === 'cache_control') continue;
1485
+ walk(x);
1486
+ }
1487
+ }
1488
+ };
1489
+ walk(payload?.system);
1490
+ walk(payload?.messages);
1491
+ walk(payload?.tools);
1492
+ return Math.ceil(chars / 4) + images * 1600;
1493
+ }
1494
+
1495
+ // ---------------- upstream event normalization ----------------
1496
+ // Every upstream response (any format, stream or JSON) -> standard events:
1497
+ // { think:[{text,sig}], text:[str], tools:[{index,id,name,args}],
1498
+ // finish: 'stop'|'length'|'tool_calls'|'content_filter'|null,
1499
+ // usage:{prompt,completion,cached}, sig, error }
1500
+ //
1501
+ // Usage convention: prompt = total input tokens (including cached, per OpenAI semantics), cached = cache read.
1502
+
1503
+ function chunkToolFull(tc, idx) {
1504
+ return { index: (typeof tc.index === 'number' ? tc.index : idx), id: tc.id || null, name: tc.name || null, args: tc.args || '', sig: tc.sig || null };
1505
+ }
1506
+
1507
+ // ---------------- Gemini thought signature cache ----------------
1508
+ // Gemini 3 requires resending the functionCall's thoughtSignature within the current turn (missing -> HTTP 400).
1509
+ // Clients like Claude Code / Codex don't carry this signature, so the gateway remembers it by tool call id (in-RAM LRU)
1510
+ // and reattaches it when history comes back. Cache lost (restart) -> use the Google-allowed dummy signature.
1511
+ // https://ai.google.dev/gemini-api/docs/generate-content/thought-signatures
1512
+ export const GEMINI_DUMMY_SIGNATURE = 'skip_thought_signature_validator';
1513
+ const MAX_SIGNATURES = 5000;
1514
+ const toolSignatures = new Map();
1515
+
1516
+ function rememberToolSignature(id, sig) {
1517
+ if (!id || !sig) return;
1518
+ toolSignatures.delete(id);
1519
+ toolSignatures.set(id, sig);
1520
+ if (toolSignatures.size > MAX_SIGNATURES) toolSignatures.delete(toolSignatures.keys().next().value);
1521
+ }
1522
+
1523
+ function lookupToolSignature(id) {
1524
+ return id ? toolSignatures.get(id) || null : null;
1525
+ }
1526
+
1527
+ function geminiRequiresSignatures(model) {
1528
+ const m = String(model || '').toLowerCase().match(/gemini-(\d+)/);
1529
+ return Boolean(m && Number(m[1]) >= 3);
1530
+ }
1531
+
1532
+ function anthropicUsage(u) {
1533
+ const o = (u && typeof u === 'object') ? u : {};
1534
+ const input = Number(o.input_tokens) || 0;
1535
+ const cacheRead = Number(o.cache_read_input_tokens) || 0;
1536
+ const cacheCreate = Number(o.cache_creation_input_tokens) || 0;
1537
+ // Anthropic reports thinking tokens under output_tokens_details.
1538
+ const det = (o.output_tokens_details && typeof o.output_tokens_details === 'object') ? o.output_tokens_details : {};
1539
+ return {
1540
+ prompt: input + cacheRead + cacheCreate,
1541
+ completion: Number(o.output_tokens) || 0,
1542
+ cached: cacheRead,
1543
+ reasoning: Number(det.thinking_tokens || det.reasoning_tokens) || 0
1544
+ };
1545
+ }
1546
+
1547
+ function upstreamError(parsed) {
1548
+ if (parsed.type === 'error') return parsed.error?.message || JSON.stringify(parsed.error || parsed);
1549
+ if (parsed.error && typeof parsed.error === 'object' && !parsed.choices && !parsed.candidates) {
1550
+ return parsed.error.message || JSON.stringify(parsed.error);
1551
+ }
1552
+ return null;
1553
+ }
1554
+
1555
+ function normalizeUpstream(parsed, outFormat) {
1556
+ const ev = { think: [], text: [], tools: [], finish: null, usage: { prompt: 0, completion: 0, cached: 0, reasoning: 0 }, sig: null, error: null };
1557
+ if (!parsed || typeof parsed !== 'object') return ev;
1558
+
1559
+ const err = upstreamError(parsed);
1560
+ if (err) {
1561
+ ev.error = err;
1562
+ return ev;
1563
+ }
1564
+
1565
+ if (outFormat === 'anthropic') {
1566
+ const t = parsed.type;
1567
+ if (t === 'content_block_delta') {
1568
+ const d = parsed.delta || {};
1569
+ if (d.type === 'thinking_delta' && d.thinking) ev.think.push({ text: d.thinking });
1570
+ else if (d.type === 'text_delta' && d.text) ev.text.push(d.text);
1571
+ else if (d.type === 'signature_delta' && d.signature) ev.sig = d.signature;
1572
+ else if (d.type === 'input_json_delta' && d.partial_json) {
1573
+ ev.tools.push({ index: parsed.index ?? 0, id: null, name: null, args: d.partial_json });
1574
+ }
1575
+ } else if (t === 'content_block_start') {
1576
+ const b = parsed.content_block || {};
1577
+ if (b.type === 'tool_use') ev.tools.push({ index: parsed.index ?? 0, id: b.id || null, name: b.name || null, args: '' });
1578
+ else if (b.type === 'thinking' && b.thinking) ev.think.push({ text: b.thinking });
1579
+ else if (b.type === 'text' && b.text) ev.text.push(b.text);
1580
+ } else if (t === 'message_start') {
1581
+ ev.usage = anthropicUsage(parsed.message?.usage);
1582
+ } else if (t === 'message_delta') {
1583
+ if (parsed.delta?.stop_reason) ev.finish = canonFinish(parsed.delta.stop_reason);
1584
+ ev.usage = anthropicUsage(parsed.usage);
1585
+ } else if (t === 'message' && Array.isArray(parsed.content)) {
1586
+ // non-stream Anthropic message
1587
+ for (const b of parsed.content) {
1588
+ if (b.type === 'thinking' && b.thinking) ev.think.push({ text: b.thinking, sig: b.signature || null });
1589
+ else if (b.type === 'text' && b.text) ev.text.push(b.text);
1590
+ else if (b.type === 'tool_use') ev.tools.push(chunkToolFull({ index: ev.tools.length, id: b.id, name: b.name, args: stringifyArgs(b.input) }));
1591
+ }
1592
+ if (parsed.stop_reason) ev.finish = canonFinish(parsed.stop_reason);
1593
+ ev.usage = anthropicUsage(parsed.usage);
1594
+ }
1595
+ return ev;
1596
+ }
1597
+
1598
+ // openai-chat | vertex (both SSE chunks and full JSON share one shape)
1599
+ // Read usage FIRST: OpenAI's final usage chunk (stream_options.include_usage) has `choices: []`.
1600
+ const u = smartUsage(parsed.usage ?? parsed.usageMetadata);
1601
+ // Forward the whole shape. Listing the fields by hand here is how `reasoning`
1602
+ // was silently dropped between smartUsage and the emitters on 2026-09-20.
1603
+ ev.usage = u;
1604
+
1605
+ const choice = firstChoice(parsed);
1606
+ const node = smartDelta(choice) || (outFormat === 'vertex' ? parsed : null);
1607
+ if (!node) return ev;
1608
+ const r = smartReasoning(node);
1609
+ if (r) {
1610
+ ev.think.push({ text: r.text, sig: r.signature || null });
1611
+ if (r.signature) ev.sig = r.signature;
1612
+ }
1613
+ const tx = smartText(node);
1614
+ if (tx) ev.text.push(tx);
1615
+ for (const tc of smartToolCalls(node)) ev.tools.push(chunkToolFull(tc));
1616
+ const f = smartFinish(parsed, choice);
1617
+ if (f) ev.finish = canonFinish(f);
1618
+ return ev;
1619
+ }
1620
+
1621
+ // Stateful normalizer for one response: renumber tool call indexes to continuous 0,1,2...
1622
+ // - Vertex returns complete functionCalls with no index -> each call is a new tool
1623
+ // (previously every call had index 0 so args from different calls got concatenated).
1624
+ // - Anthropic uses block indexes (1, 2...) -> OpenAI clients need tool_calls indexes starting at 0.
1625
+ // - OpenAI-compatible providers drop `index` but send distinct ids -> split into separate tools.
1626
+ function createUpstreamNormalizer(outFormat) {
1627
+ const slots = new Map();
1628
+ let next = 0;
1629
+ return (parsed) => {
1630
+ const ev = normalizeUpstream(parsed, outFormat);
1631
+ for (const tc of ev.tools) {
1632
+ if (outFormat === 'vertex') {
1633
+ tc.index = next++;
1634
+ // Vertex has no id -> generate a stable id for the client to send back, used as the signature-cache key.
1635
+ tc.id = tc.id || `call_${rand(24)}`;
1636
+ rememberToolSignature(tc.id, tc.sig);
1637
+ continue;
1638
+ }
1639
+ const key = tc.index ?? 0;
1640
+ const slot = slots.get(key);
1641
+ if (!slot || (tc.id && slot.id && tc.id !== slot.id)) {
1642
+ const created = { index: next++, id: tc.id || null };
1643
+ slots.set(key, created);
1644
+ tc.index = created.index;
1645
+ } else {
1646
+ if (!slot.id && tc.id) slot.id = tc.id;
1647
+ tc.index = slot.index;
1648
+ }
1649
+ rememberToolSignature(tc.id || slots.get(key)?.id, tc.sig);
1650
+ }
1651
+ return ev;
1652
+ };
1653
+ }
1654
+
1655
+ // Collect events for non-stream (also the usage accumulator for stream).
1656
+ function createCollector() {
1657
+ const C = {
1658
+ think: [], text: [], tools: new Map(), finish: null,
1659
+ prompt: 0, completionTokens: 0, cached: 0, reasoning: 0, usageSum: 0,
1660
+ sig: null, chars: 0, error: null,
1661
+ add(ev) {
1662
+ for (const t of (ev.think || [])) {
1663
+ C.think.push(t.text);
1664
+ C.chars += t.text.length;
1665
+ if (t.sig && !C.sig) C.sig = t.sig;
1666
+ }
1667
+ if (ev.sig && !C.sig) C.sig = ev.sig;
1668
+ for (const t of (ev.text || [])) {
1669
+ C.text.push(t);
1670
+ C.chars += t.length;
1671
+ }
1672
+ for (const tc of (ev.tools || [])) {
1673
+ const idx = tc.index ?? 0;
1674
+ if (!C.tools.has(idx)) C.tools.set(idx, { index: idx, id: tc.id, name: tc.name, args: '' });
1675
+ const s = C.tools.get(idx);
1676
+ if (!s.id && tc.id) s.id = tc.id;
1677
+ if (!s.sig && tc.sig) s.sig = tc.sig;
1678
+ if ((!s.name || s.name === 'tool') && tc.name) s.name = tc.name;
1679
+ if (tc.args) {
1680
+ s.args += tc.args;
1681
+ C.chars += tc.args.length;
1682
+ }
1683
+ }
1684
+ if (ev.finish) C.finish = ev.finish;
1685
+ if (ev.error) C.error = ev.error;
1686
+ if (ev.usage) {
1687
+ // Three upstream shapes, not two:
1688
+ // cumulative - Anthropic/Vertex repeat a running total in every chunk
1689
+ // final-once - OpenAI sends one usage object at the end
1690
+ // per-chunk - qwen/zai on Cloudflare send a DELTA every chunk
1691
+ // (completion_tokens: 1 x N, see docs/LLM-RESPONSE-MATRIX.md)
1692
+ // Math.max is right for the first two and reports 1 for the third. A value
1693
+ // that does not grow is the signature of a delta, so switch to summing the
1694
+ // moment a later chunk reports less completion than the running total.
1695
+ const completion = ev.usage.completion || 0;
1696
+ C.completionTokens = Math.max(C.completionTokens, completion);
1697
+ C.usageSum += completion;
1698
+ C.prompt = Math.max(C.prompt, ev.usage.prompt || 0);
1699
+ C.cached = Math.max(C.cached, ev.usage.cached || 0);
1700
+ C.reasoning = Math.max(C.reasoning, ev.usage.reasoning || 0);
1701
+ }
1702
+ },
1703
+ // Upstream returned no usage -> estimate ~4 chars / token.
1704
+ //
1705
+ // Three upstream shapes exist, not two. Anthropic and Vertex repeat a running
1706
+ // total in every chunk and OpenAI sends one object at the end: for both, the
1707
+ // maximum is the answer. But qwen and zai on Cloudflare send a per-token DELTA
1708
+ // in every chunk (`completion_tokens: 1` x N, docs/LLM-RESPONSE-MATRIX.md:126),
1709
+ // and the maximum of those is 1 no matter how long the reply was.
1710
+ //
1711
+ // Rather than guess the shape from the number pattern, compare the reported
1712
+ // total against what was actually streamed. A total far below the text we
1713
+ // received cannot be a total, so the sum is the honest figure.
1714
+ completion() {
1715
+ const estimate = Math.ceil(C.chars / 4);
1716
+ if (C.completionTokens <= 0) return estimate;
1717
+ if (C.usageSum > C.completionTokens && C.completionTokens * 4 < estimate) return C.usageSum;
1718
+ return C.completionTokens;
1719
+ }
1720
+ };
1721
+ return C;
1722
+ }
1723
+
1724
+ // ---------------- <think> tag splitter ----------------
1725
+ // Many OSS models (DeepSeek, Qwen, GLM) stuff reasoning into content as <think>...</think>.
1726
+ // The splitter tolerates tags split across 2 chunks (e.g. "<thi" + "nk>").
1727
+
1728
+ const THINK_TAGS = ['<think>', '</think>', '<thinking>', '</thinking>'];
1729
+
1730
+ function createThinkTagSplitter(onThink, onText) {
1731
+ let carry = '';
1732
+ let inTag = false;
1733
+ const emitTok = (tok) => {
1734
+ if (!tok) return;
1735
+ if (inTag) onThink(tok);
1736
+ else onText(tok);
1737
+ };
1738
+ return {
1739
+ push(raw) {
1740
+ if (!raw) return;
1741
+ let buf = carry + raw;
1742
+ carry = '';
1743
+ const lastOpen = buf.lastIndexOf('<');
1744
+ if (lastOpen !== -1) {
1745
+ const tail = buf.slice(lastOpen).toLowerCase();
1746
+ if (!tail.includes('>') && THINK_TAGS.some(t => t.startsWith(tail))) {
1747
+ carry = buf.slice(lastOpen);
1748
+ buf = buf.slice(0, lastOpen);
1749
+ }
1750
+ }
1751
+ for (const tok of buf.split(/(<\/?think(?:ing)?>)/i)) {
1752
+ if (!tok) continue;
1753
+ if (/^<think(?:ing)?>$/i.test(tok)) { inTag = true; continue; }
1754
+ if (/^<\/think(?:ing)?>$/i.test(tok)) { inTag = false; continue; }
1755
+ emitTok(tok);
1756
+ }
1757
+ },
1758
+ flush() {
1759
+ const c = carry;
1760
+ carry = '';
1761
+ emitTok(c);
1762
+ }
1763
+ };
1764
+ }
1765
+
1766
+ function splitThinkTags(text) {
1767
+ const think = [];
1768
+ const out = [];
1769
+ const sp = createThinkTagSplitter(t => think.push(t), t => out.push(t));
1770
+ sp.push(text);
1771
+ sp.flush();
1772
+ return { think: think.join(''), text: out.join('') };
1773
+ }
1774
+
1775
+ // ---------------- client renderers ----------------
1776
+ // Each renderer has: start(), think(text, sig), text(t), tool({index,id,name,args}),
1777
+ // finish(canonical, {prompt, completion, cached, hasTools}), error(message).
1778
+
1779
+ function rand(n = 6) {
1780
+ let s = '';
1781
+ while (s.length < n) s += Math.random().toString(36).slice(2);
1782
+ return s.slice(0, n);
1783
+ }
1784
+
1785
+ function anthropicStopReason(canonical, hasTools) {
1786
+ if (canonical === 'length') return 'max_tokens';
1787
+ if (hasTools) return 'tool_use';
1788
+ switch (canonical) {
1789
+ case 'length': return 'max_tokens';
1790
+ case 'tool_calls': return 'tool_use';
1791
+ default: return 'end_turn';
1792
+ }
1793
+ }
1794
+
1795
+ // --- Anthropic SSE + message ---
1796
+ // Sequential block state machine: indexes grow in content_block_start order, never
1797
+ // write deltas into a stopped block or reuse an index (thinking -> tool -> thinking -> text are all valid).
1798
+ function createAnthropicStream(emit, model) {
1799
+ const msgId = `msg_${Date.now()}_${rand()}`;
1800
+ let nextIndex = 0;
1801
+ let open = null; // { kind: 'thinking' | 'text', index }
1802
+ const tools = new Map(); // tool index -> { index, id, name, closed }
1803
+ let sig = null;
1804
+
1805
+ function closeOpen() {
1806
+ if (!open) return;
1807
+ if (open.kind === 'thinking') {
1808
+ emit('content_block_delta', { type: 'content_block_delta', index: open.index, delta: { type: 'signature_delta', signature: sig ? FOREIGN_SIG_PREFIX + sig : PLACEHOLDER_SIGNATURE } });
1809
+ }
1810
+ emit('content_block_stop', { type: 'content_block_stop', index: open.index });
1811
+ open = null;
1812
+ }
1813
+ function closeTools() {
1814
+ for (const t of tools.values()) {
1815
+ if (!t.closed) {
1816
+ emit('content_block_stop', { type: 'content_block_stop', index: t.index });
1817
+ t.closed = true;
1818
+ }
1819
+ }
1820
+ }
1821
+ function delta(kind, tok) {
1822
+ if (!open || open.kind !== kind) {
1823
+ closeOpen();
1824
+ closeTools();
1825
+ open = { kind, index: nextIndex++ };
1826
+ emit('content_block_start', {
1827
+ type: 'content_block_start', index: open.index,
1828
+ content_block: kind === 'thinking' ? { type: 'thinking', thinking: '' } : { type: 'text', text: '' }
1829
+ });
1830
+ }
1831
+ emit('content_block_delta', {
1832
+ type: 'content_block_delta', index: open.index,
1833
+ delta: kind === 'thinking' ? { type: 'thinking_delta', thinking: tok } : { type: 'text_delta', text: tok }
1834
+ });
1835
+ }
1836
+
1837
+ return {
1838
+ start() {
1839
+ emit('message_start', {
1840
+ type: 'message_start',
1841
+ message: {
1842
+ id: msgId, type: 'message', role: 'assistant', model, content: [],
1843
+ stop_reason: null, stop_sequence: null,
1844
+ usage: { input_tokens: 0, output_tokens: 0, cache_creation_input_tokens: 0, cache_read_input_tokens: 0 }
1845
+ }
1846
+ });
1847
+ emit('ping', { type: 'ping' });
1848
+ },
1849
+ think(text, s) {
1850
+ if (s) sig = s;
1851
+ if (text) delta('thinking', text);
1852
+ },
1853
+ text(t) {
1854
+ if (t) delta('text', t);
1855
+ },
1856
+ tool(tc) {
1857
+ const key = tc.index ?? 0;
1858
+ let st = tools.get(key);
1859
+ if (!st) {
1860
+ closeOpen();
1861
+ st = { index: nextIndex++, id: tc.id || `toolu_${rand(24)}`, name: tc.name || 'tool', closed: false };
1862
+ tools.set(key, st);
1863
+ emit('content_block_start', {
1864
+ type: 'content_block_start', index: st.index,
1865
+ content_block: { type: 'tool_use', id: st.id, name: st.name, input: {} }
1866
+ });
1867
+ }
1868
+ if (tc.args) {
1869
+ emit('content_block_delta', {
1870
+ type: 'content_block_delta', index: st.index,
1871
+ delta: { type: 'input_json_delta', partial_json: tc.args }
1872
+ });
1873
+ }
1874
+ },
1875
+ finish(canonical, stats = {}) {
1876
+ closeOpen();
1877
+ closeTools();
1878
+ const prompt = stats.prompt || 0;
1879
+ const cached = Math.min(stats.cached || 0, prompt);
1880
+ emit('message_delta', {
1881
+ type: 'message_delta',
1882
+ delta: { stop_reason: anthropicStopReason(canonical, stats.hasTools || tools.size > 0), stop_sequence: null },
1883
+ usage: {
1884
+ input_tokens: prompt - cached,
1885
+ output_tokens: stats.completion || 0,
1886
+ cache_read_input_tokens: cached,
1887
+ cache_creation_input_tokens: 0
1888
+ }
1889
+ });
1890
+ emit('message_stop', { type: 'message_stop' });
1891
+ },
1892
+ error(message) {
1893
+ closeOpen();
1894
+ closeTools();
1895
+ emit('error', { type: 'error', error: { type: 'api_error', message: String(message || 'Upstream stream error') } });
1896
+ }
1897
+ };
1898
+ }
1899
+
1900
+ function buildAnthropicMessage({ model, think, text, tools, finish, prompt, completion, cached, id, sig }) {
1901
+ const content = [];
1902
+ const thinking = (think || []).join('');
1903
+ if (thinking) {
1904
+ content.push({ type: 'thinking', thinking, signature: sig ? FOREIGN_SIG_PREFIX + sig : PLACEHOLDER_SIGNATURE });
1905
+ }
1906
+ const body = (text || []).join('');
1907
+ if (body) content.push({ type: 'text', text: body });
1908
+ for (const tc of (tools || [])) {
1909
+ content.push({ type: 'tool_use', id: tc.id || `toolu_${rand(24)}`, name: tc.name || 'tool', input: parseArgs(tc.args) });
1910
+ }
1911
+ if (!content.length) content.push({ type: 'text', text: '' });
1912
+ const p = prompt || 0;
1913
+ const c = Math.min(cached || 0, p);
1914
+ return {
1915
+ id: id || `msg_${Date.now()}_${rand()}`,
1916
+ type: 'message', role: 'assistant', model, content,
1917
+ stop_reason: anthropicStopReason(finish, Boolean(tools && tools.length)),
1918
+ stop_sequence: null,
1919
+ usage: { input_tokens: p - c, output_tokens: completion || 0, cache_creation_input_tokens: 0, cache_read_input_tokens: c }
1920
+ };
1921
+ }
1922
+
1923
+ // --- OpenAI Chat SSE + object ---
1924
+ function createChatStream(emit, model) {
1925
+ const id = `chatcmpl-${Date.now()}${rand(4)}`;
1926
+ const created = Math.floor(Date.now() / 1000);
1927
+ const seenTools = new Set();
1928
+ const chunk = (choices, usage) => {
1929
+ const o = { id, object: 'chat.completion.chunk', created, model, choices };
1930
+ if (usage) o.usage = usage;
1931
+ emit(null, o);
1932
+ };
1933
+ return {
1934
+ start() {
1935
+ chunk([{ index: 0, delta: { role: 'assistant', content: '' }, finish_reason: null }]);
1936
+ },
1937
+ think(t) {
1938
+ if (t) chunk([{ index: 0, delta: { reasoning_content: t }, finish_reason: null }]);
1939
+ },
1940
+ text(t) {
1941
+ if (t) chunk([{ index: 0, delta: { content: t }, finish_reason: null }]);
1942
+ },
1943
+ tool(tc) {
1944
+ const idx = tc.index ?? 0;
1945
+ if (!seenTools.has(idx)) {
1946
+ seenTools.add(idx);
1947
+ chunk([{ index: 0, delta: { tool_calls: [{ index: idx, id: tc.id || `call_${rand(24)}`, type: 'function', function: { name: tc.name || 'tool', arguments: tc.args || '' } }] }, finish_reason: null }]);
1948
+ } else if (tc.args) {
1949
+ chunk([{ index: 0, delta: { tool_calls: [{ index: idx, function: { arguments: tc.args } }] }, finish_reason: null }]);
1950
+ }
1951
+ },
1952
+ finish(canonical, stats = {}) {
1953
+ const completion = stats.completion || 0;
1954
+ const prompt = stats.prompt || 0;
1955
+ const usage = { prompt_tokens: prompt, completion_tokens: completion, total_tokens: prompt + completion };
1956
+ if (stats.cached) usage.prompt_tokens_details = { cached_tokens: stats.cached };
1957
+ if (stats.reasoning > 0) usage.completion_tokens_details = { reasoning_tokens: stats.reasoning };
1958
+ chunk([{ index: 0, delta: {}, finish_reason: chatFinish(canonical, stats.hasTools) }], usage);
1959
+ },
1960
+ error(message) {
1961
+ emit(null, { error: { message: String(message || 'Upstream stream error'), type: 'server_error', code: 'upstream_error' } });
1962
+ }
1963
+ };
1964
+ }
1965
+
1966
+ function buildChatMessage({ model, think, text, tools, finish, prompt, completion, cached, reasoning, id, stats }) {
1967
+ const msg = { role: 'assistant', content: (text || []).join('') || null };
1968
+ const thinking = (think || []).join('');
1969
+ if (thinking) {
1970
+ msg.reasoning_content = thinking;
1971
+ msg.reasoning = thinking;
1972
+ }
1973
+ if (tools && tools.length) {
1974
+ msg.tool_calls = tools.map(tc => ({
1975
+ id: tc.id || `call_${rand(24)}`,
1976
+ type: 'function',
1977
+ function: { name: tc.name || 'tool', arguments: typeof tc.args === 'string' ? tc.args : stringifyArgs(tc.args) }
1978
+ }));
1979
+ }
1980
+ if (msg.content === null && !msg.tool_calls) msg.content = '';
1981
+ const usage = { prompt_tokens: prompt || 0, completion_tokens: completion || 0, total_tokens: (prompt || 0) + (completion || 0) };
1982
+ if (cached) usage.prompt_tokens_details = { cached_tokens: cached };
1983
+ const r = reasoning ?? stats?.reasoning;
1984
+ if (r > 0) usage.completion_tokens_details = { reasoning_tokens: r };
1985
+ return {
1986
+ id: id || `chatcmpl-${Date.now()}${rand(4)}`,
1987
+ object: 'chat.completion',
1988
+ created: Math.floor(Date.now() / 1000),
1989
+ model,
1990
+ choices: [{ index: 0, message: msg, finish_reason: chatFinish(finish, tools && tools.length) }],
1991
+ usage
1992
+ };
1993
+ }
1994
+
1995
+ // --- OpenAI Responses (Codex) SSE + object ---
1996
+ // Codex CLI builds history & runs tools from `response.output_item.done`, not `response.completed.output`,
1997
+ // so each item (reasoning / message / function_call) must complete the full added -> delta -> done cycle.
1998
+ function responsesUsage(prompt, completion, cached, reasoning) {
1999
+ return {
2000
+ input_tokens: prompt || 0,
2001
+ input_tokens_details: { cached_tokens: cached || 0 },
2002
+ output_tokens: completion || 0,
2003
+ output_tokens_details: { reasoning_tokens: reasoning || 0 },
2004
+ total_tokens: (prompt || 0) + (completion || 0)
2005
+ };
2006
+ }
2007
+
2008
+ // Models often return args with literal newlines inside JSON strings (invalid) -> try escaping control chars.
2009
+ function parseArgsLenient(v) {
2010
+ const first = parseArgs(v);
2011
+ if (typeof v !== 'string' || !('raw' in first) || Object.keys(first).length !== 1) return first;
2012
+ try {
2013
+ return JSON.parse(v.replace(/[\u0000-\u001f]/g, c => JSON.stringify(c).slice(1, -1)));
2014
+ } catch {
2015
+ return first;
2016
+ }
2017
+ }
2018
+
2019
+ // Build the Codex output item for one tool call based on the original tool type declared in the request.
2020
+ function responsesToolItem(toolMeta, t, status = 'completed') {
2021
+ const meta = toolMeta?.[t.name];
2022
+ const ns = meta?.namespace ? { namespace: meta.namespace } : {};
2023
+ if (meta?.kind === 'custom') {
2024
+ const a = parseArgsLenient(t.args);
2025
+ const input = typeof a.input === 'string' ? a.input : (typeof a.raw === 'string' ? a.raw : (t.args || ''));
2026
+ return { id: t.itemId || `ctc_${rand(24)}`, type: 'custom_tool_call', status, call_id: t.callId, name: meta.name, ...ns, input: status === 'completed' ? input : '' };
2027
+ }
2028
+ if (meta?.kind === 'local_shell') {
2029
+ const a = parseArgsLenient(t.args);
2030
+ const command = Array.isArray(a.command) ? a.command.map(String) : (a.command ? ['bash', '-lc', String(a.command)] : []);
2031
+ return {
2032
+ id: t.itemId || `lsh_${rand(24)}`, type: 'local_shell_call', status, call_id: t.callId,
2033
+ action: { type: 'exec', command, timeout_ms: a.timeout_ms ?? null, working_directory: a.workdir ?? null, env: null, user: null }
2034
+ };
2035
+ }
2036
+ return {
2037
+ id: t.itemId || `fc_${rand(24)}`, type: 'function_call', status, name: meta?.name || t.name || 'tool', ...ns,
2038
+ arguments: status === 'completed' ? (t.args || '{}') : '', call_id: t.callId
2039
+ };
2040
+ }
2041
+
2042
+ function createResponsesStream(emit, model, opts = {}) {
2043
+ const toolMeta = opts.toolMeta || null;
2044
+ const respId = `resp_${Date.now()}${rand(8)}`;
2045
+ const created = Math.floor(Date.now() / 1000);
2046
+ let seq = 0;
2047
+ let nextOutput = 0;
2048
+ const output = [];
2049
+ let reasoning = null; // { id, index, text }
2050
+ let message = null; // { id, index, text }
2051
+ const tools = new Map(); // tool index -> { id, callId, index, name, args, done }
2052
+
2053
+ const send = (obj) => emit(obj.type, { ...obj, sequence_number: seq++ });
2054
+ const snapshot = (status, extra = {}) => ({
2055
+ id: respId, object: 'response', created_at: created, model, status,
2056
+ output: output.filter(Boolean), parallel_tool_calls: true, tool_choice: 'auto', tools: [],
2057
+ ...extra
2058
+ });
2059
+
2060
+ function closeReasoning() {
2061
+ if (!reasoning) return;
2062
+ const r = reasoning;
2063
+ reasoning = null;
2064
+ send({ type: 'response.reasoning_summary_text.done', item_id: r.id, output_index: r.index, summary_index: 0, text: r.text });
2065
+ send({ type: 'response.reasoning_summary_part.done', item_id: r.id, output_index: r.index, summary_index: 0, part: { type: 'summary_text', text: r.text } });
2066
+ const item = { id: r.id, type: 'reasoning', summary: [{ type: 'summary_text', text: r.text }] };
2067
+ output[r.index] = item;
2068
+ send({ type: 'response.output_item.done', output_index: r.index, item });
2069
+ }
2070
+ function closeMessage() {
2071
+ if (!message) return;
2072
+ const m = message;
2073
+ message = null;
2074
+ const part = { type: 'output_text', text: m.text, annotations: [] };
2075
+ send({ type: 'response.output_text.done', item_id: m.id, output_index: m.index, content_index: 0, text: m.text });
2076
+ send({ type: 'response.content_part.done', item_id: m.id, output_index: m.index, content_index: 0, part });
2077
+ const item = { id: m.id, type: 'message', status: 'completed', role: 'assistant', content: [part] };
2078
+ output[m.index] = item;
2079
+ send({ type: 'response.output_item.done', output_index: m.index, item });
2080
+ }
2081
+ function closeTool(t) {
2082
+ if (t.done) return;
2083
+ t.done = true;
2084
+ const item = responsesToolItem(toolMeta, { itemId: t.id, callId: t.callId, name: t.name, args: t.args });
2085
+ if (item.type === 'function_call') {
2086
+ send({ type: 'response.function_call_arguments.done', item_id: t.id, output_index: t.index, arguments: t.args });
2087
+ }
2088
+ output[t.index] = item;
2089
+ // Codex only runs the tool upon receiving output_item.done with the complete item.
2090
+ send({ type: 'response.output_item.done', output_index: t.index, item });
2091
+ }
2092
+ const closeTools = () => { for (const t of tools.values()) closeTool(t); };
2093
+
2094
+ return {
2095
+ start() {
2096
+ send({ type: 'response.created', response: snapshot('in_progress') });
2097
+ send({ type: 'response.in_progress', response: snapshot('in_progress') });
2098
+ },
2099
+ think(t) {
2100
+ if (!t) return;
2101
+ if (!reasoning) {
2102
+ closeMessage();
2103
+ closeTools();
2104
+ reasoning = { id: `rs_${rand(24)}`, index: nextOutput++, text: '' };
2105
+ send({ type: 'response.output_item.added', output_index: reasoning.index, item: { id: reasoning.id, type: 'reasoning', summary: [] } });
2106
+ send({ type: 'response.reasoning_summary_part.added', item_id: reasoning.id, output_index: reasoning.index, summary_index: 0, part: { type: 'summary_text', text: '' } });
2107
+ }
2108
+ reasoning.text += t;
2109
+ send({ type: 'response.reasoning_summary_text.delta', item_id: reasoning.id, output_index: reasoning.index, summary_index: 0, delta: t });
2110
+ },
2111
+ text(t) {
2112
+ if (!t) return;
2113
+ if (!message) {
2114
+ closeReasoning();
2115
+ closeTools();
2116
+ message = { id: `msg_${rand(24)}`, index: nextOutput++, text: '' };
2117
+ send({ type: 'response.output_item.added', output_index: message.index, item: { id: message.id, type: 'message', status: 'in_progress', role: 'assistant', content: [] } });
2118
+ send({ type: 'response.content_part.added', item_id: message.id, output_index: message.index, content_index: 0, part: { type: 'output_text', text: '', annotations: [] } });
2119
+ }
2120
+ message.text += t;
2121
+ send({ type: 'response.output_text.delta', item_id: message.id, output_index: message.index, content_index: 0, delta: t });
2122
+ },
2123
+ tool(tc) {
2124
+ const key = tc.index ?? 0;
2125
+ let st = tools.get(key);
2126
+ if (!st) {
2127
+ closeReasoning();
2128
+ closeMessage();
2129
+ const kind = toolMeta?.[tc.name]?.kind || 'function';
2130
+ const prefix = kind === 'custom' ? 'ctc' : kind === 'local_shell' ? 'lsh' : 'fc';
2131
+ st = { id: `${prefix}_${rand(24)}`, callId: tc.id || `call_${rand(24)}`, index: nextOutput++, name: tc.name || '', args: '', done: false, kind };
2132
+ tools.set(key, st);
2133
+ if (kind !== 'local_shell') {
2134
+ const added = responsesToolItem(toolMeta, { itemId: st.id, callId: st.callId, name: st.name, args: '' }, 'in_progress');
2135
+ send({ type: 'response.output_item.added', output_index: st.index, item: added });
2136
+ }
2137
+ }
2138
+ if (!st.name && tc.name) st.name = tc.name;
2139
+ if (tc.args && !st.done) {
2140
+ st.args += tc.args;
2141
+ if (st.kind === 'function') {
2142
+ send({ type: 'response.function_call_arguments.delta', item_id: st.id, output_index: st.index, delta: tc.args });
2143
+ }
2144
+ }
2145
+ },
2146
+ finish(canonical, stats = {}) {
2147
+ closeReasoning();
2148
+ closeMessage();
2149
+ const incomplete = canonical === 'length';
2150
+ // Codex runs a tool on output_item.done. After a length stop, drop any tool call whose arguments
2151
+ // do not parse (upstream arguments are JSON for every tool kind) instead of running it cut off.
2152
+ // Its output_item.added then gets no done. A done would run the call, so this is the safe side.
2153
+ if (incomplete) {
2154
+ for (const t of tools.values()) {
2155
+ if (t.done) continue;
2156
+ try { JSON.parse(t.args || '{}'); } catch { t.done = true; }
2157
+ }
2158
+ }
2159
+ closeTools();
2160
+ send({
2161
+ type: 'response.completed',
2162
+ response: snapshot(incomplete ? 'incomplete' : 'completed', {
2163
+ incomplete_details: incomplete ? { reason: 'max_output_tokens' } : null,
2164
+ usage: responsesUsage(stats.prompt, stats.completion, stats.cached, stats.reasoning)
2165
+ })
2166
+ });
2167
+ },
2168
+ error(message, code = 'server_error') {
2169
+ send({ type: 'response.failed', response: snapshot('failed', { error: { code: String(code || 'server_error'), message: String(message || 'Upstream stream error') } }) });
2170
+ }
2171
+ };
2172
+ }
2173
+
2174
+ function buildResponsesMessage({ model, think, text, tools, finish, prompt, completion, cached, reasoning, id, toolMeta, stats }) {
2175
+ const output = [];
2176
+ const thinking = (think || []).join('');
2177
+ if (thinking) output.push({ id: `rs_${rand(24)}`, type: 'reasoning', summary: [{ type: 'summary_text', text: thinking }] });
2178
+ const body = (text || []).join('');
2179
+ if (body || !(tools && tools.length)) {
2180
+ output.push({
2181
+ id: `msg_${rand(24)}`, type: 'message', status: 'completed', role: 'assistant',
2182
+ content: [{ type: 'output_text', text: body, annotations: [] }]
2183
+ });
2184
+ }
2185
+ const incomplete = finish === 'length';
2186
+ for (const tc of (tools || [])) {
2187
+ const item = responsesToolItem(toolMeta, {
2188
+ callId: tc.id || `call_${rand(24)}`, name: tc.name,
2189
+ args: typeof tc.args === 'string' ? tc.args : stringifyArgs(tc.args)
2190
+ });
2191
+ // Same rule as the stream: a length stop never hands Codex a call with truncated arguments.
2192
+ if (incomplete && typeof tc.args === 'string') {
2193
+ try { JSON.parse(tc.args || '{}'); } catch { continue; }
2194
+ }
2195
+ output.push(item);
2196
+ }
2197
+ const r = reasoning ?? stats?.reasoning;
2198
+ return {
2199
+ id: id || `resp_${Date.now()}${rand(8)}`, object: 'response', created_at: Math.floor(Date.now() / 1000), model,
2200
+ status: incomplete ? 'incomplete' : 'completed',
2201
+ incomplete_details: incomplete ? { reason: 'max_output_tokens' } : null,
2202
+ output,
2203
+ output_text: body,
2204
+ usage: responsesUsage(prompt, completion, cached, r)
2205
+ };
2206
+ }
2207
+
2208
+ // --- Vertex generateContent (native) ---
2209
+ function vertexFinish(canonical) {
2210
+ switch (canonical) {
2211
+ case 'length': return 'MAX_TOKENS';
2212
+ case 'content_filter': return 'SAFETY';
2213
+ default: return 'STOP';
2214
+ }
2215
+ }
2216
+
2217
+ function vertexUsage(prompt, completion, cached, reasoning) {
2218
+ const thoughts = Math.min(reasoning || 0, completion || 0);
2219
+ const u = { promptTokenCount: prompt || 0, candidatesTokenCount: (completion || 0) - thoughts };
2220
+ if (thoughts) u.thoughtsTokenCount = thoughts;
2221
+ u.totalTokenCount = (prompt || 0) + (completion || 0);
2222
+ if (cached) u.cachedContentTokenCount = cached;
2223
+ return u;
2224
+ }
2225
+
2226
+ function buildVertexMessage({ model, think, text, tools, finish, prompt, completion, cached, reasoning, sig }) {
2227
+ const parts = [];
2228
+ const thinking = (think || []).join('');
2229
+ if (thinking) {
2230
+ const tp = { text: thinking, thought: true };
2231
+ if (sig) tp.thoughtSignature = sig;
2232
+ parts.push(tp);
2233
+ }
2234
+ const body = (text || []).join('');
2235
+ if (body) parts.push({ text: body });
2236
+ for (const tc of (tools || [])) {
2237
+ const part = { functionCall: { name: tc.name || 'tool', args: parseArgs(tc.args) } };
2238
+ const tsig = tc.sig || lookupToolSignature(tc.id);
2239
+ if (tsig) part.thoughtSignature = tsig;
2240
+ parts.push(part);
2241
+ }
2242
+ if (!parts.length) parts.push({ text: '' });
2243
+ return {
2244
+ candidates: [{ content: { role: 'model', parts }, finishReason: vertexFinish(finish), index: 0 }],
2245
+ usageMetadata: vertexUsage(prompt, completion, cached, reasoning),
2246
+ modelVersion: model
2247
+ };
2248
+ }
2249
+
2250
+ function createVertexStream(emit, model) {
2251
+ const cand = (parts) => emit(null, { candidates: [{ content: { role: 'model', parts }, index: 0 }], modelVersion: model });
2252
+ // Tool args arrive from OpenAI/Anthropic as deltas; Vertex needs complete functionCalls -> buffer them, emit at the end.
2253
+ const tools = new Map();
2254
+ return {
2255
+ start() {},
2256
+ think(t, sig) {
2257
+ if (t) {
2258
+ const p = { text: t, thought: true };
2259
+ if (sig) p.thoughtSignature = sig;
2260
+ cand([p]);
2261
+ }
2262
+ },
2263
+ text(t) { if (t) cand([{ text: t }]); },
2264
+ tool(tc) {
2265
+ const key = tc.index ?? 0;
2266
+ if (!tools.has(key)) tools.set(key, { id: tc.id || null, name: tc.name || '', args: '', sig: null });
2267
+ const st = tools.get(key);
2268
+ if (!st.id && tc.id) st.id = tc.id;
2269
+ if (!st.sig && tc.sig) st.sig = tc.sig;
2270
+ if (!st.name && tc.name) st.name = tc.name;
2271
+ if (tc.args) st.args += tc.args;
2272
+ },
2273
+ finish(canonical, stats = {}) {
2274
+ const parts = [...tools.values()]
2275
+ .filter(t => t.name)
2276
+ .map(t => {
2277
+ const part = { functionCall: { name: t.name, args: parseArgs(t.args || '{}') } };
2278
+ const tsig = t.sig || lookupToolSignature(t.id);
2279
+ if (tsig) part.thoughtSignature = tsig;
2280
+ return part;
2281
+ });
2282
+ emit(null, {
2283
+ candidates: [{ content: { role: 'model', parts }, finishReason: vertexFinish(canonical), index: 0 }],
2284
+ usageMetadata: vertexUsage(stats.prompt, stats.completion, stats.cached, stats.reasoning),
2285
+ modelVersion: model
2286
+ });
2287
+ },
2288
+ error(message) {
2289
+ emit(null, { error: { code: 500, message: String(message || 'Upstream stream error'), status: 'INTERNAL' } });
2290
+ }
2291
+ };
2292
+ }
2293
+
2294
+ export {
2295
+ canonFinish, chatFinish, anthropicStopReason,
2296
+ smartReasoning, smartText, smartToolCalls, smartUsage, smartFinish,
2297
+ firstChoice, smartDelta, sanitizeJsonSchema, toGeminiSchema, splitParts,
2298
+ budgetToEffort, effortToBudget, clampBudget, parseArgs, stringifyArgs,
2299
+ anthropicToIR, chatToIR, responsesToIR, vertexToIR, parseToIR,
2300
+ healToolPairs, healAnthropicPayload, rememberToolSignature, lookupToolSignature, estimateTokens, irToChatBody, irToAnthropicBody, irToVertexBody, emitUpstreamBody,
2301
+ normalizeUpstream, createUpstreamNormalizer, createCollector,
2302
+ createThinkTagSplitter, splitThinkTags,
2303
+ createAnthropicStream, buildAnthropicMessage,
2304
+ createChatStream, buildChatMessage,
2305
+ createResponsesStream, buildResponsesMessage,
2306
+ vertexFinish, createVertexStream, buildVertexMessage,
2307
+ isAntigravityModel
2308
+ };