llm-switcher 1.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.gitattributes +16 -0
- package/LICENSE +21 -0
- package/README.md +587 -0
- package/README.vi.md +585 -0
- package/blindfold/blindfold.mjs +633 -0
- package/blindfold/make-certs.sh +88 -0
- package/blindfold/wsframe.mjs +176 -0
- package/codex-catalog-template.json +1 -0
- package/config.example.json +84 -0
- package/contract-exclusions.json +41 -0
- package/contract.mjs +561 -0
- package/docs/LLM-RESPONSE-MATRIX.md +165 -0
- package/docs/TOKEN-OPTIMIZER-INTEROP.md +110 -0
- package/docs/codex-blindfold.md +214 -0
- package/docs/cross-platform.md +136 -0
- package/docs/diagrams/blindfold-request-routing.html +14972 -0
- package/docs/diagrams/blindfold-request-routing.sequence.json +175 -0
- package/docs/diagrams/blindfold-switch-lifecycle.html +14958 -0
- package/docs/diagrams/blindfold-switch-lifecycle.lifecycle.json +159 -0
- package/docs/diagrams/codex-model-name-resolution.html +15005 -0
- package/docs/diagrams/codex-model-name-resolution.workflow.json +71 -0
- package/docs/response-matrix.json +1131 -0
- package/formats.mjs +2308 -0
- package/mcp.mjs +340 -0
- package/package.json +36 -0
- package/proxy.mjs +1743 -0
- package/service.mjs +132 -0
- package/shim.mjs +292 -0
- package/skills/llm-switcher/SKILL.md +88 -0
- package/state.mjs +978 -0
- package/switch +5 -0
- package/switch.cmd +2 -0
- package/switch.mjs +930 -0
- package/tests/blindfold.test.mjs +307 -0
- package/tests/blindfold.wire.test.mjs +170 -0
- package/tests/contract/run.test.mjs +214 -0
- package/tests/contract-check.test.mjs +458 -0
- package/tests/contract-lab.test.mjs +755 -0
- package/tests/datadir.test.mjs +37 -0
- package/tests/formats.test.mjs +794 -0
- package/tests/gateway.e2e.test.mjs +999 -0
- package/tests/helpers.mjs +24 -0
- package/tests/lifecycle.test.mjs +416 -0
- package/tests/live-optimizer-interop.mjs +205 -0
- package/tests/mcp.test.mjs +91 -0
- package/tests/service.test.mjs +69 -0
- package/tests/shim.test.mjs +228 -0
- package/tests/state.test.mjs +675 -0
- package/tests/switch.test.mjs +156 -0
- package/tests/wsframe.test.mjs +154 -0
- package/ui.html +2234 -0
|
@@ -0,0 +1,794 @@
|
|
|
1
|
+
// Unit tests for formats.mjs — runs offline: node --test tests/
|
|
2
|
+
import { test } from 'node:test';
|
|
3
|
+
import assert from 'node:assert/strict';
|
|
4
|
+
import {
|
|
5
|
+
anthropicToIR, chatToIR, responsesToIR, vertexToIR,
|
|
6
|
+
healToolPairs, irToChatBody, irToAnthropicBody, irToVertexBody, toGeminiSchema,
|
|
7
|
+
createUpstreamNormalizer, createCollector, createThinkTagSplitter,
|
|
8
|
+
createAnthropicStream, createResponsesStream, createVertexStream,
|
|
9
|
+
healAnthropicPayload, estimateTokens, PLACEHOLDER_SIGNATURE, GEMINI_DUMMY_SIGNATURE,
|
|
10
|
+
buildResponsesMessage, createUpstreamNormalizer as makeNormalizer, emitUpstreamBody,
|
|
11
|
+
isAntigravityModel, smartUsage,
|
|
12
|
+
createChatStream, buildChatMessage, chatFinish, anthropicStopReason,
|
|
13
|
+
smartText, smartReasoning, sanitizeJsonSchema, normalizeUpstream, buildVertexMessage
|
|
14
|
+
} from '../formats.mjs';
|
|
15
|
+
import { assertValidAnthropicEvents } from './helpers.mjs';
|
|
16
|
+
|
|
17
|
+
test('anthropicToIR: stream defaults to false and thinking budget_tokens is preserved', () => {
|
|
18
|
+
const ir = anthropicToIR({ model: 'claude-opus-4-6', max_tokens: 64000, messages: [{ role: 'user', content: 'hi' }], thinking: { type: 'enabled', budget_tokens: 31999 } });
|
|
19
|
+
assert.equal(ir.stream, false);
|
|
20
|
+
assert.deepEqual(ir.thinking, { type: 'enabled', budget: 31999 });
|
|
21
|
+
const body = irToChatBody(ir, 'ag/claude-opus-4-6-thinking');
|
|
22
|
+
assert.equal(body.thinking.budget_tokens, 31999);
|
|
23
|
+
});
|
|
24
|
+
|
|
25
|
+
test('anthropicToIR: explicit thinking disabled is not "restored" for reasoning models', () => {
|
|
26
|
+
const ir = anthropicToIR({ model: 'x', messages: [{ role: 'user', content: 'hi' }], thinking: { type: 'disabled' } });
|
|
27
|
+
const body = irToChatBody(ir, 'ag/claude-opus-4-6-thinking');
|
|
28
|
+
assert.equal(body.thinking, undefined);
|
|
29
|
+
assert.equal(body.reasoning_effort, undefined);
|
|
30
|
+
});
|
|
31
|
+
|
|
32
|
+
test('Chat emitter restores thinking for versioned Claude Opus model IDs', () => {
|
|
33
|
+
const ir = anthropicToIR({ model: 'x', messages: [{ role: 'user', content: 'hi' }] });
|
|
34
|
+
for (const model of ['ag/claude-opus-4-6', 'ag/claude-opus-4-7', 'claude-opus-5']) {
|
|
35
|
+
const body = irToChatBody(ir, model);
|
|
36
|
+
assert.ok(body.thinking, `${model} should receive restored thinking settings`);
|
|
37
|
+
}
|
|
38
|
+
});
|
|
39
|
+
|
|
40
|
+
test('emitUpstreamBody: billing header stripped only for antigravity (ag/) models', () => {
|
|
41
|
+
const header = 'x-anthropic-billing-header: cc_version=2.1.275.f15; cc_entrypoint=cli;';
|
|
42
|
+
const fromArray = anthropicToIR({
|
|
43
|
+
model: 'x', messages: [{ role: 'user', content: 'hi' }],
|
|
44
|
+
system: [{ type: 'text', text: header }, { type: 'text', text: 'You are Claude Code.' }]
|
|
45
|
+
});
|
|
46
|
+
const fromString = anthropicToIR({ model: 'x', messages: [{ role: 'user', content: 'hi' }], system: `${header}\n\nYou are Claude Code.` });
|
|
47
|
+
for (const ir of [fromArray, fromString]) {
|
|
48
|
+
assert.equal(emitUpstreamBody('openai-chat', ir, 'ag/gemini-3.8-flash').messages[0].content, 'You are Claude Code.');
|
|
49
|
+
assert.equal(emitUpstreamBody('vertex', ir, 'antigravity/gemini-3.8-flash').systemInstruction.parts[0].text, 'You are Claude Code.');
|
|
50
|
+
for (const model of ['cc/claude-opus-4-7', 'claude-sonnet-4-6', 'openrouter/x-ai/grok']) {
|
|
51
|
+
assert.ok(emitUpstreamBody('openai-chat', ir, model).messages[0].content.startsWith(header), model);
|
|
52
|
+
}
|
|
53
|
+
}
|
|
54
|
+
});
|
|
55
|
+
|
|
56
|
+
test('healAnthropicPayload: native Anthropic passthrough keeps the billing header', () => {
|
|
57
|
+
const header = 'x-anthropic-billing-header: cc_version=2.1.275.f15; cc_entrypoint=cli;';
|
|
58
|
+
const { payload } = healAnthropicPayload({
|
|
59
|
+
model: 'claude-opus-4-6', messages: [{ role: 'user', content: 'hi' }],
|
|
60
|
+
system: [{ type: 'text', text: header }, { type: 'text', text: 'You are Claude Code.' }]
|
|
61
|
+
});
|
|
62
|
+
assert.equal(payload.system[0].text, header);
|
|
63
|
+
});
|
|
64
|
+
|
|
65
|
+
test('chatToIR: reasoning_effort "none" disables thinking', () => {
|
|
66
|
+
const ir = chatToIR({ model: 'x', messages: [{ role: 'user', content: 'hi' }], reasoning_effort: 'none' });
|
|
67
|
+
assert.equal(ir.thinking.type, 'disabled');
|
|
68
|
+
});
|
|
69
|
+
|
|
70
|
+
test('healToolPairs: orphan result -> user text, missing result -> placeholder, adjacency kept', () => {
|
|
71
|
+
const healed = healToolPairs([
|
|
72
|
+
{ role: 'user', content: 'start' },
|
|
73
|
+
{ role: 'tool', toolCallId: 'gone', content: 'orphan output' },
|
|
74
|
+
{ role: 'assistant', toolCalls: [{ id: 'a', name: 'f', args: {} }, { id: 'b', name: 'g', args: {} }] },
|
|
75
|
+
{ role: 'tool', toolCallId: 'x', content: 'stray' },
|
|
76
|
+
{ role: 'tool', toolCallId: 'a', content: 'A' },
|
|
77
|
+
{ role: 'user', content: 'next' }
|
|
78
|
+
]);
|
|
79
|
+
assert.deepEqual(healed.map(m => m.role), ['user', 'user', 'assistant', 'tool', 'tool', 'user', 'user']);
|
|
80
|
+
assert.match(healed[1].content, /orphan output/);
|
|
81
|
+
assert.equal(healed[3].toolCallId, 'a');
|
|
82
|
+
assert.equal(healed[4].toolCallId, 'b');
|
|
83
|
+
assert.match(healed[4].content, /unavailable/);
|
|
84
|
+
assert.match(healed[5].content, /stray/);
|
|
85
|
+
});
|
|
86
|
+
|
|
87
|
+
test('irToChatBody: every tool_calls message is immediately followed by its tool results', () => {
|
|
88
|
+
const ir = anthropicToIR({
|
|
89
|
+
model: 'x', messages: [
|
|
90
|
+
{ role: 'user', content: 'go' },
|
|
91
|
+
{ role: 'assistant', content: [{ type: 'text', text: 'calling' }, { type: 'tool_use', id: 't1', name: 'read', input: { p: 1 } }] },
|
|
92
|
+
{ role: 'user', content: [{ type: 'text', text: 'interrupt' }] },
|
|
93
|
+
{ role: 'user', content: [{ type: 'tool_result', tool_use_id: 't1', content: 'late' }] }
|
|
94
|
+
]
|
|
95
|
+
});
|
|
96
|
+
const { messages } = irToChatBody(ir, 'gpt-4o');
|
|
97
|
+
const i = messages.findIndex(m => m.tool_calls);
|
|
98
|
+
assert.equal(messages[i + 1].role, 'tool');
|
|
99
|
+
assert.equal(messages[i + 1].tool_call_id, 't1');
|
|
100
|
+
assert.ok(!messages.slice(i + 2).some(m => m.role === 'tool'), 'late result must not be emitted as a dangling tool message');
|
|
101
|
+
});
|
|
102
|
+
|
|
103
|
+
test('responsesToIR: parallel function_call items merge into one assistant turn; developer -> system', () => {
|
|
104
|
+
const ir = responsesToIR({
|
|
105
|
+
model: 'x', instructions: 'sys', input: [
|
|
106
|
+
{ type: 'message', role: 'developer', content: [{ type: 'input_text', text: 'dev rules' }] },
|
|
107
|
+
{ role: 'user', content: 'do two things' },
|
|
108
|
+
{ type: 'function_call', call_id: 'c1', name: 'a', arguments: '{}' },
|
|
109
|
+
{ type: 'function_call', call_id: 'c2', name: 'b', arguments: '{}' },
|
|
110
|
+
{ type: 'function_call_output', call_id: 'c1', output: '1' },
|
|
111
|
+
{ type: 'function_call_output', call_id: 'c2', output: '2' }
|
|
112
|
+
]
|
|
113
|
+
});
|
|
114
|
+
assert.equal(ir.system, 'sys\n\ndev rules');
|
|
115
|
+
const { messages } = irToChatBody(ir, 'gpt-4o');
|
|
116
|
+
const roles = messages.map(m => m.role);
|
|
117
|
+
assert.deepEqual(roles, ['system', 'user', 'assistant', 'tool', 'tool']);
|
|
118
|
+
assert.equal(messages[2].tool_calls.length, 2);
|
|
119
|
+
});
|
|
120
|
+
|
|
121
|
+
test('vertexToIR: functionCall/functionResponse are paired by generated ids', () => {
|
|
122
|
+
const ir = vertexToIR({
|
|
123
|
+
contents: [
|
|
124
|
+
{ role: 'user', parts: [{ text: 'weather?' }] },
|
|
125
|
+
{ role: 'model', parts: [{ functionCall: { name: 'w', args: { c: 'A' } } }, { functionCall: { name: 'w', args: { c: 'B' } } }] },
|
|
126
|
+
{ role: 'function', parts: [{ functionResponse: { name: 'w', response: { t: 1 } } }, { functionResponse: { name: 'w', response: { t: 2 } } }] }
|
|
127
|
+
]
|
|
128
|
+
});
|
|
129
|
+
const calls = ir.messages[1].toolCalls.map(t => t.id);
|
|
130
|
+
const results = ir.messages.filter(m => m.role === 'tool').map(m => m.toolCallId);
|
|
131
|
+
assert.deepEqual(results, calls);
|
|
132
|
+
const { messages } = irToChatBody(ir, 'gpt-4o');
|
|
133
|
+
assert.equal(messages.filter(m => m.role === 'tool').length, 2);
|
|
134
|
+
});
|
|
135
|
+
|
|
136
|
+
test('irToVertexBody: functionResponse uses the function name and parallel responses share one content', () => {
|
|
137
|
+
const ir = anthropicToIR({
|
|
138
|
+
model: 'x', messages: [
|
|
139
|
+
{ role: 'user', content: 'go' },
|
|
140
|
+
{ role: 'assistant', content: [{ type: 'tool_use', id: 'toolu_1', name: 'read', input: {} }, { type: 'tool_use', id: 'toolu_2', name: 'grep', input: {} }] },
|
|
141
|
+
{ role: 'user', content: [{ type: 'tool_result', tool_use_id: 'toolu_1', content: '[1,2]' }, { type: 'tool_result', tool_use_id: 'toolu_2', content: 'ok' }] }
|
|
142
|
+
],
|
|
143
|
+
thinking: { type: 'enabled', budget_tokens: 2048 }
|
|
144
|
+
});
|
|
145
|
+
const body = irToVertexBody(ir);
|
|
146
|
+
const fr = body.contents[2].parts.map(p => p.functionResponse);
|
|
147
|
+
assert.deepEqual(fr.map(f => f.name), ['read', 'grep']);
|
|
148
|
+
assert.deepEqual(fr[0].response, { result: [1, 2] });
|
|
149
|
+
assert.equal(body.generationConfig.thinkingConfig.includeThoughts, true);
|
|
150
|
+
});
|
|
151
|
+
|
|
152
|
+
test('toGeminiSchema strips unsupported JSON Schema keywords', () => {
|
|
153
|
+
const s = toGeminiSchema({
|
|
154
|
+
$schema: 'http://json-schema.org/draft-07/schema#', type: 'object', additionalProperties: false,
|
|
155
|
+
properties: { a: { type: ['string', 'null'], format: 'uri' }, b: { const: 'x' }, additionalProperties: { type: 'number' } },
|
|
156
|
+
required: ['a', 'zzz']
|
|
157
|
+
});
|
|
158
|
+
assert.equal(s.$schema, undefined);
|
|
159
|
+
assert.equal(s.additionalProperties, undefined);
|
|
160
|
+
assert.deepEqual(s.properties.a, { type: 'string', nullable: true });
|
|
161
|
+
assert.deepEqual(s.properties.b, { enum: ['x'], type: 'string' });
|
|
162
|
+
assert.ok(s.properties.additionalProperties, 'a property literally named additionalProperties is kept');
|
|
163
|
+
assert.deepEqual(s.required, ['a']);
|
|
164
|
+
});
|
|
165
|
+
|
|
166
|
+
test('irToAnthropicBody: thinking constraints (budget < max_tokens, no top_k, temperature 1)', () => {
|
|
167
|
+
const ir = chatToIR({ model: 'x', messages: [{ role: 'user', content: 'hi' }], max_tokens: 3000, temperature: 0.2, reasoning_effort: 'high' });
|
|
168
|
+
ir.params.topK = 5;
|
|
169
|
+
const body = irToAnthropicBody(ir, 'claude-opus-4-6');
|
|
170
|
+
assert.ok(body.thinking.budget_tokens < body.max_tokens);
|
|
171
|
+
assert.equal(body.temperature, undefined);
|
|
172
|
+
assert.equal(body.top_k, undefined);
|
|
173
|
+
|
|
174
|
+
const small = irToAnthropicBody(chatToIR({ model: 'x', messages: [{ role: 'user', content: 'hi' }], max_tokens: 100, reasoning_effort: 'high' }), 'claude');
|
|
175
|
+
assert.equal(small.thinking, undefined, 'thinking dropped when max_tokens <= 1024');
|
|
176
|
+
});
|
|
177
|
+
|
|
178
|
+
test('irToAnthropicBody: never emits empty text blocks', () => {
|
|
179
|
+
const ir = chatToIR({ model: 'x', messages: [{ role: 'user', content: [{ type: 'text', text: '' }, { type: 'text', text: 'hi' }] }] });
|
|
180
|
+
const body = irToAnthropicBody(ir, 'claude');
|
|
181
|
+
assert.ok(body.messages.every(m => m.content.every(b => b.type !== 'text' || b.text)));
|
|
182
|
+
});
|
|
183
|
+
|
|
184
|
+
test('normalizer: usage-only OpenAI chunk is counted; vertex tool calls get distinct indexes', () => {
|
|
185
|
+
const chat = createUpstreamNormalizer('openai-chat');
|
|
186
|
+
const col = createCollector();
|
|
187
|
+
col.add(chat({ choices: [{ delta: { content: 'hi' } }] }));
|
|
188
|
+
col.add(chat({ choices: [], usage: { prompt_tokens: 1200, completion_tokens: 7, prompt_tokens_details: { cached_tokens: 1000 } } }));
|
|
189
|
+
assert.equal(col.prompt, 1200);
|
|
190
|
+
assert.equal(col.completion(), 7);
|
|
191
|
+
assert.equal(col.cached, 1000);
|
|
192
|
+
|
|
193
|
+
const vtx = createUpstreamNormalizer('vertex');
|
|
194
|
+
const a = vtx({ candidates: [{ content: { parts: [{ functionCall: { name: 'f', args: { x: 1 } } }] } }] });
|
|
195
|
+
const b = vtx({ candidates: [{ content: { parts: [{ functionCall: { name: 'g', args: { y: 2 } } }] } }] });
|
|
196
|
+
assert.notEqual(a.tools[0].index, b.tools[0].index);
|
|
197
|
+
});
|
|
198
|
+
|
|
199
|
+
test('normalizer: Anthropic block indexes are remapped to 0-based tool indexes', () => {
|
|
200
|
+
const n = createUpstreamNormalizer('anthropic');
|
|
201
|
+
const start = n({ type: 'content_block_start', index: 2, content_block: { type: 'tool_use', id: 'toolu_x', name: 'f' } });
|
|
202
|
+
const delta = n({ type: 'content_block_delta', index: 2, delta: { type: 'input_json_delta', partial_json: '{}' } });
|
|
203
|
+
assert.equal(start.tools[0].index, 0);
|
|
204
|
+
assert.equal(delta.tools[0].index, 0);
|
|
205
|
+
});
|
|
206
|
+
|
|
207
|
+
test('think tag splitter handles tags split across chunks', () => {
|
|
208
|
+
const think = [];
|
|
209
|
+
const text = [];
|
|
210
|
+
const sp = createThinkTagSplitter(t => think.push(t), t => text.push(t));
|
|
211
|
+
for (const c of ['<thi', 'nk>plan', ' A</th', 'ink>', 'answer a < b']) sp.push(c);
|
|
212
|
+
sp.flush();
|
|
213
|
+
assert.equal(think.join(''), 'plan A');
|
|
214
|
+
assert.equal(text.join(''), 'answer a < b');
|
|
215
|
+
});
|
|
216
|
+
|
|
217
|
+
test('Anthropic renderer: think -> tool -> think -> text produces a valid event sequence', () => {
|
|
218
|
+
const events = [];
|
|
219
|
+
const r = createAnthropicStream((event, data) => events.push({ event, data }), 'm');
|
|
220
|
+
r.start();
|
|
221
|
+
r.think('a');
|
|
222
|
+
r.tool({ index: 0, id: 'c1', name: 'f', args: '{"x"' });
|
|
223
|
+
r.tool({ index: 0, args: ':1}' });
|
|
224
|
+
r.think('b', 'sig123');
|
|
225
|
+
r.text('hello');
|
|
226
|
+
r.tool({ index: 1, id: 'c2', name: 'g', args: '{}' });
|
|
227
|
+
r.text('more');
|
|
228
|
+
r.finish('stop', { prompt: 100, completion: 5, cached: 40, hasTools: true });
|
|
229
|
+
assertValidAnthropicEvents(events);
|
|
230
|
+
const md = events.find(e => e.event === 'message_delta').data;
|
|
231
|
+
assert.equal(md.delta.stop_reason, 'tool_use');
|
|
232
|
+
assert.equal(md.usage.input_tokens, 60);
|
|
233
|
+
assert.equal(md.usage.cache_read_input_tokens, 40);
|
|
234
|
+
assert.ok(events.some(e => e.data.delta?.signature === 'lsw1.sig123'), 'foreign signature is wrapped, never passed off as an Anthropic signature');
|
|
235
|
+
});
|
|
236
|
+
|
|
237
|
+
test('Responses renderer: emits output_item.done for every item with full content', () => {
|
|
238
|
+
const events = [];
|
|
239
|
+
const r = createResponsesStream((event, data) => events.push({ event, data }), 'm');
|
|
240
|
+
r.start();
|
|
241
|
+
r.think('reason');
|
|
242
|
+
r.text('Hel');
|
|
243
|
+
r.text('lo');
|
|
244
|
+
r.tool({ index: 0, id: 'call_1', name: 'shell', args: '{"cmd":' });
|
|
245
|
+
r.tool({ index: 0, args: '"ls"}' });
|
|
246
|
+
r.finish('stop', { prompt: 10, completion: 3 });
|
|
247
|
+
|
|
248
|
+
const seqs = events.map(e => e.data.sequence_number);
|
|
249
|
+
assert.deepEqual(seqs, [...seqs].sort((a, b) => a - b));
|
|
250
|
+
assert.equal(new Set(seqs).size, seqs.length);
|
|
251
|
+
const done = events.filter(e => e.event === 'response.output_item.done').map(e => e.data.item);
|
|
252
|
+
assert.deepEqual(done.map(i => i.type), ['reasoning', 'message', 'function_call']);
|
|
253
|
+
assert.equal(done[1].content[0].text, 'Hello');
|
|
254
|
+
assert.equal(done[2].arguments, '{"cmd":"ls"}');
|
|
255
|
+
assert.equal(done[2].call_id, 'call_1');
|
|
256
|
+
const completed = events.at(-1).data.response;
|
|
257
|
+
assert.equal(completed.output.length, 3);
|
|
258
|
+
assert.equal(completed.usage.total_tokens, 13);
|
|
259
|
+
});
|
|
260
|
+
|
|
261
|
+
test('Vertex renderer: streamed tool deltas are emitted as one complete functionCall', () => {
|
|
262
|
+
const chunks = [];
|
|
263
|
+
const r = createVertexStream((_, d) => chunks.push(d), 'm');
|
|
264
|
+
r.tool({ index: 0, id: 'c', name: 'f', args: '{"a"' });
|
|
265
|
+
r.tool({ index: 0, args: ':1}' });
|
|
266
|
+
r.finish('tool_calls', {});
|
|
267
|
+
const calls = chunks.flatMap(c => c.candidates[0].content.parts).filter(p => p.functionCall);
|
|
268
|
+
assert.deepEqual(calls, [{ functionCall: { name: 'f', args: { a: 1 } } }]);
|
|
269
|
+
});
|
|
270
|
+
|
|
271
|
+
test('thinkingMode: native sends only reasoning_effort, no prompt injection; off sends nothing', () => {
|
|
272
|
+
const ir = anthropicToIR({ model: 'x', max_tokens: 8000, system: 'SYS', messages: [{ role: 'user', content: 'hi' }], thinking: { type: 'enabled', budget_tokens: 10000 } });
|
|
273
|
+
const auto = irToChatBody(ir, 'gpt-4o');
|
|
274
|
+
assert.ok(auto.thinking && auto.messages[0].content.includes('<think>'));
|
|
275
|
+
|
|
276
|
+
const native = irToChatBody(ir, 'gpt-4o', { thinkingMode: 'native' });
|
|
277
|
+
assert.equal(native.thinking, undefined);
|
|
278
|
+
assert.equal(native.reasoning_effort, 'high');
|
|
279
|
+
assert.equal(native.messages[0].content, 'SYS');
|
|
280
|
+
assert.equal(native.max_completion_tokens, 8000);
|
|
281
|
+
assert.equal(native.max_tokens, undefined);
|
|
282
|
+
|
|
283
|
+
const restoreIr = anthropicToIR({ model: 'x', messages: [{ role: 'user', content: 'hi' }] });
|
|
284
|
+
assert.equal(irToChatBody(restoreIr, 'ag/claude-opus-4-6-thinking', { thinkingMode: 'native' }).reasoning_effort, undefined, 'native does not restore');
|
|
285
|
+
|
|
286
|
+
const off = irToChatBody(ir, 'ag/claude-opus-4-6-thinking', { thinkingMode: 'off' });
|
|
287
|
+
assert.equal(off.thinking, undefined);
|
|
288
|
+
assert.equal(off.reasoning_effort, undefined);
|
|
289
|
+
});
|
|
290
|
+
|
|
291
|
+
test('healAnthropicPayload: untouched when valid (bytes can be forwarded as-is)', () => {
|
|
292
|
+
const payload = {
|
|
293
|
+
model: 'claude', thinking: { type: 'enabled', budget_tokens: 2048 }, messages: [
|
|
294
|
+
{ role: 'user', content: 'go' },
|
|
295
|
+
{ role: 'assistant', content: [{ type: 'thinking', thinking: 'x', signature: 'EqRealSig' }, { type: 'tool_use', id: 't1', name: 'f', input: {} }] },
|
|
296
|
+
{ role: 'user', content: [{ type: 'tool_result', tool_use_id: 't1', content: 'ok', cache_control: { type: 'ephemeral' } }] }
|
|
297
|
+
]
|
|
298
|
+
};
|
|
299
|
+
const r = healAnthropicPayload(payload);
|
|
300
|
+
assert.equal(r.changed, false);
|
|
301
|
+
assert.equal(r.payload, payload);
|
|
302
|
+
});
|
|
303
|
+
|
|
304
|
+
test('healAnthropicPayload: orphan/missing/misplaced tool_result and placeholder thinking', () => {
|
|
305
|
+
const r = healAnthropicPayload({
|
|
306
|
+
model: 'claude', thinking: { type: 'enabled', budget_tokens: 2048 }, messages: [
|
|
307
|
+
{ role: 'user', content: [{ type: 'tool_result', tool_use_id: 'pruned', content: 'old' }, { type: 'text', text: 'hi' }] },
|
|
308
|
+
{ role: 'assistant', content: [{ type: 'thinking', thinking: 'fake', signature: PLACEHOLDER_SIGNATURE }, { type: 'tool_use', id: 'a', name: 'f', input: {} }, { type: 'tool_use', id: 'b', name: 'g', input: {} }] },
|
|
309
|
+
{ role: 'user', content: [{ type: 'text', text: 'note' }] },
|
|
310
|
+
{ role: 'user', content: [{ type: 'tool_result', tool_use_id: 'a', content: 'A' }] }
|
|
311
|
+
]
|
|
312
|
+
});
|
|
313
|
+
assert.equal(r.changed, true);
|
|
314
|
+
const [first, assistant, last] = r.payload.messages;
|
|
315
|
+
assert.deepEqual(first.content.map(b => b.type), ['text', 'text']);
|
|
316
|
+
assert.deepEqual(assistant.content.map(b => b.type), ['tool_use', 'tool_use'], 'placeholder-signed thinking stripped');
|
|
317
|
+
assert.deepEqual(last.content.map(b => b.type), ['tool_result', 'tool_result', 'text'], 'results first, merged user turns');
|
|
318
|
+
assert.equal(last.content[0].tool_use_id, 'a');
|
|
319
|
+
assert.equal(last.content[1].tool_use_id, 'b');
|
|
320
|
+
assert.equal(r.payload.thinking, undefined, 'thinking disabled because last assistant turn lost its thinking block');
|
|
321
|
+
assert.equal(r.payload.messages.length, 3);
|
|
322
|
+
});
|
|
323
|
+
|
|
324
|
+
test('estimateTokens ignores base64 payloads', () => {
|
|
325
|
+
const big = 'A'.repeat(400000);
|
|
326
|
+
const n = estimateTokens({ messages: [{ role: 'user', content: [{ type: 'text', text: 'x'.repeat(400) }, { type: 'image', source: { type: 'base64', data: big } }] }] });
|
|
327
|
+
assert.ok(n < 2000, String(n));
|
|
328
|
+
});
|
|
329
|
+
|
|
330
|
+
const APPLY_PATCH_TOOL = { type: 'custom', name: 'apply_patch', description: 'Edit files.', format: { type: 'grammar', syntax: 'lark', definition: 'start: begin_patch hunk+ end_patch' } };
|
|
331
|
+
|
|
332
|
+
test('Codex tools: custom -> function(input), namespace flattened, local_shell, hosted tools dropped', () => {
|
|
333
|
+
const ir = responsesToIR({
|
|
334
|
+
model: 'x',
|
|
335
|
+
tools: [
|
|
336
|
+
APPLY_PATCH_TOOL,
|
|
337
|
+
{ type: 'namespace', name: 'mcp_fs', description: 'fs', tools: [{ type: 'function', name: 'read', parameters: { type: 'object', properties: { p: { type: 'string' } } } }] },
|
|
338
|
+
{ type: 'local_shell' },
|
|
339
|
+
{ type: 'web_search', external_web_access: true }
|
|
340
|
+
],
|
|
341
|
+
input: [
|
|
342
|
+
{ type: 'message', role: 'user', content: [{ type: 'input_text', text: 'patch it' }] },
|
|
343
|
+
{ type: 'custom_tool_call', call_id: 'c1', name: 'apply_patch', input: '*** Begin Patch\n*** End Patch' },
|
|
344
|
+
{ type: 'function_call', call_id: 'c2', namespace: 'mcp_fs', name: 'read', arguments: '{"p":"a"}' },
|
|
345
|
+
{ type: 'custom_tool_call_output', call_id: 'c1', output: [{ type: 'input_text', text: 'Done!' }] },
|
|
346
|
+
{ type: 'function_call_output', call_id: 'c2', output: 'contents' }
|
|
347
|
+
]
|
|
348
|
+
});
|
|
349
|
+
assert.deepEqual(ir.tools.map(t => t.name), ['apply_patch', 'mcp_fs__read', 'local_shell']);
|
|
350
|
+
assert.deepEqual(ir.tools[0].parameters.required, ['input']);
|
|
351
|
+
assert.match(ir.tools[0].description, /start: begin_patch/);
|
|
352
|
+
const { messages } = irToChatBody(ir, 'gpt-4o');
|
|
353
|
+
const call = messages.find(m => m.tool_calls);
|
|
354
|
+
assert.deepEqual(call.tool_calls.map(t => t.function.name), ['apply_patch', 'mcp_fs__read']);
|
|
355
|
+
assert.deepEqual(JSON.parse(call.tool_calls[0].function.arguments), { input: '*** Begin Patch\n*** End Patch' });
|
|
356
|
+
assert.equal(messages.find(m => m.tool_call_id === 'c1').content, 'Done!');
|
|
357
|
+
});
|
|
358
|
+
|
|
359
|
+
test('Codex tools: responses output items restore custom_tool_call / namespace / local_shell_call', () => {
|
|
360
|
+
const ir = responsesToIR({ model: 'x', input: 'go', tools: [APPLY_PATCH_TOOL, { type: 'namespace', name: 'mcp_fs', tools: [{ type: 'function', name: 'read' }] }, { type: 'local_shell' }] });
|
|
361
|
+
const events = [];
|
|
362
|
+
const r = createResponsesStream((event, data) => events.push({ event, data }), 'm', { toolMeta: ir.toolMeta });
|
|
363
|
+
r.start();
|
|
364
|
+
r.tool({ index: 0, id: 'call_p', name: 'apply_patch', args: '{"input":"*** Begin' });
|
|
365
|
+
r.tool({ index: 0, args: ' Patch\n*** End Patch"}' });
|
|
366
|
+
r.tool({ index: 1, id: 'call_r', name: 'mcp_fs__read', args: '{"p":"a"}' });
|
|
367
|
+
r.tool({ index: 2, id: 'call_s', name: 'local_shell', args: '{"command":["ls","-la"]}' });
|
|
368
|
+
r.finish('tool_calls', {});
|
|
369
|
+
const done = events.filter(e => e.event === 'response.output_item.done').map(e => e.data.item);
|
|
370
|
+
assert.deepEqual(done[0], { id: done[0].id, type: 'custom_tool_call', status: 'completed', call_id: 'call_p', name: 'apply_patch', input: '*** Begin Patch\n*** End Patch' });
|
|
371
|
+
assert.equal(done[1].type, 'function_call');
|
|
372
|
+
assert.equal(done[1].name, 'read');
|
|
373
|
+
assert.equal(done[1].namespace, 'mcp_fs');
|
|
374
|
+
assert.deepEqual(done[2].action.command, ['ls', '-la']);
|
|
375
|
+
assert.equal(done[2].type, 'local_shell_call');
|
|
376
|
+
assert.ok(!events.some(e => e.event === 'response.function_call_arguments.delta' && e.data.item_id === done[0].id), 'no function deltas for custom tools');
|
|
377
|
+
|
|
378
|
+
const msg = buildResponsesMessage({ model: 'm', text: [''], tools: [{ id: 'call_p', name: 'apply_patch', args: '{"input":"X"}' }], toolMeta: ir.toolMeta });
|
|
379
|
+
assert.deepEqual(msg.output.map(o => o.type), ['custom_tool_call']);
|
|
380
|
+
assert.equal(msg.output[0].input, 'X');
|
|
381
|
+
});
|
|
382
|
+
|
|
383
|
+
test('Gemini: captured thoughtSignature is replayed on the functionCall part; dummy only for Gemini 3 current turn', () => {
|
|
384
|
+
const n = makeNormalizer('vertex');
|
|
385
|
+
const ev = n({ candidates: [{ content: { role: 'model', parts: [{ functionCall: { name: 'w', args: {} }, thoughtSignature: 'SIG_1' }, { functionCall: { name: 'w', args: { b: 1 } } }] } }] });
|
|
386
|
+
const [a, b] = ev.tools;
|
|
387
|
+
assert.ok(a.id && b.id && a.id !== b.id);
|
|
388
|
+
|
|
389
|
+
const history = (ids) => anthropicToIR({
|
|
390
|
+
model: 'x', messages: [
|
|
391
|
+
{ role: 'user', content: 'weather' },
|
|
392
|
+
{ role: 'assistant', content: ids.map((id, i) => ({ type: 'tool_use', id, name: 'w', input: { i } })) },
|
|
393
|
+
{ role: 'user', content: ids.map(id => ({ type: 'tool_result', tool_use_id: id, content: '{"t":1}' })) }
|
|
394
|
+
]
|
|
395
|
+
});
|
|
396
|
+
|
|
397
|
+
const body = irToVertexBody(history([a.id, b.id]), 'gemini-3-pro');
|
|
398
|
+
assert.equal(body.contents[1].parts[0].thoughtSignature, 'SIG_1');
|
|
399
|
+
assert.equal(body.contents[1].parts[1].thoughtSignature, undefined, 'only first parallel call carries a signature');
|
|
400
|
+
assert.equal(body.contents[2].role, 'user');
|
|
401
|
+
assert.equal(body.contents[2].parts.length, 2);
|
|
402
|
+
|
|
403
|
+
const unknown = irToVertexBody(history(['toolu_from_claude']), 'gemini-3-flash');
|
|
404
|
+
assert.equal(unknown.contents[1].parts[0].thoughtSignature, GEMINI_DUMMY_SIGNATURE);
|
|
405
|
+
const old = irToVertexBody(history(['toolu_from_claude']), 'gemini-2.5-pro');
|
|
406
|
+
assert.equal(old.contents[1].parts[0].thoughtSignature, undefined);
|
|
407
|
+
|
|
408
|
+
const past = anthropicToIR({
|
|
409
|
+
model: 'x', messages: [
|
|
410
|
+
{ role: 'user', content: 'first' },
|
|
411
|
+
{ role: 'assistant', content: [{ type: 'tool_use', id: 'old_call', name: 'w', input: {} }] },
|
|
412
|
+
{ role: 'user', content: [{ type: 'tool_result', tool_use_id: 'old_call', content: 'ok' }] },
|
|
413
|
+
{ role: 'assistant', content: 'answer' },
|
|
414
|
+
{ role: 'user', content: 'new question' }
|
|
415
|
+
]
|
|
416
|
+
});
|
|
417
|
+
assert.equal(irToVertexBody(past, 'gemini-3-pro').contents[1].parts[0].thoughtSignature, undefined, 'previous turns are not validated');
|
|
418
|
+
});
|
|
419
|
+
|
|
420
|
+
test('OpenAI-compatible Gemini: extra_content.google.thought_signature is captured and echoed back', () => {
|
|
421
|
+
const n = makeNormalizer('openai-chat');
|
|
422
|
+
n({ choices: [{ delta: { tool_calls: [{ index: 0, id: 'call_g', type: 'function', function: { name: 'f', arguments: '{}' }, extra_content: { google: { thought_signature: 'SIG_OAI' } } }] } }] });
|
|
423
|
+
const ir = anthropicToIR({
|
|
424
|
+
model: 'x', messages: [
|
|
425
|
+
{ role: 'user', content: 'go' },
|
|
426
|
+
{ role: 'assistant', content: [{ type: 'tool_use', id: 'call_g', name: 'f', input: {} }] },
|
|
427
|
+
{ role: 'user', content: [{ type: 'tool_result', tool_use_id: 'call_g', content: 'ok' }] }
|
|
428
|
+
]
|
|
429
|
+
});
|
|
430
|
+
const body = irToChatBody(ir, 'gemini-3-pro');
|
|
431
|
+
assert.equal(body.messages.find(m => m.tool_calls).tool_calls[0].extra_content.google.thought_signature, 'SIG_OAI');
|
|
432
|
+
const plain = irToChatBody(anthropicToIR({ model: 'x', messages: [{ role: 'user', content: 'go' }, { role: 'assistant', content: [{ type: 'tool_use', id: 'other', name: 'f', input: {} }] }, { role: 'user', content: [{ type: 'tool_result', tool_use_id: 'other', content: 'ok' }] }] }), 'gpt-4o');
|
|
433
|
+
assert.equal(plain.messages.find(m => m.tool_calls).tool_calls[0].extra_content, undefined);
|
|
434
|
+
});
|
|
435
|
+
|
|
436
|
+
test('toGeminiSchema: bare string shorthands become Schema objects (Vertex 400 fix)', () => {
|
|
437
|
+
assert.deepEqual(toGeminiSchema('object'), { type: 'object', properties: {} });
|
|
438
|
+
assert.deepEqual(toGeminiSchema('string'), { type: 'string' });
|
|
439
|
+
assert.deepEqual(
|
|
440
|
+
toGeminiSchema({ type: 'object', properties: { tags: { type: 'array', items: 'object' }, n: 'string' } }),
|
|
441
|
+
{ type: 'object', properties: { tags: { type: 'array', items: { type: 'object', properties: {} } }, n: { type: 'string' } } }
|
|
442
|
+
);
|
|
443
|
+
});
|
|
444
|
+
|
|
445
|
+
test('toGeminiSchema: local $refs are inlined, $defs dropped', () => {
|
|
446
|
+
const schema = {
|
|
447
|
+
type: 'object',
|
|
448
|
+
properties: { msg: { $ref: '#/$defs/Part', description: 'a part' } },
|
|
449
|
+
$defs: { Part: { type: 'object', properties: { text: { type: 'string' } } } }
|
|
450
|
+
};
|
|
451
|
+
assert.deepEqual(toGeminiSchema(schema), {
|
|
452
|
+
type: 'object',
|
|
453
|
+
properties: { msg: { type: 'object', properties: { text: { type: 'string' } }, description: 'a part' } }
|
|
454
|
+
});
|
|
455
|
+
assert.deepEqual(toGeminiSchema({ $ref: '#/$defs/Missing' }), {});
|
|
456
|
+
});
|
|
457
|
+
|
|
458
|
+
test('toGeminiSchema: anyOf-null unions collapse to nullable', () => {
|
|
459
|
+
assert.deepEqual(
|
|
460
|
+
toGeminiSchema({ anyOf: [{ type: 'array', items: { type: 'string' } }, { type: 'null' }] }),
|
|
461
|
+
{ type: 'array', items: { type: 'string' }, nullable: true }
|
|
462
|
+
);
|
|
463
|
+
const multi = toGeminiSchema({ anyOf: [{ type: 'string' }, { type: 'integer' }] });
|
|
464
|
+
assert.equal(multi.anyOf.length, 2);
|
|
465
|
+
});
|
|
466
|
+
|
|
467
|
+
test('toGeminiSchema: garbage required entries and extra keys are stripped', () => {
|
|
468
|
+
const out = toGeminiSchema({
|
|
469
|
+
type: 'object',
|
|
470
|
+
properties: { a: { type: 'string' } },
|
|
471
|
+
required: [{ type: 'a' }, 'b'],
|
|
472
|
+
additionalProperties: true
|
|
473
|
+
});
|
|
474
|
+
// No listed name survives, and an empty required list is dropped (audit BR-05).
|
|
475
|
+
assert.equal(out.required, undefined);
|
|
476
|
+
assert.equal(out.additionalProperties, undefined);
|
|
477
|
+
});
|
|
478
|
+
|
|
479
|
+
test('isAntigravityModel gates the Gemini-safe tool rewrite', () => {
|
|
480
|
+
assert.ok(isAntigravityModel('ag/gemini-3.8-flash'));
|
|
481
|
+
assert.ok(isAntigravityModel('antigravity/x'));
|
|
482
|
+
assert.ok(!isAntigravityModel('gpt-5-codex'));
|
|
483
|
+
});
|
|
484
|
+
|
|
485
|
+
test('createResponsesStream.error carries a mapped code (Codex retryable failures)', () => {
|
|
486
|
+
const events = [];
|
|
487
|
+
const s = createResponsesStream((e, d) => events.push({ event: e, data: d }), 'ag/mock');
|
|
488
|
+
s.start();
|
|
489
|
+
s.error('slow down', 'rate_limit_exceeded');
|
|
490
|
+
assert.deepEqual(events.slice(0, 2).map(e => e.event), ['response.created', 'response.in_progress']);
|
|
491
|
+
const failed = events.find(e => e.event === 'response.failed');
|
|
492
|
+
assert.equal(failed.data.response.status, 'failed');
|
|
493
|
+
assert.equal(failed.data.response.error.code, 'rate_limit_exceeded');
|
|
494
|
+
assert.match(failed.data.response.id, /^resp_/);
|
|
495
|
+
});
|
|
496
|
+
|
|
497
|
+
// ---------------------------------------------------------------------------
|
|
498
|
+
// Usage accounting. Measured against real upstreams on 2026-09-20.
|
|
499
|
+
// ---------------------------------------------------------------------------
|
|
500
|
+
|
|
501
|
+
// Reasoning tokens are the expensive half of a thinking model's output, and every
|
|
502
|
+
// upstream reports them in its own place. Measured live through 9Router:
|
|
503
|
+
// {"completion_tokens":96,"completion_tokens_details":{"reasoning_tokens":95}}
|
|
504
|
+
// 95 of 96 output tokens were reasoning. Dropping the field makes the client
|
|
505
|
+
// believe a reasoning model did no reasoning.
|
|
506
|
+
test('smartUsage reads reasoning tokens from every upstream shape', () => {
|
|
507
|
+
const openaiChat = smartUsage({
|
|
508
|
+
prompt_tokens: 2012, completion_tokens: 96, total_tokens: 2108,
|
|
509
|
+
completion_tokens_details: { reasoning_tokens: 95 }
|
|
510
|
+
});
|
|
511
|
+
assert.equal(openaiChat.reasoning, 95, 'OpenAI chat: completion_tokens_details');
|
|
512
|
+
|
|
513
|
+
const responses = smartUsage({
|
|
514
|
+
input_tokens: 10, output_tokens: 50,
|
|
515
|
+
output_tokens_details: { reasoning_tokens: 40 }
|
|
516
|
+
});
|
|
517
|
+
assert.equal(responses.reasoning, 40, 'OpenAI Responses: output_tokens_details');
|
|
518
|
+
|
|
519
|
+
const anthropic = smartUsage({
|
|
520
|
+
input_tokens: 10, output_tokens: 50,
|
|
521
|
+
output_tokens_details: { thinking_tokens: 33 }
|
|
522
|
+
});
|
|
523
|
+
assert.equal(anthropic.reasoning, 33, 'Anthropic: thinking_tokens');
|
|
524
|
+
|
|
525
|
+
const vertex = smartUsage({
|
|
526
|
+
promptTokenCount: 10, candidatesTokenCount: 50, thoughtsTokenCount: 21
|
|
527
|
+
});
|
|
528
|
+
assert.equal(vertex.reasoning, 21, 'Vertex: thoughtsTokenCount');
|
|
529
|
+
|
|
530
|
+
assert.equal(smartUsage({ prompt_tokens: 5, completion_tokens: 5 }).reasoning, 0,
|
|
531
|
+
'no reasoning reported means zero, not undefined');
|
|
532
|
+
});
|
|
533
|
+
|
|
534
|
+
// docs/LLM-RESPONSE-MATRIX.md:126 records qwen/zai on Cloudflare sending usage in
|
|
535
|
+
// EVERY chunk as a per-token delta (completion_tokens: 1 x N). Math.max over those
|
|
536
|
+
// yields 1 regardless of how long the answer was.
|
|
537
|
+
test('collector accumulates per-chunk usage deltas instead of taking the max', () => {
|
|
538
|
+
const c = createCollector();
|
|
539
|
+
for (let i = 0; i < 40; i++) {
|
|
540
|
+
c.add({ text: 'x', usage: { prompt: i === 0 ? 120 : 0, completion: 1, cached: 0 } });
|
|
541
|
+
}
|
|
542
|
+
assert.equal(c.completion(), 40,
|
|
543
|
+
'forty per-token deltas must total 40, not 1');
|
|
544
|
+
assert.equal(c.prompt, 120, 'the prompt total is reported once and must not be summed away');
|
|
545
|
+
});
|
|
546
|
+
|
|
547
|
+
// Anthropic and Vertex report a running total in every chunk. Summing those would
|
|
548
|
+
// multiply the count, so the collector must still take the max for that shape.
|
|
549
|
+
test('collector keeps taking the max for cumulative usage', () => {
|
|
550
|
+
const c = createCollector();
|
|
551
|
+
c.add({ text: 'a', usage: { prompt: 100, completion: 10, cached: 0 } });
|
|
552
|
+
c.add({ text: 'b', usage: { prompt: 100, completion: 25, cached: 0 } });
|
|
553
|
+
c.add({ text: 'c', usage: { prompt: 100, completion: 60, cached: 0 } });
|
|
554
|
+
assert.equal(c.completion(), 60, 'a cumulative series must report its last value');
|
|
555
|
+
assert.equal(c.prompt, 100);
|
|
556
|
+
});
|
|
557
|
+
|
|
558
|
+
test('emitters carry reasoning tokens through to the client', () => {
|
|
559
|
+
const respMsg = buildResponsesMessage({
|
|
560
|
+
model: 'm',
|
|
561
|
+
text: ['hello'],
|
|
562
|
+
prompt: 100,
|
|
563
|
+
completion: 50,
|
|
564
|
+
cached: 20,
|
|
565
|
+
reasoning: 35
|
|
566
|
+
});
|
|
567
|
+
assert.equal(respMsg.usage.output_tokens_details.reasoning_tokens, 35);
|
|
568
|
+
|
|
569
|
+
let streamResp;
|
|
570
|
+
const respStream = createResponsesStream((event, data) => {
|
|
571
|
+
if (event === 'response.completed') streamResp = data.response;
|
|
572
|
+
}, 'm');
|
|
573
|
+
respStream.start();
|
|
574
|
+
respStream.finish('stop', { prompt: 100, completion: 50, cached: 20, reasoning: 35 });
|
|
575
|
+
assert.equal(streamResp.usage.output_tokens_details.reasoning_tokens, 35);
|
|
576
|
+
|
|
577
|
+
const chatMsg = buildChatMessage({
|
|
578
|
+
model: 'm',
|
|
579
|
+
text: ['hello'],
|
|
580
|
+
prompt: 100,
|
|
581
|
+
completion: 50,
|
|
582
|
+
cached: 20,
|
|
583
|
+
reasoning: 35
|
|
584
|
+
});
|
|
585
|
+
assert.deepEqual(chatMsg.usage.completion_tokens_details, { reasoning_tokens: 35 });
|
|
586
|
+
|
|
587
|
+
const chatMsgZero = buildChatMessage({
|
|
588
|
+
model: 'm',
|
|
589
|
+
text: ['hello'],
|
|
590
|
+
prompt: 100,
|
|
591
|
+
completion: 50,
|
|
592
|
+
cached: 20,
|
|
593
|
+
reasoning: 0
|
|
594
|
+
});
|
|
595
|
+
assert.equal(chatMsgZero.usage.completion_tokens_details, undefined);
|
|
596
|
+
|
|
597
|
+
let chatStreamChunk;
|
|
598
|
+
const chatStream = createChatStream((err, chunk) => {
|
|
599
|
+
if (chunk?.usage) chatStreamChunk = chunk;
|
|
600
|
+
}, 'm');
|
|
601
|
+
chatStream.start();
|
|
602
|
+
chatStream.finish('stop', { prompt: 100, completion: 50, cached: 20, reasoning: 35 });
|
|
603
|
+
assert.deepEqual(chatStreamChunk.usage.completion_tokens_details, { reasoning_tokens: 35 });
|
|
604
|
+
|
|
605
|
+
let chatStreamZeroChunk;
|
|
606
|
+
const chatStreamZero = createChatStream((err, chunk) => {
|
|
607
|
+
if (chunk?.usage) chatStreamZeroChunk = chunk;
|
|
608
|
+
}, 'm');
|
|
609
|
+
chatStreamZero.start();
|
|
610
|
+
chatStreamZero.finish('stop', { prompt: 100, completion: 50, cached: 20, reasoning: 0 });
|
|
611
|
+
assert.equal(chatStreamZeroChunk.usage.completion_tokens_details, undefined);
|
|
612
|
+
});
|
|
613
|
+
|
|
614
|
+
|
|
615
|
+
// Four separate places rebuilt the usage object by listing its fields by hand, and
|
|
616
|
+
// each one silently dropped `reasoning` on 2026-09-20. This test walks the whole
|
|
617
|
+
// path — parse, collect, emit — so a fifth hand-written copy fails here instead of
|
|
618
|
+
// reaching a user as a zero.
|
|
619
|
+
test('reasoning tokens survive the full parse -> collect -> emit path', () => {
|
|
620
|
+
const norm = createUpstreamNormalizer('openai-chat');
|
|
621
|
+
const collector = createCollector();
|
|
622
|
+
const chunks = [
|
|
623
|
+
{ choices: [{ delta: { content: 'O' } }] },
|
|
624
|
+
{ choices: [{ delta: { content: 'K' } }], usage: {
|
|
625
|
+
prompt_tokens: 2012, completion_tokens: 96,
|
|
626
|
+
completion_tokens_details: { reasoning_tokens: 95 }
|
|
627
|
+
} }
|
|
628
|
+
];
|
|
629
|
+
for (const c of chunks) collector.add(norm(c));
|
|
630
|
+
|
|
631
|
+
assert.equal(collector.reasoning, 95, 'the collector must keep the reasoning count');
|
|
632
|
+
|
|
633
|
+
const built = buildResponsesMessage({
|
|
634
|
+
model: 'gpt-5.6-sol', think: [], text: ['OK'], tools: [], finish: 'stop',
|
|
635
|
+
prompt: collector.prompt, completion: collector.completion(),
|
|
636
|
+
cached: collector.cached, reasoning: collector.reasoning
|
|
637
|
+
});
|
|
638
|
+
assert.equal(built.usage.output_tokens_details.reasoning_tokens, 95,
|
|
639
|
+
'the Responses emitter must report what the upstream billed');
|
|
640
|
+
});
|
|
641
|
+
|
|
642
|
+
// A max_tokens stop can cut a tool call's arguments. Reporting it as a finished tool call makes
|
|
643
|
+
// the client run the tool with truncated JSON (9router backends do stop with `length`).
|
|
644
|
+
test('a length stop is never reported as a finished tool call', () => {
|
|
645
|
+
assert.equal(chatFinish('length', true), 'length');
|
|
646
|
+
assert.equal(chatFinish('stop', true), 'tool_calls');
|
|
647
|
+
assert.equal(anthropicStopReason('length', true), 'max_tokens');
|
|
648
|
+
assert.equal(anthropicStopReason('stop', true), 'tool_use');
|
|
649
|
+
|
|
650
|
+
const events = [];
|
|
651
|
+
const r = createResponsesStream((event, data) => events.push({ event, data }), 'm');
|
|
652
|
+
r.start();
|
|
653
|
+
r.tool({ index: 0, id: 'call_cut', name: 'write_file', args: '{"file_path":"/x","content":"par' });
|
|
654
|
+
r.finish('length', { prompt: 10, completion: 3, hasTools: true });
|
|
655
|
+
const done = events.filter(e => e.event === 'response.output_item.done').map(e => e.data.item);
|
|
656
|
+
assert.ok(!done.some(i => i.type === 'function_call'), 'a function_call with unparseable arguments is not emitted as done');
|
|
657
|
+
const completed = events.at(-1).data.response;
|
|
658
|
+
assert.equal(completed.status, 'incomplete');
|
|
659
|
+
assert.ok(!completed.output.some(i => i?.type === 'function_call'), 'nor listed in the final output');
|
|
660
|
+
|
|
661
|
+
const ok = [];
|
|
662
|
+
const r2 = createResponsesStream((event, data) => ok.push({ event, data }), 'm');
|
|
663
|
+
r2.start();
|
|
664
|
+
r2.tool({ index: 0, id: 'call_ok', name: 'shell', args: '{"cmd":"ls"}' });
|
|
665
|
+
r2.finish('length', { hasTools: true });
|
|
666
|
+
assert.ok(ok.some(e => e.event === 'response.output_item.done' && e.data.item.type === 'function_call'), 'complete arguments still run');
|
|
667
|
+
|
|
668
|
+
// Every tool kind, not only function calls: apply_patch is a custom tool whose JSON arguments can be cut too.
|
|
669
|
+
const custom = [];
|
|
670
|
+
const r3 = createResponsesStream((event, data) => custom.push({ event, data }), 'm', { toolMeta: { apply_patch: { kind: 'custom', name: 'apply_patch' } } });
|
|
671
|
+
r3.start();
|
|
672
|
+
r3.tool({ index: 0, id: 'call_patch', name: 'apply_patch', args: '{"input":"*** Begin Patch\\n*** Upd' });
|
|
673
|
+
r3.tool({ index: 1, id: 'call_noargs', name: 'list_files', args: '' });
|
|
674
|
+
r3.finish('length', { hasTools: true });
|
|
675
|
+
const doneItems = custom.filter(e => e.event === 'response.output_item.done').map(e => e.data.item);
|
|
676
|
+
assert.ok(!doneItems.some(i => i.type === 'custom_tool_call'), 'a cut custom tool call is not emitted as done');
|
|
677
|
+
assert.ok(doneItems.some(i => i.type === 'function_call' && i.name === 'list_files'), 'a call with no arguments is complete, as in the non-stream builder');
|
|
678
|
+
|
|
679
|
+
const msg = buildResponsesMessage({ model: 'm', text: [], tools: [{ id: 'c1', name: 'write_file', args: '{"a":"tru' }], finish: 'length' });
|
|
680
|
+
assert.equal(msg.status, 'incomplete');
|
|
681
|
+
assert.ok(!msg.output.some(i => i.type === 'function_call'), 'the non-stream builder applies the same rule');
|
|
682
|
+
});
|
|
683
|
+
|
|
684
|
+
// ---- Format edge cases from the audit (G01-G04, T07, T08, BR-04..08, S6, null tool id) ----
|
|
685
|
+
|
|
686
|
+
const fn = (name) => ({ type: 'function', name, parameters: { type: 'object', properties: {} } });
|
|
687
|
+
|
|
688
|
+
test('Responses tool_choice: allowed_tools restricts the tools and keeps its mode; hosted and object modes are not forced', () => {
|
|
689
|
+
const base = { model: 'm', input: 'x', tools: [fn('a'), fn('b')] };
|
|
690
|
+
const allowed = responsesToIR({ ...base, tool_choice: { type: 'allowed_tools', mode: 'auto', tools: [{ type: 'function', name: 'a' }] } });
|
|
691
|
+
assert.deepEqual(allowed.tools.map(t => t.name), ['a']);
|
|
692
|
+
assert.equal(allowed.toolChoice, 'auto');
|
|
693
|
+
const required = responsesToIR({ ...base, tool_choice: { type: 'allowed_tools', mode: 'required', tools: [{ type: 'function', name: 'b' }] } });
|
|
694
|
+
assert.deepEqual(required.tools.map(t => t.name), ['b']);
|
|
695
|
+
assert.equal(required.toolChoice, 'required');
|
|
696
|
+
assert.equal(responsesToIR({ ...base, tool_choice: { type: 'web_search_preview' } }).toolChoice, 'auto');
|
|
697
|
+
assert.equal(responsesToIR({ ...base, tool_choice: { type: 'none' } }).toolChoice, 'none');
|
|
698
|
+
assert.equal(responsesToIR({ ...base, tool_choice: { type: 'auto' } }).toolChoice, 'auto');
|
|
699
|
+
assert.deepEqual(responsesToIR({ ...base, tool_choice: { type: 'function', name: 'b' } }).toolChoice, { name: 'b' });
|
|
700
|
+
});
|
|
701
|
+
|
|
702
|
+
test('Chat tool_choice: allowed_tools restricts the tools and keeps its mode', () => {
|
|
703
|
+
const tools = ['a', 'b'].map(name => ({ type: 'function', function: { name, parameters: { type: 'object', properties: {} } } }));
|
|
704
|
+
const ir = chatToIR({ model: 'm', messages: [{ role: 'user', content: 'x' }], tools,
|
|
705
|
+
tool_choice: { type: 'allowed_tools', allowed_tools: { mode: 'auto', tools: [{ type: 'function', function: { name: 'b' } }] } } });
|
|
706
|
+
assert.deepEqual(ir.tools.map(t => t.name), ['b']);
|
|
707
|
+
assert.equal(ir.toolChoice, 'auto');
|
|
708
|
+
});
|
|
709
|
+
|
|
710
|
+
test('smartText and smartReasoning return each string once', () => {
|
|
711
|
+
assert.equal(smartText({ content: ['line 1', 'line 2'] }), 'line 1\nline 2');
|
|
712
|
+
assert.equal(smartReasoning({ reasoning_content: 'step 1', parts: [{ text: 'step 1', thought: true }] }).text, 'step 1');
|
|
713
|
+
});
|
|
714
|
+
|
|
715
|
+
test('an empty stop string is dropped in every input format', () => {
|
|
716
|
+
const msgs = [{ role: 'user', content: 'x' }];
|
|
717
|
+
assert.deepEqual(chatToIR({ model: 'm', messages: msgs, stop: '' }).params.stop, []);
|
|
718
|
+
assert.deepEqual(chatToIR({ model: 'm', messages: msgs, stop: ['', 'END'] }).params.stop, ['END']);
|
|
719
|
+
assert.deepEqual(anthropicToIR({ model: 'm', max_tokens: 10, messages: msgs, stop_sequences: ['', 'END'] }).params.stop, ['END']);
|
|
720
|
+
assert.deepEqual(vertexToIR({ contents: [{ role: 'user', parts: [{ text: 'x' }] }], generationConfig: { stopSequences: [''] } }).params.stop, []);
|
|
721
|
+
const body = irToAnthropicBody(chatToIR({ model: 'm', messages: msgs, stop: '' }), 'claude-x');
|
|
722
|
+
assert.equal(body.stop_sequences, undefined);
|
|
723
|
+
});
|
|
724
|
+
|
|
725
|
+
test('a tool parameter named "properties" is a parameter, not a schema', () => {
|
|
726
|
+
const out = sanitizeJsonSchema({ type: 'object', properties: { parent: { type: 'string' }, properties: { type: 'string' } } });
|
|
727
|
+
assert.deepEqual(Object.keys(out.properties), ['parent', 'properties']);
|
|
728
|
+
});
|
|
729
|
+
|
|
730
|
+
test('Vertex usage: thoughts count as output, and the Vertex reply splits them back out', () => {
|
|
731
|
+
const u = smartUsage({ promptTokenCount: 5, candidatesTokenCount: 10, thoughtsTokenCount: 90 });
|
|
732
|
+
assert.equal(u.completion, 100);
|
|
733
|
+
assert.equal(u.reasoning, 90);
|
|
734
|
+
const msg = buildVertexMessage({ model: 'm', think: [], text: ['ok'], tools: [], finish: 'stop', prompt: 5, completion: 100, reasoning: 90, cached: 0 });
|
|
735
|
+
assert.deepEqual(msg.usageMetadata, { promptTokenCount: 5, candidatesTokenCount: 10, thoughtsTokenCount: 90, totalTokenCount: 105 });
|
|
736
|
+
});
|
|
737
|
+
|
|
738
|
+
test('reasoning model families get no <think> guide', () => {
|
|
739
|
+
const ir = chatToIR({ model: 'm', messages: [{ role: 'user', content: 'x' }], reasoning_effort: 'high' });
|
|
740
|
+
for (const model of ['o3-mini', 'openai/o1', 'o4-mini-high', 'deepseek-r1', 'deepseek-reasoner', 'qwq-32b']) {
|
|
741
|
+
const sys = irToChatBody(ir, model).messages.find(m => m.role === 'system')?.content || '';
|
|
742
|
+
assert.ok(!sys.includes('<think>'), `${model} must not get the guide`);
|
|
743
|
+
}
|
|
744
|
+
const plain = irToChatBody(ir, 'gpt-4o').messages.find(m => m.role === 'system')?.content || '';
|
|
745
|
+
assert.ok(plain.includes('<think>'), 'a model without native reasoning keeps the guide');
|
|
746
|
+
});
|
|
747
|
+
|
|
748
|
+
test('toGeminiSchema: no empty or misplaced required, const null counts as nullable, non-string enums survive as text', () => {
|
|
749
|
+
const noMatch = toGeminiSchema({ type: 'object', properties: { a: { type: 'string' } }, required: ['x'] });
|
|
750
|
+
assert.equal(Object.hasOwn(noMatch, 'required'), false);
|
|
751
|
+
assert.equal(Object.hasOwn(toGeminiSchema({ type: 'string', required: ['x'] }), 'required'), false);
|
|
752
|
+
assert.deepEqual(toGeminiSchema({ anyOf: [{ type: 'string' }, { const: null }] }), { type: 'string', nullable: true });
|
|
753
|
+
const num = toGeminiSchema({ type: 'integer', enum: [1, 2, 3], description: 'Level.' });
|
|
754
|
+
assert.equal(num.enum, undefined);
|
|
755
|
+
assert.equal(num.description, 'Level. Allowed values: 1, 2, 3.');
|
|
756
|
+
});
|
|
757
|
+
|
|
758
|
+
test('normalizeUpstream usage always carries reasoning', () => {
|
|
759
|
+
assert.equal(normalizeUpstream({ choices: [{ delta: { content: 'x' } }] }, 'openai-chat').usage.reasoning, 0);
|
|
760
|
+
assert.equal(normalizeUpstream({ type: 'content_block_delta', index: 0, delta: { type: 'text_delta', text: 'x' } }, 'anthropic').usage.reasoning, 0);
|
|
761
|
+
});
|
|
762
|
+
|
|
763
|
+
test('healToolPairs pairs a tool result with no id to the call with no id', () => {
|
|
764
|
+
const out = healToolPairs([
|
|
765
|
+
{ role: 'user', content: 'x' },
|
|
766
|
+
{ role: 'assistant', toolCalls: [{ id: null, name: 'f', args: '{}' }] },
|
|
767
|
+
{ role: 'tool', toolCallId: null, content: 'real result' }
|
|
768
|
+
]);
|
|
769
|
+
const call = out[1].toolCalls[0];
|
|
770
|
+
assert.ok(call.id);
|
|
771
|
+
assert.equal(out.length, 3);
|
|
772
|
+
assert.deepEqual({ role: out[2].role, toolCallId: out[2].toolCallId, content: out[2].content }, { role: 'tool', toolCallId: call.id, content: 'real result' });
|
|
773
|
+
});
|
|
774
|
+
|
|
775
|
+
test('healAnthropicPayload keeps an unchanged user turn as the same object', () => {
|
|
776
|
+
const user = { role: 'user', content: [{ type: 'tool_result', tool_use_id: 't1', content: 'ok' }] };
|
|
777
|
+
const payload = { messages: [{ role: 'user', content: 'go' }, { role: 'assistant', content: [{ type: 'tool_use', id: 't1', name: 'f', input: {} }] }, user] };
|
|
778
|
+
const healed = healAnthropicPayload(payload);
|
|
779
|
+
assert.equal(healed.changed, false);
|
|
780
|
+
assert.equal(healed.payload.messages[2], user);
|
|
781
|
+
});
|
|
782
|
+
|
|
783
|
+
test('allowed_tools that matches no declared tool gives a request without tools, not a crash', () => {
|
|
784
|
+
const chat = chatToIR({ model: 'm', messages: [{ role: 'user', content: 'x' }],
|
|
785
|
+
tools: [{ type: 'function', function: { name: 'a', parameters: { type: 'object', properties: {} } } }],
|
|
786
|
+
tool_choice: { type: 'allowed_tools', allowed_tools: { mode: 'auto', tools: [{ type: 'function', function: { name: 'b' } }] } } });
|
|
787
|
+
const hosted = responsesToIR({ model: 'm', input: 'x', tools: [fn('a')], tool_choice: { type: 'allowed_tools', mode: 'required', tools: [{ type: 'web_search' }] } });
|
|
788
|
+
for (const ir of [chat, hosted]) {
|
|
789
|
+
for (const body of [irToChatBody(ir, 'm'), irToAnthropicBody(ir, 'claude-x'), irToVertexBody(ir, 'gemini-x')]) {
|
|
790
|
+
assert.equal(body.tools, undefined);
|
|
791
|
+
assert.equal(body.tool_choice, undefined);
|
|
792
|
+
}
|
|
793
|
+
}
|
|
794
|
+
});
|