llm-switcher 1.2.0 → 1.2.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +12 -0
- package/classifier.mjs +238 -0
- package/docs/TOKEN-OPTIMIZER-INTEROP.md +110 -110
- package/docs/response-matrix.json +1130 -1130
- package/formats.mjs +30 -1
- package/icons/antigravity.png +0 -0
- package/icons/claude.png +0 -0
- package/icons/codex.png +0 -0
- package/icons/deepseek.png +0 -0
- package/icons/gemini.png +0 -0
- package/icons/github.png +0 -0
- package/icons/groq.png +0 -0
- package/icons/intact.svg +1 -0
- package/icons/ollama.png +0 -0
- package/icons/openai.png +0 -0
- package/icons/openrouter.png +0 -0
- package/icons/qwen.png +0 -0
- package/icons/vertex.png +0 -0
- package/package.json +1 -1
- package/proxy.mjs +27 -15
- package/skills/llm-switcher/SKILL.md +93 -93
- package/switch +0 -0
- package/tests/classifier.test.mjs +210 -0
- package/tests/formats.test.mjs +30 -0
- package/tests/helpers.mjs +24 -24
- package/tests/live-optimizer-interop.mjs +205 -205
- package/ui.html +1700 -1623
package/tests/helpers.mjs
CHANGED
|
@@ -1,24 +1,24 @@
|
|
|
1
|
-
import assert from 'node:assert/strict';
|
|
2
|
-
|
|
3
|
-
// Validate an Anthropic event sequence: indexes increase in start order, deltas only target an open block.
|
|
4
|
-
export function assertValidAnthropicEvents(events) {
|
|
5
|
-
const open = new Set();
|
|
6
|
-
const seen = new Set();
|
|
7
|
-
let expectedNext = 0;
|
|
8
|
-
for (const { event, data } of events) {
|
|
9
|
-
if (event === 'content_block_start') {
|
|
10
|
-
assert.equal(data.index, expectedNext, `block index must be sequential (got ${data.index}, want ${expectedNext})`);
|
|
11
|
-
assert.ok(!seen.has(data.index), `index ${data.index} reused`);
|
|
12
|
-
seen.add(data.index);
|
|
13
|
-
open.add(data.index);
|
|
14
|
-
expectedNext++;
|
|
15
|
-
} else if (event === 'content_block_delta') {
|
|
16
|
-
assert.ok(open.has(data.index), `delta for block ${data.index} which is not open`);
|
|
17
|
-
} else if (event === 'content_block_stop') {
|
|
18
|
-
assert.ok(open.has(data.index), `stop for block ${data.index} which is not open`);
|
|
19
|
-
open.delete(data.index);
|
|
20
|
-
} else if (event === 'message_stop') {
|
|
21
|
-
assert.equal(open.size, 0, 'all blocks must be closed before message_stop');
|
|
22
|
-
}
|
|
23
|
-
}
|
|
24
|
-
}
|
|
1
|
+
import assert from 'node:assert/strict';
|
|
2
|
+
|
|
3
|
+
// Validate an Anthropic event sequence: indexes increase in start order, deltas only target an open block.
|
|
4
|
+
export function assertValidAnthropicEvents(events) {
|
|
5
|
+
const open = new Set();
|
|
6
|
+
const seen = new Set();
|
|
7
|
+
let expectedNext = 0;
|
|
8
|
+
for (const { event, data } of events) {
|
|
9
|
+
if (event === 'content_block_start') {
|
|
10
|
+
assert.equal(data.index, expectedNext, `block index must be sequential (got ${data.index}, want ${expectedNext})`);
|
|
11
|
+
assert.ok(!seen.has(data.index), `index ${data.index} reused`);
|
|
12
|
+
seen.add(data.index);
|
|
13
|
+
open.add(data.index);
|
|
14
|
+
expectedNext++;
|
|
15
|
+
} else if (event === 'content_block_delta') {
|
|
16
|
+
assert.ok(open.has(data.index), `delta for block ${data.index} which is not open`);
|
|
17
|
+
} else if (event === 'content_block_stop') {
|
|
18
|
+
assert.ok(open.has(data.index), `stop for block ${data.index} which is not open`);
|
|
19
|
+
open.delete(data.index);
|
|
20
|
+
} else if (event === 'message_stop') {
|
|
21
|
+
assert.equal(open.size, 0, 'all blocks must be closed before message_stop');
|
|
22
|
+
}
|
|
23
|
+
}
|
|
24
|
+
}
|
|
@@ -1,205 +1,205 @@
|
|
|
1
|
-
// LIVE test: requires a running gateway (switch on) and a real upstream (costs tokens).
|
|
2
|
-
// Offline test, no network needed: npm test
|
|
3
|
-
const GATEWAY_PORT = Number(process.env.LLM_SWITCHER_PORT) || 3456;
|
|
4
|
-
const BASE_URL = `http://127.0.0.1:${GATEWAY_PORT}`;
|
|
5
|
-
|
|
6
|
-
console.log('================================================================');
|
|
7
|
-
console.log(' LLM SWITCHER — TOKEN OPTIMIZER INTEROPERABILITY TEST SUITE ');
|
|
8
|
-
console.log(' Simulating failure modes from Headroom, RTK, and Ponytail ');
|
|
9
|
-
console.log('================================================================\n');
|
|
10
|
-
|
|
11
|
-
let passed = 0;
|
|
12
|
-
let failed = 0;
|
|
13
|
-
|
|
14
|
-
async function runTest(name, fn) {
|
|
15
|
-
process.stdout.write(`[TEST] ${name}... `);
|
|
16
|
-
const t0 = Date.now();
|
|
17
|
-
try {
|
|
18
|
-
const detail = await fn();
|
|
19
|
-
console.log(`PASS (${Date.now() - t0}ms)`);
|
|
20
|
-
if (detail) console.log(` ↳ ${detail}`);
|
|
21
|
-
passed++;
|
|
22
|
-
} catch (err) {
|
|
23
|
-
console.log(`FAIL (${Date.now() - t0}ms)`);
|
|
24
|
-
console.log(` ↳ ERROR: ${err.message}`);
|
|
25
|
-
failed++;
|
|
26
|
-
}
|
|
27
|
-
}
|
|
28
|
-
|
|
29
|
-
// -----------------------------------------------------------------------------
|
|
30
|
-
// TEST 1: Headroom Failure Mode — Orphaned tool_result
|
|
31
|
-
// When Headroom prunes history to save tokens, it drops the assistant tool_use turn.
|
|
32
|
-
// Anthropic strictly throws: 400 invalid_request_error: 'tool_use_id does not correspond to any tool_use'
|
|
33
|
-
// LLM Switcher Healer Engine: Repairs orphaned result into context text -> 200 OK
|
|
34
|
-
// -----------------------------------------------------------------------------
|
|
35
|
-
await runTest('Headroom Simulation: Orphaned tool_result turn', async () => {
|
|
36
|
-
const payload = {
|
|
37
|
-
model: 'claude-opus-4-6',
|
|
38
|
-
max_tokens: 40,
|
|
39
|
-
messages: [
|
|
40
|
-
{ role: 'user', content: 'Context turn before pruning' },
|
|
41
|
-
{
|
|
42
|
-
role: 'user',
|
|
43
|
-
content: [
|
|
44
|
-
{ type: 'tool_result', tool_use_id: 'headroom_pruned_call_99', content: 'Database query result: 42 rows found' },
|
|
45
|
-
{ type: 'text', text: 'Reply: healed successfully' }
|
|
46
|
-
]
|
|
47
|
-
}
|
|
48
|
-
],
|
|
49
|
-
stream: false
|
|
50
|
-
};
|
|
51
|
-
|
|
52
|
-
const res = await fetch(`${BASE_URL}/v1/messages`, {
|
|
53
|
-
method: 'POST',
|
|
54
|
-
headers: { 'Content-Type': 'application/json', 'anthropic-version': '2023-06-01' },
|
|
55
|
-
body: JSON.stringify(payload)
|
|
56
|
-
});
|
|
57
|
-
|
|
58
|
-
if (res.status !== 200) {
|
|
59
|
-
const err = await res.text();
|
|
60
|
-
throw new Error(`Expected HTTP 200, got ${res.status}: ${err}`);
|
|
61
|
-
}
|
|
62
|
-
const json = await res.json();
|
|
63
|
-
const text = json.content?.map(c => c.text).join('') || '';
|
|
64
|
-
return `Healed orphaned tool_result. Response HTTP 200: "${text.slice(0, 50).replace(/\n/g, ' ')}..."`;
|
|
65
|
-
});
|
|
66
|
-
|
|
67
|
-
// -----------------------------------------------------------------------------
|
|
68
|
-
// TEST 2: Headroom Failure Mode — Consecutive User turns (Roles must alternate)
|
|
69
|
-
// When Headroom collapses history, multiple user turns occur back-to-back.
|
|
70
|
-
// Anthropic strictly throws: 400 invalid_request_error: 'roles must alternate'
|
|
71
|
-
// LLM Switcher Healer Engine: Merges consecutive same-role turns -> 200 OK
|
|
72
|
-
// -----------------------------------------------------------------------------
|
|
73
|
-
await runTest('Headroom Simulation: Consecutive User turns (Role alternation violation)', async () => {
|
|
74
|
-
const payload = {
|
|
75
|
-
model: 'claude-opus-4-6',
|
|
76
|
-
max_tokens: 30,
|
|
77
|
-
messages: [
|
|
78
|
-
{ role: 'user', content: 'User message turn 1' },
|
|
79
|
-
{ role: 'user', content: 'User message turn 2 (no assistant in between)' },
|
|
80
|
-
{ role: 'user', content: 'User message turn 3: reply pong' }
|
|
81
|
-
],
|
|
82
|
-
stream: false
|
|
83
|
-
};
|
|
84
|
-
|
|
85
|
-
const res = await fetch(`${BASE_URL}/v1/messages`, {
|
|
86
|
-
method: 'POST',
|
|
87
|
-
headers: { 'Content-Type': 'application/json', 'anthropic-version': '2023-06-01' },
|
|
88
|
-
body: JSON.stringify(payload)
|
|
89
|
-
});
|
|
90
|
-
|
|
91
|
-
if (res.status !== 200) {
|
|
92
|
-
const err = await res.text();
|
|
93
|
-
throw new Error(`Expected HTTP 200, got ${res.status}: ${err}`);
|
|
94
|
-
}
|
|
95
|
-
const json = await res.json();
|
|
96
|
-
return `Merged consecutive turns seamlessly. Stop reason: ${json.stop_reason}`;
|
|
97
|
-
});
|
|
98
|
-
|
|
99
|
-
// -----------------------------------------------------------------------------
|
|
100
|
-
// TEST 3: Headroom / RTK Failure Mode — Stripped Thinking Parameter
|
|
101
|
-
// Optimizer stripped the 'thinking' object to reduce tokens.
|
|
102
|
-
// LLM Switcher Thinking Guard: Detects reasoning model (ag/claude-opus-4-6-thinking)
|
|
103
|
-
// and restores thinking.budget_tokens automatically -> thinking_delta emitted!
|
|
104
|
-
// -----------------------------------------------------------------------------
|
|
105
|
-
await runTest('Thinking Guard: Restoring stripped thinking parameter on reasoning models', async () => {
|
|
106
|
-
const payload = {
|
|
107
|
-
model: 'claude-opus-4-6',
|
|
108
|
-
max_tokens: 200,
|
|
109
|
-
// Note: NO thinking parameter included (simulating optimizer stripping it)
|
|
110
|
-
messages: [
|
|
111
|
-
{ role: 'user', content: 'Solve step by step: what is 17 * 23? Show reasoning.' }
|
|
112
|
-
],
|
|
113
|
-
stream: true
|
|
114
|
-
};
|
|
115
|
-
|
|
116
|
-
const res = await fetch(`${BASE_URL}/v1/messages`, {
|
|
117
|
-
method: 'POST',
|
|
118
|
-
headers: { 'Content-Type': 'application/json', 'anthropic-version': '2023-06-01' },
|
|
119
|
-
body: JSON.stringify(payload)
|
|
120
|
-
});
|
|
121
|
-
|
|
122
|
-
if (res.status !== 200) {
|
|
123
|
-
throw new Error(`Expected HTTP 200, got ${res.status}`);
|
|
124
|
-
}
|
|
125
|
-
|
|
126
|
-
const text = await res.text();
|
|
127
|
-
const hasThinkingDelta = text.includes('thinking_delta');
|
|
128
|
-
const hasSignatureDelta = text.includes('signature_delta');
|
|
129
|
-
const hasTextDelta = text.includes('text_delta');
|
|
130
|
-
|
|
131
|
-
if (!hasThinkingDelta) {
|
|
132
|
-
throw new Error('Thinking Guard failed: thinking_delta was NOT emitted in stream!');
|
|
133
|
-
}
|
|
134
|
-
|
|
135
|
-
return `Automatically restored thinking: thinking_delta=${hasThinkingDelta}, signature_delta=${hasSignatureDelta}, text_delta=${hasTextDelta}`;
|
|
136
|
-
});
|
|
137
|
-
|
|
138
|
-
// -----------------------------------------------------------------------------
|
|
139
|
-
// TEST 4: RTK & Intermediary Headers Passthrough
|
|
140
|
-
// RTK / tracing tools inject headers: x-rtk-version, traceparent, x-request-id.
|
|
141
|
-
// LLM Switcher: Transparently preserves all tracking headers without rejection.
|
|
142
|
-
// -----------------------------------------------------------------------------
|
|
143
|
-
await runTest('RTK Intermediary: Custom headers and traceparent passthrough', async () => {
|
|
144
|
-
const payload = {
|
|
145
|
-
model: 'claude-opus-4-6',
|
|
146
|
-
max_tokens: 20,
|
|
147
|
-
messages: [{ role: 'user', content: 'ping' }],
|
|
148
|
-
stream: false
|
|
149
|
-
};
|
|
150
|
-
|
|
151
|
-
const res = await fetch(`${BASE_URL}/v1/messages`, {
|
|
152
|
-
method: 'POST',
|
|
153
|
-
headers: {
|
|
154
|
-
'Content-Type': 'application/json',
|
|
155
|
-
'anthropic-version': '2023-06-01',
|
|
156
|
-
'x-rtk-version': '0.37.2',
|
|
157
|
-
'x-optimizer-id': 'rtk-cli-hook',
|
|
158
|
-
'traceparent': '00-4bf92f3577b34da6a3ce929d0e0e4736-00f067aa0ba902b7-01'
|
|
159
|
-
},
|
|
160
|
-
body: JSON.stringify(payload)
|
|
161
|
-
});
|
|
162
|
-
|
|
163
|
-
if (res.status !== 200) {
|
|
164
|
-
throw new Error(`Expected HTTP 200, got ${res.status}`);
|
|
165
|
-
}
|
|
166
|
-
return `Headers accepted cleanly with HTTP 200 OK`;
|
|
167
|
-
});
|
|
168
|
-
|
|
169
|
-
// -----------------------------------------------------------------------------
|
|
170
|
-
// TEST 5: OpenAI Chat Healer Mode — Orphaned Tool Result in Chat Completions
|
|
171
|
-
// Same test but over /v1/chat/completions: tool role message without prior tool_calls assistant message.
|
|
172
|
-
// OpenAI API strictly throws: 400 'messages with role tool must be a response to a preceding message with tool_calls'
|
|
173
|
-
// LLM Switcher: Converts orphaned tool to user text message -> 200 OK
|
|
174
|
-
// -----------------------------------------------------------------------------
|
|
175
|
-
await runTest('OpenAI Chat Healer: Orphaned role tool without preceding assistant tool_calls', async () => {
|
|
176
|
-
const payload = {
|
|
177
|
-
model: 'ag/gpt-oss-120b-medium',
|
|
178
|
-
max_tokens: 30,
|
|
179
|
-
messages: [
|
|
180
|
-
{ role: 'user', content: 'Turn 1 context' },
|
|
181
|
-
{ role: 'tool', tool_call_id: 'orphaned_chat_call_77', content: 'Tool execution logs: OK' },
|
|
182
|
-
{ role: 'user', content: 'reply pong' }
|
|
183
|
-
],
|
|
184
|
-
stream: false
|
|
185
|
-
};
|
|
186
|
-
|
|
187
|
-
const res = await fetch(`${BASE_URL}/v1/chat/completions`, {
|
|
188
|
-
method: 'POST',
|
|
189
|
-
headers: { 'Content-Type': 'application/json' },
|
|
190
|
-
body: JSON.stringify(payload)
|
|
191
|
-
});
|
|
192
|
-
|
|
193
|
-
if (res.status !== 200) {
|
|
194
|
-
const err = await res.text();
|
|
195
|
-
throw new Error(`Expected HTTP 200, got ${res.status}: ${err}`);
|
|
196
|
-
}
|
|
197
|
-
const json = await res.json();
|
|
198
|
-
return `Chat Healer rescued orphaned tool role. Stop reason: ${json.choices?.[0]?.finish_reason}`;
|
|
199
|
-
});
|
|
200
|
-
|
|
201
|
-
console.log('\n================================================================');
|
|
202
|
-
console.log(` TEST RESULTS: ${passed} PASSED / ${failed} FAILED `);
|
|
203
|
-
console.log('================================================================\n');
|
|
204
|
-
|
|
205
|
-
process.exit(failed ? 1 : 0);
|
|
1
|
+
// LIVE test: requires a running gateway (switch on) and a real upstream (costs tokens).
|
|
2
|
+
// Offline test, no network needed: npm test
|
|
3
|
+
const GATEWAY_PORT = Number(process.env.LLM_SWITCHER_PORT) || 3456;
|
|
4
|
+
const BASE_URL = `http://127.0.0.1:${GATEWAY_PORT}`;
|
|
5
|
+
|
|
6
|
+
console.log('================================================================');
|
|
7
|
+
console.log(' LLM SWITCHER — TOKEN OPTIMIZER INTEROPERABILITY TEST SUITE ');
|
|
8
|
+
console.log(' Simulating failure modes from Headroom, RTK, and Ponytail ');
|
|
9
|
+
console.log('================================================================\n');
|
|
10
|
+
|
|
11
|
+
let passed = 0;
|
|
12
|
+
let failed = 0;
|
|
13
|
+
|
|
14
|
+
async function runTest(name, fn) {
|
|
15
|
+
process.stdout.write(`[TEST] ${name}... `);
|
|
16
|
+
const t0 = Date.now();
|
|
17
|
+
try {
|
|
18
|
+
const detail = await fn();
|
|
19
|
+
console.log(`PASS (${Date.now() - t0}ms)`);
|
|
20
|
+
if (detail) console.log(` ↳ ${detail}`);
|
|
21
|
+
passed++;
|
|
22
|
+
} catch (err) {
|
|
23
|
+
console.log(`FAIL (${Date.now() - t0}ms)`);
|
|
24
|
+
console.log(` ↳ ERROR: ${err.message}`);
|
|
25
|
+
failed++;
|
|
26
|
+
}
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
// -----------------------------------------------------------------------------
|
|
30
|
+
// TEST 1: Headroom Failure Mode — Orphaned tool_result
|
|
31
|
+
// When Headroom prunes history to save tokens, it drops the assistant tool_use turn.
|
|
32
|
+
// Anthropic strictly throws: 400 invalid_request_error: 'tool_use_id does not correspond to any tool_use'
|
|
33
|
+
// LLM Switcher Healer Engine: Repairs orphaned result into context text -> 200 OK
|
|
34
|
+
// -----------------------------------------------------------------------------
|
|
35
|
+
await runTest('Headroom Simulation: Orphaned tool_result turn', async () => {
|
|
36
|
+
const payload = {
|
|
37
|
+
model: 'claude-opus-4-6',
|
|
38
|
+
max_tokens: 40,
|
|
39
|
+
messages: [
|
|
40
|
+
{ role: 'user', content: 'Context turn before pruning' },
|
|
41
|
+
{
|
|
42
|
+
role: 'user',
|
|
43
|
+
content: [
|
|
44
|
+
{ type: 'tool_result', tool_use_id: 'headroom_pruned_call_99', content: 'Database query result: 42 rows found' },
|
|
45
|
+
{ type: 'text', text: 'Reply: healed successfully' }
|
|
46
|
+
]
|
|
47
|
+
}
|
|
48
|
+
],
|
|
49
|
+
stream: false
|
|
50
|
+
};
|
|
51
|
+
|
|
52
|
+
const res = await fetch(`${BASE_URL}/v1/messages`, {
|
|
53
|
+
method: 'POST',
|
|
54
|
+
headers: { 'Content-Type': 'application/json', 'anthropic-version': '2023-06-01' },
|
|
55
|
+
body: JSON.stringify(payload)
|
|
56
|
+
});
|
|
57
|
+
|
|
58
|
+
if (res.status !== 200) {
|
|
59
|
+
const err = await res.text();
|
|
60
|
+
throw new Error(`Expected HTTP 200, got ${res.status}: ${err}`);
|
|
61
|
+
}
|
|
62
|
+
const json = await res.json();
|
|
63
|
+
const text = json.content?.map(c => c.text).join('') || '';
|
|
64
|
+
return `Healed orphaned tool_result. Response HTTP 200: "${text.slice(0, 50).replace(/\n/g, ' ')}..."`;
|
|
65
|
+
});
|
|
66
|
+
|
|
67
|
+
// -----------------------------------------------------------------------------
|
|
68
|
+
// TEST 2: Headroom Failure Mode — Consecutive User turns (Roles must alternate)
|
|
69
|
+
// When Headroom collapses history, multiple user turns occur back-to-back.
|
|
70
|
+
// Anthropic strictly throws: 400 invalid_request_error: 'roles must alternate'
|
|
71
|
+
// LLM Switcher Healer Engine: Merges consecutive same-role turns -> 200 OK
|
|
72
|
+
// -----------------------------------------------------------------------------
|
|
73
|
+
await runTest('Headroom Simulation: Consecutive User turns (Role alternation violation)', async () => {
|
|
74
|
+
const payload = {
|
|
75
|
+
model: 'claude-opus-4-6',
|
|
76
|
+
max_tokens: 30,
|
|
77
|
+
messages: [
|
|
78
|
+
{ role: 'user', content: 'User message turn 1' },
|
|
79
|
+
{ role: 'user', content: 'User message turn 2 (no assistant in between)' },
|
|
80
|
+
{ role: 'user', content: 'User message turn 3: reply pong' }
|
|
81
|
+
],
|
|
82
|
+
stream: false
|
|
83
|
+
};
|
|
84
|
+
|
|
85
|
+
const res = await fetch(`${BASE_URL}/v1/messages`, {
|
|
86
|
+
method: 'POST',
|
|
87
|
+
headers: { 'Content-Type': 'application/json', 'anthropic-version': '2023-06-01' },
|
|
88
|
+
body: JSON.stringify(payload)
|
|
89
|
+
});
|
|
90
|
+
|
|
91
|
+
if (res.status !== 200) {
|
|
92
|
+
const err = await res.text();
|
|
93
|
+
throw new Error(`Expected HTTP 200, got ${res.status}: ${err}`);
|
|
94
|
+
}
|
|
95
|
+
const json = await res.json();
|
|
96
|
+
return `Merged consecutive turns seamlessly. Stop reason: ${json.stop_reason}`;
|
|
97
|
+
});
|
|
98
|
+
|
|
99
|
+
// -----------------------------------------------------------------------------
|
|
100
|
+
// TEST 3: Headroom / RTK Failure Mode — Stripped Thinking Parameter
|
|
101
|
+
// Optimizer stripped the 'thinking' object to reduce tokens.
|
|
102
|
+
// LLM Switcher Thinking Guard: Detects reasoning model (ag/claude-opus-4-6-thinking)
|
|
103
|
+
// and restores thinking.budget_tokens automatically -> thinking_delta emitted!
|
|
104
|
+
// -----------------------------------------------------------------------------
|
|
105
|
+
await runTest('Thinking Guard: Restoring stripped thinking parameter on reasoning models', async () => {
|
|
106
|
+
const payload = {
|
|
107
|
+
model: 'claude-opus-4-6',
|
|
108
|
+
max_tokens: 200,
|
|
109
|
+
// Note: NO thinking parameter included (simulating optimizer stripping it)
|
|
110
|
+
messages: [
|
|
111
|
+
{ role: 'user', content: 'Solve step by step: what is 17 * 23? Show reasoning.' }
|
|
112
|
+
],
|
|
113
|
+
stream: true
|
|
114
|
+
};
|
|
115
|
+
|
|
116
|
+
const res = await fetch(`${BASE_URL}/v1/messages`, {
|
|
117
|
+
method: 'POST',
|
|
118
|
+
headers: { 'Content-Type': 'application/json', 'anthropic-version': '2023-06-01' },
|
|
119
|
+
body: JSON.stringify(payload)
|
|
120
|
+
});
|
|
121
|
+
|
|
122
|
+
if (res.status !== 200) {
|
|
123
|
+
throw new Error(`Expected HTTP 200, got ${res.status}`);
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
const text = await res.text();
|
|
127
|
+
const hasThinkingDelta = text.includes('thinking_delta');
|
|
128
|
+
const hasSignatureDelta = text.includes('signature_delta');
|
|
129
|
+
const hasTextDelta = text.includes('text_delta');
|
|
130
|
+
|
|
131
|
+
if (!hasThinkingDelta) {
|
|
132
|
+
throw new Error('Thinking Guard failed: thinking_delta was NOT emitted in stream!');
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
return `Automatically restored thinking: thinking_delta=${hasThinkingDelta}, signature_delta=${hasSignatureDelta}, text_delta=${hasTextDelta}`;
|
|
136
|
+
});
|
|
137
|
+
|
|
138
|
+
// -----------------------------------------------------------------------------
|
|
139
|
+
// TEST 4: RTK & Intermediary Headers Passthrough
|
|
140
|
+
// RTK / tracing tools inject headers: x-rtk-version, traceparent, x-request-id.
|
|
141
|
+
// LLM Switcher: Transparently preserves all tracking headers without rejection.
|
|
142
|
+
// -----------------------------------------------------------------------------
|
|
143
|
+
await runTest('RTK Intermediary: Custom headers and traceparent passthrough', async () => {
|
|
144
|
+
const payload = {
|
|
145
|
+
model: 'claude-opus-4-6',
|
|
146
|
+
max_tokens: 20,
|
|
147
|
+
messages: [{ role: 'user', content: 'ping' }],
|
|
148
|
+
stream: false
|
|
149
|
+
};
|
|
150
|
+
|
|
151
|
+
const res = await fetch(`${BASE_URL}/v1/messages`, {
|
|
152
|
+
method: 'POST',
|
|
153
|
+
headers: {
|
|
154
|
+
'Content-Type': 'application/json',
|
|
155
|
+
'anthropic-version': '2023-06-01',
|
|
156
|
+
'x-rtk-version': '0.37.2',
|
|
157
|
+
'x-optimizer-id': 'rtk-cli-hook',
|
|
158
|
+
'traceparent': '00-4bf92f3577b34da6a3ce929d0e0e4736-00f067aa0ba902b7-01'
|
|
159
|
+
},
|
|
160
|
+
body: JSON.stringify(payload)
|
|
161
|
+
});
|
|
162
|
+
|
|
163
|
+
if (res.status !== 200) {
|
|
164
|
+
throw new Error(`Expected HTTP 200, got ${res.status}`);
|
|
165
|
+
}
|
|
166
|
+
return `Headers accepted cleanly with HTTP 200 OK`;
|
|
167
|
+
});
|
|
168
|
+
|
|
169
|
+
// -----------------------------------------------------------------------------
|
|
170
|
+
// TEST 5: OpenAI Chat Healer Mode — Orphaned Tool Result in Chat Completions
|
|
171
|
+
// Same test but over /v1/chat/completions: tool role message without prior tool_calls assistant message.
|
|
172
|
+
// OpenAI API strictly throws: 400 'messages with role tool must be a response to a preceding message with tool_calls'
|
|
173
|
+
// LLM Switcher: Converts orphaned tool to user text message -> 200 OK
|
|
174
|
+
// -----------------------------------------------------------------------------
|
|
175
|
+
await runTest('OpenAI Chat Healer: Orphaned role tool without preceding assistant tool_calls', async () => {
|
|
176
|
+
const payload = {
|
|
177
|
+
model: 'ag/gpt-oss-120b-medium',
|
|
178
|
+
max_tokens: 30,
|
|
179
|
+
messages: [
|
|
180
|
+
{ role: 'user', content: 'Turn 1 context' },
|
|
181
|
+
{ role: 'tool', tool_call_id: 'orphaned_chat_call_77', content: 'Tool execution logs: OK' },
|
|
182
|
+
{ role: 'user', content: 'reply pong' }
|
|
183
|
+
],
|
|
184
|
+
stream: false
|
|
185
|
+
};
|
|
186
|
+
|
|
187
|
+
const res = await fetch(`${BASE_URL}/v1/chat/completions`, {
|
|
188
|
+
method: 'POST',
|
|
189
|
+
headers: { 'Content-Type': 'application/json' },
|
|
190
|
+
body: JSON.stringify(payload)
|
|
191
|
+
});
|
|
192
|
+
|
|
193
|
+
if (res.status !== 200) {
|
|
194
|
+
const err = await res.text();
|
|
195
|
+
throw new Error(`Expected HTTP 200, got ${res.status}: ${err}`);
|
|
196
|
+
}
|
|
197
|
+
const json = await res.json();
|
|
198
|
+
return `Chat Healer rescued orphaned tool role. Stop reason: ${json.choices?.[0]?.finish_reason}`;
|
|
199
|
+
});
|
|
200
|
+
|
|
201
|
+
console.log('\n================================================================');
|
|
202
|
+
console.log(` TEST RESULTS: ${passed} PASSED / ${failed} FAILED `);
|
|
203
|
+
console.log('================================================================\n');
|
|
204
|
+
|
|
205
|
+
process.exit(failed ? 1 : 0);
|