llm-switcher 1.2.0 → 1.2.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +23 -0
- package/README.md +1 -0
- package/blindfold/make-certs.sh +3 -1
- package/catalog.mjs +132 -42
- package/classifier.mjs +238 -0
- package/docs/TOKEN-OPTIMIZER-INTEROP.md +110 -110
- package/docs/response-matrix.json +1130 -1130
- package/formats.mjs +30 -1
- package/icons/antigravity.png +0 -0
- package/icons/claude.png +0 -0
- package/icons/codex.png +0 -0
- package/icons/deepseek.png +0 -0
- package/icons/gemini.png +0 -0
- package/icons/github.png +0 -0
- package/icons/groq.png +0 -0
- package/icons/intact.svg +1 -0
- package/icons/ollama.png +0 -0
- package/icons/openai.png +0 -0
- package/icons/openrouter.png +0 -0
- package/icons/qwen.png +0 -0
- package/icons/vertex.png +0 -0
- package/mcp.mjs +2 -2
- package/package.json +1 -1
- package/proxy.mjs +34 -20
- package/scripts/run-tests.mjs +3 -1
- package/skills/llm-switcher/SKILL.md +93 -93
- package/state.mjs +58 -4
- package/switch +0 -0
- package/switch.mjs +28 -4
- package/tests/blindfold-e2e.test.mjs +2 -2
- package/tests/blindfold-task5.test.mjs +1 -1
- package/tests/blindfold.task3.test.mjs +2 -2
- package/tests/blindfold.wire.test.mjs +3 -1
- package/tests/catalog.test.mjs +117 -19
- package/tests/classifier.test.mjs +210 -0
- package/tests/codex-daemon.test.mjs +92 -0
- package/tests/formats.test.mjs +30 -0
- package/tests/helpers.mjs +24 -24
- package/tests/lifecycle.test.mjs +49 -37
- package/tests/live-optimizer-interop.mjs +205 -205
- package/tests/make-certs.test.mjs +22 -0
- package/tests/real-user-sim.test.mjs +3 -0
- package/tests/shim.test.mjs +8 -3
- package/tests/switch.test.mjs +30 -24
- package/tests/ui.test.mjs +90 -0
- package/tests/version.test.mjs +78 -0
- package/ui.html +1895 -1616
- package/version.mjs +53 -0
package/tests/lifecycle.test.mjs
CHANGED
|
@@ -54,16 +54,16 @@ function envFor(ws) {
|
|
|
54
54
|
return { ...process.env, HOME: path.join(ws.dir, 'home'), USERPROFILE: path.join(ws.dir, 'home'), LLM_SWITCHER_CONFIG: ws.cfgPath, LLM_SWITCHER_STATE_DIR: ws.dir, LLM_SWITCHER_BLINDFOLD_CERTS: ws.certDir, CLAUDE_CONFIG_DIR: ws.claudeDir, LLM_SWITCHER_PORT: '', PORT: '' };
|
|
55
55
|
}
|
|
56
56
|
|
|
57
|
-
|
|
57
|
+
// The 1.2 schema: one pointer per tool and one interceptor port for the whole config. A file in the
|
|
58
|
+
// older schema is migrated on load, which rewrites it and breaks every "changes nothing" assertion.
|
|
59
|
+
function writeConfig(ws, gwPort, bfPort, activeCodex = null) {
|
|
58
60
|
fs.writeFileSync(ws.cfgPath, JSON.stringify({
|
|
59
61
|
port: gwPort,
|
|
60
|
-
|
|
62
|
+
blindfold: { port: bfPort },
|
|
63
|
+
activeProfiles: { claude: null, codex: activeCodex },
|
|
61
64
|
profiles: {
|
|
62
|
-
plain: { name: 'Plain', mode: 'convert',
|
|
63
|
-
bf: {
|
|
64
|
-
name: 'Blindfold', mode: 'convert', inFormat: 'responses', baseURL: 'http://127.0.0.1:9/v1', apiKey: 'k',
|
|
65
|
-
defaultModels: { main: 'm' }, blindfold: true, blindfoldPort: bfPort
|
|
66
|
-
}
|
|
65
|
+
plain: { name: 'Plain', mode: 'convert', tool: 'claude', baseURL: 'http://127.0.0.1:9/v1', apiKey: 'k', defaultModels: { opus: 'o' } },
|
|
66
|
+
bf: { name: 'Blindfold', mode: 'convert', tool: 'codex', baseURL: 'http://127.0.0.1:9/v1', apiKey: 'k', defaultModels: { main: 'm' } }
|
|
67
67
|
}
|
|
68
68
|
}, null, 2), { mode: 0o600 });
|
|
69
69
|
}
|
|
@@ -121,7 +121,7 @@ test('the gateway proves its identity to a fresh challenge; a replayed /health i
|
|
|
121
121
|
const body = await (await fetch(`http://127.0.0.1:${gwPort}/health?challenge=abc`)).json();
|
|
122
122
|
// The proof binds the role, the port it answers on and its pid, so it cannot be relayed or edited.
|
|
123
123
|
assert.equal(body.port, gwPort);
|
|
124
|
-
assert.equal(body.proof, crypto.createHmac('sha256', token(ws)).update(['gateway', gwPort, body.pid, '', '', '
|
|
124
|
+
assert.equal(body.proof, crypto.createHmac('sha256', token(ws)).update(['gateway', gwPort, body.pid, '', '', 'abc'].join('|')).digest('hex'));
|
|
125
125
|
assert.equal(await probe(ws, `s.probeGateway(${gwPort})`), 'ours');
|
|
126
126
|
assert.equal(await probe(ws, `s.probeGateway(${replayPort})`), 'foreign');
|
|
127
127
|
assert.equal(await probe(ws, `s.probeGateway(${await freePort()})`), 'free');
|
|
@@ -210,18 +210,21 @@ test('the gateway owns the interceptor: dashboard changes, a lost interceptor an
|
|
|
210
210
|
let gw = await startGateway(ws, gwPort);
|
|
211
211
|
const bfState = () => probe(ws, `s.probeBlindfold(${bfPort})`);
|
|
212
212
|
try {
|
|
213
|
-
// 1. the dashboard turns
|
|
214
|
-
const on = await api(ws, gwPort, '/api/switch', { target: '
|
|
213
|
+
// 1. the dashboard turns Codex on
|
|
214
|
+
const on = await api(ws, gwPort, '/api/switch', { target: 'codex', profile: 'bf' });
|
|
215
215
|
assert.equal(on.success, true, JSON.stringify(on));
|
|
216
216
|
let st = await waitFor(async () => { const s = await bfState(); return s.state === 'ours' && s; });
|
|
217
217
|
assert.equal(st.gatewayPort, gwPort);
|
|
218
|
-
assert.equal(st.
|
|
218
|
+
assert.equal(st.activeTools, 'codex');
|
|
219
219
|
|
|
220
|
-
// 2. the dashboard
|
|
221
|
-
const
|
|
222
|
-
assert.notEqual(
|
|
223
|
-
st = await waitFor(async () => { const s = await bfState(); return s.
|
|
224
|
-
assert.equal(st.
|
|
220
|
+
// 2. the dashboard turns Claude on as well: the running interceptor takes the new tool set in place
|
|
221
|
+
const both = await api(ws, gwPort, '/api/switch', { target: 'claude', profile: 'plain' });
|
|
222
|
+
assert.notEqual(both.success, false, JSON.stringify(both));
|
|
223
|
+
st = await waitFor(async () => { const s = await bfState(); return s.activeTools === 'claude,codex' && s; });
|
|
224
|
+
assert.equal(st.activeTools, 'claude,codex');
|
|
225
|
+
await api(ws, gwPort, '/api/switch', { target: 'claude', profile: null });
|
|
226
|
+
st = await waitFor(async () => { const s = await bfState(); return s.activeTools === 'codex' && s; });
|
|
227
|
+
assert.equal(st.activeTools, 'codex');
|
|
225
228
|
|
|
226
229
|
// 3. the interceptor dies; a sync brings it back
|
|
227
230
|
process.kill(st.pid, 'SIGTERM');
|
|
@@ -238,31 +241,30 @@ test('the gateway owns the interceptor: dashboard changes, a lost interceptor an
|
|
|
238
241
|
st = await waitFor(async () => { const s = await bfState(); return s.state === 'ours' && s; });
|
|
239
242
|
assert.equal(st.gatewayPort, gwPort, 'the gateway starts the interceptor at boot');
|
|
240
243
|
|
|
241
|
-
// 5.
|
|
244
|
+
// 5. config.json names an interceptor port that a foreign process holds: a dashboard change fails
|
|
242
245
|
const squatter = net.createServer(s => s.end());
|
|
243
246
|
const squatPort = await new Promise(r => squatter.listen(0, '127.0.0.1', () => r(squatter.address().port)));
|
|
247
|
+
const good = fs.readFileSync(ws.cfgPath);
|
|
244
248
|
try {
|
|
249
|
+
fs.writeFileSync(ws.cfgPath, JSON.stringify({ ...JSON.parse(good), blindfold: { port: squatPort } }, null, 2), { mode: 0o600 });
|
|
250
|
+
const edited = fs.readFileSync(ws.cfgPath);
|
|
245
251
|
const badRes = await fetch(`http://127.0.0.1:${gwPort}/api/save-profile`, {
|
|
246
252
|
method: 'POST', headers: { 'Content-Type': 'application/json', 'x-llm-switcher-token': token(ws) },
|
|
247
|
-
body: JSON.stringify({ key: 'bf', profile: {
|
|
253
|
+
body: JSON.stringify({ key: 'bf', profile: { name: 'Renamed' } })
|
|
248
254
|
});
|
|
249
255
|
assert.equal(badRes.status, 502, 'a failure is not answered with 200');
|
|
250
256
|
const bad = await badRes.json();
|
|
251
257
|
assert.equal(bad.success, false);
|
|
252
258
|
assert.match(bad.error, /held by another process/);
|
|
253
|
-
|
|
254
|
-
assert.equal(onDisk, bfPort, 'a refused change is not saved, so HTTPS_PROXY never points at the squatter');
|
|
259
|
+
assert.deepEqual(fs.readFileSync(ws.cfgPath), edited, 'a refused change is not saved');
|
|
255
260
|
assert.equal((await bfState()).state, 'ours', 'the working interceptor stays up');
|
|
256
|
-
const status = await fetch(`http://127.0.0.1:${gwPort}/api/status`, { headers: { 'x-llm-switcher-token': token(ws) } }).then(r => r.json());
|
|
257
|
-
assert.equal(status.config.profiles.bf.blindfoldPort, bfPort, 'the gateway does not keep the refused change in memory');
|
|
258
261
|
} finally {
|
|
259
262
|
squatter.close();
|
|
263
|
+
fs.writeFileSync(ws.cfgPath, good, { mode: 0o600 });
|
|
260
264
|
}
|
|
261
265
|
|
|
262
266
|
// 6. turning the Codex target off stops the interceptor
|
|
263
|
-
await api(ws, gwPort, '/api/
|
|
264
|
-
await waitFor(async () => (await bfState()).state === 'ours');
|
|
265
|
-
await api(ws, gwPort, '/api/switch', { target: 'responses', profile: null });
|
|
267
|
+
await api(ws, gwPort, '/api/switch', { target: 'codex', profile: null });
|
|
266
268
|
assert.equal(await waitFor(async () => (await bfState()).state === 'free'), true);
|
|
267
269
|
} finally {
|
|
268
270
|
try { const s = await bfState(); if (s.state === 'ours') process.kill(s.pid, 'SIGTERM'); } catch {}
|
|
@@ -276,32 +278,42 @@ test('the gateway owns the interceptor: dashboard changes, a lost interceptor an
|
|
|
276
278
|
test('concurrent admin changes: a refused change never reaches disk, and no accepted change is lost', { skip: !HAS_OPENSSL && 'posix + openssl' }, async () => {
|
|
277
279
|
const ws = makeWorkspace({ certs: true });
|
|
278
280
|
const gwPort = await freePort();
|
|
279
|
-
|
|
280
|
-
writeConfig(ws, gwPort, bfPort, 'bf');
|
|
281
|
-
const launch = snapshotLaunchFiles(ws.dir);
|
|
282
|
-
const gw = await startGateway(ws, gwPort);
|
|
281
|
+
let bfPort;
|
|
283
282
|
// A squatter that accepts and never answers keeps the identity probe waiting.
|
|
284
283
|
const silent = net.createServer(() => {});
|
|
285
284
|
const silentPort = await new Promise(r => silent.listen(0, '127.0.0.1', () => r(silent.address().port)));
|
|
285
|
+
writeConfig(ws, gwPort, silentPort);
|
|
286
|
+
const launch = snapshotLaunchFiles(ws.dir);
|
|
287
|
+
const gw = await startGateway(ws, gwPort);
|
|
286
288
|
try {
|
|
287
|
-
|
|
288
|
-
const post = (body) => fetch(`http://127.0.0.1:${gwPort}/api/save-profile`, {
|
|
289
|
+
const post = (p, body) => fetch(`http://127.0.0.1:${gwPort}${p}`, {
|
|
289
290
|
method: 'POST', headers: { 'Content-Type': 'application/json', 'x-llm-switcher-token': token(ws) }, body: JSON.stringify(body)
|
|
290
291
|
});
|
|
291
|
-
|
|
292
|
+
// Turning Codex on waits for the silent port and is refused; the rename that arrives meanwhile is kept.
|
|
293
|
+
const refused = post('/api/switch', { target: 'codex', profile: 'bf' });
|
|
292
294
|
await new Promise(r => setTimeout(r, 300));
|
|
293
|
-
const other = post({ key: 'plain', profile: { name: 'Renamed' } });
|
|
295
|
+
const other = post('/api/save-profile', { key: 'plain', profile: { name: 'Renamed' } });
|
|
294
296
|
const [a, b] = await Promise.all([refused, other]);
|
|
295
297
|
assert.equal(a.status, 502);
|
|
296
298
|
assert.equal(b.status, 200);
|
|
297
|
-
|
|
298
|
-
assert.equal(onDisk.
|
|
299
|
+
let onDisk = JSON.parse(fs.readFileSync(ws.cfgPath, 'utf8'));
|
|
300
|
+
assert.equal(onDisk.activeProfiles.codex, null, 'the refused switch is not saved by the other request');
|
|
299
301
|
assert.equal(onDisk.profiles.plain.name, 'Renamed');
|
|
300
302
|
|
|
303
|
+
// A usable interceptor port, and Codex on, for the second half. Chosen only now: other test files
|
|
304
|
+
// run at the same time, and a port picked seconds earlier can be taken in between.
|
|
305
|
+
bfPort = await freePort();
|
|
306
|
+
onDisk.blindfold = { port: bfPort };
|
|
307
|
+
fs.writeFileSync(ws.cfgPath, JSON.stringify(onDisk, null, 2), { mode: 0o600 });
|
|
308
|
+
const on = await post('/api/switch', { target: 'codex', profile: 'bf' });
|
|
309
|
+
assert.equal(on.status, 200, await on.text());
|
|
310
|
+
await waitFor(async () => (await probe(ws, `s.probeBlindfold(${bfPort})`)).state === 'ours');
|
|
311
|
+
const postProfile = (body) => post('/api/save-profile', body);
|
|
312
|
+
|
|
301
313
|
// Two accepted changes to the active profile: the one that saves last keeps the other.
|
|
302
314
|
const [c, d] = await Promise.all([
|
|
303
|
-
|
|
304
|
-
|
|
315
|
+
postProfile({ key: 'bf', profile: { name: 'First' } }),
|
|
316
|
+
postProfile({ key: 'bf', profile: { defaultModels: { main: 'm2' } } })
|
|
305
317
|
]);
|
|
306
318
|
assert.equal(c.status, 200);
|
|
307
319
|
assert.equal(d.status, 200);
|
|
@@ -1,205 +1,205 @@
|
|
|
1
|
-
// LIVE test: requires a running gateway (switch on) and a real upstream (costs tokens).
|
|
2
|
-
// Offline test, no network needed: npm test
|
|
3
|
-
const GATEWAY_PORT = Number(process.env.LLM_SWITCHER_PORT) || 3456;
|
|
4
|
-
const BASE_URL = `http://127.0.0.1:${GATEWAY_PORT}`;
|
|
5
|
-
|
|
6
|
-
console.log('================================================================');
|
|
7
|
-
console.log(' LLM SWITCHER — TOKEN OPTIMIZER INTEROPERABILITY TEST SUITE ');
|
|
8
|
-
console.log(' Simulating failure modes from Headroom, RTK, and Ponytail ');
|
|
9
|
-
console.log('================================================================\n');
|
|
10
|
-
|
|
11
|
-
let passed = 0;
|
|
12
|
-
let failed = 0;
|
|
13
|
-
|
|
14
|
-
async function runTest(name, fn) {
|
|
15
|
-
process.stdout.write(`[TEST] ${name}... `);
|
|
16
|
-
const t0 = Date.now();
|
|
17
|
-
try {
|
|
18
|
-
const detail = await fn();
|
|
19
|
-
console.log(`PASS (${Date.now() - t0}ms)`);
|
|
20
|
-
if (detail) console.log(` ↳ ${detail}`);
|
|
21
|
-
passed++;
|
|
22
|
-
} catch (err) {
|
|
23
|
-
console.log(`FAIL (${Date.now() - t0}ms)`);
|
|
24
|
-
console.log(` ↳ ERROR: ${err.message}`);
|
|
25
|
-
failed++;
|
|
26
|
-
}
|
|
27
|
-
}
|
|
28
|
-
|
|
29
|
-
// -----------------------------------------------------------------------------
|
|
30
|
-
// TEST 1: Headroom Failure Mode — Orphaned tool_result
|
|
31
|
-
// When Headroom prunes history to save tokens, it drops the assistant tool_use turn.
|
|
32
|
-
// Anthropic strictly throws: 400 invalid_request_error: 'tool_use_id does not correspond to any tool_use'
|
|
33
|
-
// LLM Switcher Healer Engine: Repairs orphaned result into context text -> 200 OK
|
|
34
|
-
// -----------------------------------------------------------------------------
|
|
35
|
-
await runTest('Headroom Simulation: Orphaned tool_result turn', async () => {
|
|
36
|
-
const payload = {
|
|
37
|
-
model: 'claude-opus-4-6',
|
|
38
|
-
max_tokens: 40,
|
|
39
|
-
messages: [
|
|
40
|
-
{ role: 'user', content: 'Context turn before pruning' },
|
|
41
|
-
{
|
|
42
|
-
role: 'user',
|
|
43
|
-
content: [
|
|
44
|
-
{ type: 'tool_result', tool_use_id: 'headroom_pruned_call_99', content: 'Database query result: 42 rows found' },
|
|
45
|
-
{ type: 'text', text: 'Reply: healed successfully' }
|
|
46
|
-
]
|
|
47
|
-
}
|
|
48
|
-
],
|
|
49
|
-
stream: false
|
|
50
|
-
};
|
|
51
|
-
|
|
52
|
-
const res = await fetch(`${BASE_URL}/v1/messages`, {
|
|
53
|
-
method: 'POST',
|
|
54
|
-
headers: { 'Content-Type': 'application/json', 'anthropic-version': '2023-06-01' },
|
|
55
|
-
body: JSON.stringify(payload)
|
|
56
|
-
});
|
|
57
|
-
|
|
58
|
-
if (res.status !== 200) {
|
|
59
|
-
const err = await res.text();
|
|
60
|
-
throw new Error(`Expected HTTP 200, got ${res.status}: ${err}`);
|
|
61
|
-
}
|
|
62
|
-
const json = await res.json();
|
|
63
|
-
const text = json.content?.map(c => c.text).join('') || '';
|
|
64
|
-
return `Healed orphaned tool_result. Response HTTP 200: "${text.slice(0, 50).replace(/\n/g, ' ')}..."`;
|
|
65
|
-
});
|
|
66
|
-
|
|
67
|
-
// -----------------------------------------------------------------------------
|
|
68
|
-
// TEST 2: Headroom Failure Mode — Consecutive User turns (Roles must alternate)
|
|
69
|
-
// When Headroom collapses history, multiple user turns occur back-to-back.
|
|
70
|
-
// Anthropic strictly throws: 400 invalid_request_error: 'roles must alternate'
|
|
71
|
-
// LLM Switcher Healer Engine: Merges consecutive same-role turns -> 200 OK
|
|
72
|
-
// -----------------------------------------------------------------------------
|
|
73
|
-
await runTest('Headroom Simulation: Consecutive User turns (Role alternation violation)', async () => {
|
|
74
|
-
const payload = {
|
|
75
|
-
model: 'claude-opus-4-6',
|
|
76
|
-
max_tokens: 30,
|
|
77
|
-
messages: [
|
|
78
|
-
{ role: 'user', content: 'User message turn 1' },
|
|
79
|
-
{ role: 'user', content: 'User message turn 2 (no assistant in between)' },
|
|
80
|
-
{ role: 'user', content: 'User message turn 3: reply pong' }
|
|
81
|
-
],
|
|
82
|
-
stream: false
|
|
83
|
-
};
|
|
84
|
-
|
|
85
|
-
const res = await fetch(`${BASE_URL}/v1/messages`, {
|
|
86
|
-
method: 'POST',
|
|
87
|
-
headers: { 'Content-Type': 'application/json', 'anthropic-version': '2023-06-01' },
|
|
88
|
-
body: JSON.stringify(payload)
|
|
89
|
-
});
|
|
90
|
-
|
|
91
|
-
if (res.status !== 200) {
|
|
92
|
-
const err = await res.text();
|
|
93
|
-
throw new Error(`Expected HTTP 200, got ${res.status}: ${err}`);
|
|
94
|
-
}
|
|
95
|
-
const json = await res.json();
|
|
96
|
-
return `Merged consecutive turns seamlessly. Stop reason: ${json.stop_reason}`;
|
|
97
|
-
});
|
|
98
|
-
|
|
99
|
-
// -----------------------------------------------------------------------------
|
|
100
|
-
// TEST 3: Headroom / RTK Failure Mode — Stripped Thinking Parameter
|
|
101
|
-
// Optimizer stripped the 'thinking' object to reduce tokens.
|
|
102
|
-
// LLM Switcher Thinking Guard: Detects reasoning model (ag/claude-opus-4-6-thinking)
|
|
103
|
-
// and restores thinking.budget_tokens automatically -> thinking_delta emitted!
|
|
104
|
-
// -----------------------------------------------------------------------------
|
|
105
|
-
await runTest('Thinking Guard: Restoring stripped thinking parameter on reasoning models', async () => {
|
|
106
|
-
const payload = {
|
|
107
|
-
model: 'claude-opus-4-6',
|
|
108
|
-
max_tokens: 200,
|
|
109
|
-
// Note: NO thinking parameter included (simulating optimizer stripping it)
|
|
110
|
-
messages: [
|
|
111
|
-
{ role: 'user', content: 'Solve step by step: what is 17 * 23? Show reasoning.' }
|
|
112
|
-
],
|
|
113
|
-
stream: true
|
|
114
|
-
};
|
|
115
|
-
|
|
116
|
-
const res = await fetch(`${BASE_URL}/v1/messages`, {
|
|
117
|
-
method: 'POST',
|
|
118
|
-
headers: { 'Content-Type': 'application/json', 'anthropic-version': '2023-06-01' },
|
|
119
|
-
body: JSON.stringify(payload)
|
|
120
|
-
});
|
|
121
|
-
|
|
122
|
-
if (res.status !== 200) {
|
|
123
|
-
throw new Error(`Expected HTTP 200, got ${res.status}`);
|
|
124
|
-
}
|
|
125
|
-
|
|
126
|
-
const text = await res.text();
|
|
127
|
-
const hasThinkingDelta = text.includes('thinking_delta');
|
|
128
|
-
const hasSignatureDelta = text.includes('signature_delta');
|
|
129
|
-
const hasTextDelta = text.includes('text_delta');
|
|
130
|
-
|
|
131
|
-
if (!hasThinkingDelta) {
|
|
132
|
-
throw new Error('Thinking Guard failed: thinking_delta was NOT emitted in stream!');
|
|
133
|
-
}
|
|
134
|
-
|
|
135
|
-
return `Automatically restored thinking: thinking_delta=${hasThinkingDelta}, signature_delta=${hasSignatureDelta}, text_delta=${hasTextDelta}`;
|
|
136
|
-
});
|
|
137
|
-
|
|
138
|
-
// -----------------------------------------------------------------------------
|
|
139
|
-
// TEST 4: RTK & Intermediary Headers Passthrough
|
|
140
|
-
// RTK / tracing tools inject headers: x-rtk-version, traceparent, x-request-id.
|
|
141
|
-
// LLM Switcher: Transparently preserves all tracking headers without rejection.
|
|
142
|
-
// -----------------------------------------------------------------------------
|
|
143
|
-
await runTest('RTK Intermediary: Custom headers and traceparent passthrough', async () => {
|
|
144
|
-
const payload = {
|
|
145
|
-
model: 'claude-opus-4-6',
|
|
146
|
-
max_tokens: 20,
|
|
147
|
-
messages: [{ role: 'user', content: 'ping' }],
|
|
148
|
-
stream: false
|
|
149
|
-
};
|
|
150
|
-
|
|
151
|
-
const res = await fetch(`${BASE_URL}/v1/messages`, {
|
|
152
|
-
method: 'POST',
|
|
153
|
-
headers: {
|
|
154
|
-
'Content-Type': 'application/json',
|
|
155
|
-
'anthropic-version': '2023-06-01',
|
|
156
|
-
'x-rtk-version': '0.37.2',
|
|
157
|
-
'x-optimizer-id': 'rtk-cli-hook',
|
|
158
|
-
'traceparent': '00-4bf92f3577b34da6a3ce929d0e0e4736-00f067aa0ba902b7-01'
|
|
159
|
-
},
|
|
160
|
-
body: JSON.stringify(payload)
|
|
161
|
-
});
|
|
162
|
-
|
|
163
|
-
if (res.status !== 200) {
|
|
164
|
-
throw new Error(`Expected HTTP 200, got ${res.status}`);
|
|
165
|
-
}
|
|
166
|
-
return `Headers accepted cleanly with HTTP 200 OK`;
|
|
167
|
-
});
|
|
168
|
-
|
|
169
|
-
// -----------------------------------------------------------------------------
|
|
170
|
-
// TEST 5: OpenAI Chat Healer Mode — Orphaned Tool Result in Chat Completions
|
|
171
|
-
// Same test but over /v1/chat/completions: tool role message without prior tool_calls assistant message.
|
|
172
|
-
// OpenAI API strictly throws: 400 'messages with role tool must be a response to a preceding message with tool_calls'
|
|
173
|
-
// LLM Switcher: Converts orphaned tool to user text message -> 200 OK
|
|
174
|
-
// -----------------------------------------------------------------------------
|
|
175
|
-
await runTest('OpenAI Chat Healer: Orphaned role tool without preceding assistant tool_calls', async () => {
|
|
176
|
-
const payload = {
|
|
177
|
-
model: 'ag/gpt-oss-120b-medium',
|
|
178
|
-
max_tokens: 30,
|
|
179
|
-
messages: [
|
|
180
|
-
{ role: 'user', content: 'Turn 1 context' },
|
|
181
|
-
{ role: 'tool', tool_call_id: 'orphaned_chat_call_77', content: 'Tool execution logs: OK' },
|
|
182
|
-
{ role: 'user', content: 'reply pong' }
|
|
183
|
-
],
|
|
184
|
-
stream: false
|
|
185
|
-
};
|
|
186
|
-
|
|
187
|
-
const res = await fetch(`${BASE_URL}/v1/chat/completions`, {
|
|
188
|
-
method: 'POST',
|
|
189
|
-
headers: { 'Content-Type': 'application/json' },
|
|
190
|
-
body: JSON.stringify(payload)
|
|
191
|
-
});
|
|
192
|
-
|
|
193
|
-
if (res.status !== 200) {
|
|
194
|
-
const err = await res.text();
|
|
195
|
-
throw new Error(`Expected HTTP 200, got ${res.status}: ${err}`);
|
|
196
|
-
}
|
|
197
|
-
const json = await res.json();
|
|
198
|
-
return `Chat Healer rescued orphaned tool role. Stop reason: ${json.choices?.[0]?.finish_reason}`;
|
|
199
|
-
});
|
|
200
|
-
|
|
201
|
-
console.log('\n================================================================');
|
|
202
|
-
console.log(` TEST RESULTS: ${passed} PASSED / ${failed} FAILED `);
|
|
203
|
-
console.log('================================================================\n');
|
|
204
|
-
|
|
205
|
-
process.exit(failed ? 1 : 0);
|
|
1
|
+
// LIVE test: requires a running gateway (switch on) and a real upstream (costs tokens).
|
|
2
|
+
// Offline test, no network needed: npm test
|
|
3
|
+
const GATEWAY_PORT = Number(process.env.LLM_SWITCHER_PORT) || 3456;
|
|
4
|
+
const BASE_URL = `http://127.0.0.1:${GATEWAY_PORT}`;
|
|
5
|
+
|
|
6
|
+
console.log('================================================================');
|
|
7
|
+
console.log(' LLM SWITCHER — TOKEN OPTIMIZER INTEROPERABILITY TEST SUITE ');
|
|
8
|
+
console.log(' Simulating failure modes from Headroom, RTK, and Ponytail ');
|
|
9
|
+
console.log('================================================================\n');
|
|
10
|
+
|
|
11
|
+
let passed = 0;
|
|
12
|
+
let failed = 0;
|
|
13
|
+
|
|
14
|
+
async function runTest(name, fn) {
|
|
15
|
+
process.stdout.write(`[TEST] ${name}... `);
|
|
16
|
+
const t0 = Date.now();
|
|
17
|
+
try {
|
|
18
|
+
const detail = await fn();
|
|
19
|
+
console.log(`PASS (${Date.now() - t0}ms)`);
|
|
20
|
+
if (detail) console.log(` ↳ ${detail}`);
|
|
21
|
+
passed++;
|
|
22
|
+
} catch (err) {
|
|
23
|
+
console.log(`FAIL (${Date.now() - t0}ms)`);
|
|
24
|
+
console.log(` ↳ ERROR: ${err.message}`);
|
|
25
|
+
failed++;
|
|
26
|
+
}
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
// -----------------------------------------------------------------------------
|
|
30
|
+
// TEST 1: Headroom Failure Mode — Orphaned tool_result
|
|
31
|
+
// When Headroom prunes history to save tokens, it drops the assistant tool_use turn.
|
|
32
|
+
// Anthropic strictly throws: 400 invalid_request_error: 'tool_use_id does not correspond to any tool_use'
|
|
33
|
+
// LLM Switcher Healer Engine: Repairs orphaned result into context text -> 200 OK
|
|
34
|
+
// -----------------------------------------------------------------------------
|
|
35
|
+
await runTest('Headroom Simulation: Orphaned tool_result turn', async () => {
|
|
36
|
+
const payload = {
|
|
37
|
+
model: 'claude-opus-4-6',
|
|
38
|
+
max_tokens: 40,
|
|
39
|
+
messages: [
|
|
40
|
+
{ role: 'user', content: 'Context turn before pruning' },
|
|
41
|
+
{
|
|
42
|
+
role: 'user',
|
|
43
|
+
content: [
|
|
44
|
+
{ type: 'tool_result', tool_use_id: 'headroom_pruned_call_99', content: 'Database query result: 42 rows found' },
|
|
45
|
+
{ type: 'text', text: 'Reply: healed successfully' }
|
|
46
|
+
]
|
|
47
|
+
}
|
|
48
|
+
],
|
|
49
|
+
stream: false
|
|
50
|
+
};
|
|
51
|
+
|
|
52
|
+
const res = await fetch(`${BASE_URL}/v1/messages`, {
|
|
53
|
+
method: 'POST',
|
|
54
|
+
headers: { 'Content-Type': 'application/json', 'anthropic-version': '2023-06-01' },
|
|
55
|
+
body: JSON.stringify(payload)
|
|
56
|
+
});
|
|
57
|
+
|
|
58
|
+
if (res.status !== 200) {
|
|
59
|
+
const err = await res.text();
|
|
60
|
+
throw new Error(`Expected HTTP 200, got ${res.status}: ${err}`);
|
|
61
|
+
}
|
|
62
|
+
const json = await res.json();
|
|
63
|
+
const text = json.content?.map(c => c.text).join('') || '';
|
|
64
|
+
return `Healed orphaned tool_result. Response HTTP 200: "${text.slice(0, 50).replace(/\n/g, ' ')}..."`;
|
|
65
|
+
});
|
|
66
|
+
|
|
67
|
+
// -----------------------------------------------------------------------------
|
|
68
|
+
// TEST 2: Headroom Failure Mode — Consecutive User turns (Roles must alternate)
|
|
69
|
+
// When Headroom collapses history, multiple user turns occur back-to-back.
|
|
70
|
+
// Anthropic strictly throws: 400 invalid_request_error: 'roles must alternate'
|
|
71
|
+
// LLM Switcher Healer Engine: Merges consecutive same-role turns -> 200 OK
|
|
72
|
+
// -----------------------------------------------------------------------------
|
|
73
|
+
await runTest('Headroom Simulation: Consecutive User turns (Role alternation violation)', async () => {
|
|
74
|
+
const payload = {
|
|
75
|
+
model: 'claude-opus-4-6',
|
|
76
|
+
max_tokens: 30,
|
|
77
|
+
messages: [
|
|
78
|
+
{ role: 'user', content: 'User message turn 1' },
|
|
79
|
+
{ role: 'user', content: 'User message turn 2 (no assistant in between)' },
|
|
80
|
+
{ role: 'user', content: 'User message turn 3: reply pong' }
|
|
81
|
+
],
|
|
82
|
+
stream: false
|
|
83
|
+
};
|
|
84
|
+
|
|
85
|
+
const res = await fetch(`${BASE_URL}/v1/messages`, {
|
|
86
|
+
method: 'POST',
|
|
87
|
+
headers: { 'Content-Type': 'application/json', 'anthropic-version': '2023-06-01' },
|
|
88
|
+
body: JSON.stringify(payload)
|
|
89
|
+
});
|
|
90
|
+
|
|
91
|
+
if (res.status !== 200) {
|
|
92
|
+
const err = await res.text();
|
|
93
|
+
throw new Error(`Expected HTTP 200, got ${res.status}: ${err}`);
|
|
94
|
+
}
|
|
95
|
+
const json = await res.json();
|
|
96
|
+
return `Merged consecutive turns seamlessly. Stop reason: ${json.stop_reason}`;
|
|
97
|
+
});
|
|
98
|
+
|
|
99
|
+
// -----------------------------------------------------------------------------
|
|
100
|
+
// TEST 3: Headroom / RTK Failure Mode — Stripped Thinking Parameter
|
|
101
|
+
// Optimizer stripped the 'thinking' object to reduce tokens.
|
|
102
|
+
// LLM Switcher Thinking Guard: Detects reasoning model (ag/claude-opus-4-6-thinking)
|
|
103
|
+
// and restores thinking.budget_tokens automatically -> thinking_delta emitted!
|
|
104
|
+
// -----------------------------------------------------------------------------
|
|
105
|
+
await runTest('Thinking Guard: Restoring stripped thinking parameter on reasoning models', async () => {
|
|
106
|
+
const payload = {
|
|
107
|
+
model: 'claude-opus-4-6',
|
|
108
|
+
max_tokens: 200,
|
|
109
|
+
// Note: NO thinking parameter included (simulating optimizer stripping it)
|
|
110
|
+
messages: [
|
|
111
|
+
{ role: 'user', content: 'Solve step by step: what is 17 * 23? Show reasoning.' }
|
|
112
|
+
],
|
|
113
|
+
stream: true
|
|
114
|
+
};
|
|
115
|
+
|
|
116
|
+
const res = await fetch(`${BASE_URL}/v1/messages`, {
|
|
117
|
+
method: 'POST',
|
|
118
|
+
headers: { 'Content-Type': 'application/json', 'anthropic-version': '2023-06-01' },
|
|
119
|
+
body: JSON.stringify(payload)
|
|
120
|
+
});
|
|
121
|
+
|
|
122
|
+
if (res.status !== 200) {
|
|
123
|
+
throw new Error(`Expected HTTP 200, got ${res.status}`);
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
const text = await res.text();
|
|
127
|
+
const hasThinkingDelta = text.includes('thinking_delta');
|
|
128
|
+
const hasSignatureDelta = text.includes('signature_delta');
|
|
129
|
+
const hasTextDelta = text.includes('text_delta');
|
|
130
|
+
|
|
131
|
+
if (!hasThinkingDelta) {
|
|
132
|
+
throw new Error('Thinking Guard failed: thinking_delta was NOT emitted in stream!');
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
return `Automatically restored thinking: thinking_delta=${hasThinkingDelta}, signature_delta=${hasSignatureDelta}, text_delta=${hasTextDelta}`;
|
|
136
|
+
});
|
|
137
|
+
|
|
138
|
+
// -----------------------------------------------------------------------------
|
|
139
|
+
// TEST 4: RTK & Intermediary Headers Passthrough
|
|
140
|
+
// RTK / tracing tools inject headers: x-rtk-version, traceparent, x-request-id.
|
|
141
|
+
// LLM Switcher: Transparently preserves all tracking headers without rejection.
|
|
142
|
+
// -----------------------------------------------------------------------------
|
|
143
|
+
await runTest('RTK Intermediary: Custom headers and traceparent passthrough', async () => {
|
|
144
|
+
const payload = {
|
|
145
|
+
model: 'claude-opus-4-6',
|
|
146
|
+
max_tokens: 20,
|
|
147
|
+
messages: [{ role: 'user', content: 'ping' }],
|
|
148
|
+
stream: false
|
|
149
|
+
};
|
|
150
|
+
|
|
151
|
+
const res = await fetch(`${BASE_URL}/v1/messages`, {
|
|
152
|
+
method: 'POST',
|
|
153
|
+
headers: {
|
|
154
|
+
'Content-Type': 'application/json',
|
|
155
|
+
'anthropic-version': '2023-06-01',
|
|
156
|
+
'x-rtk-version': '0.37.2',
|
|
157
|
+
'x-optimizer-id': 'rtk-cli-hook',
|
|
158
|
+
'traceparent': '00-4bf92f3577b34da6a3ce929d0e0e4736-00f067aa0ba902b7-01'
|
|
159
|
+
},
|
|
160
|
+
body: JSON.stringify(payload)
|
|
161
|
+
});
|
|
162
|
+
|
|
163
|
+
if (res.status !== 200) {
|
|
164
|
+
throw new Error(`Expected HTTP 200, got ${res.status}`);
|
|
165
|
+
}
|
|
166
|
+
return `Headers accepted cleanly with HTTP 200 OK`;
|
|
167
|
+
});
|
|
168
|
+
|
|
169
|
+
// -----------------------------------------------------------------------------
|
|
170
|
+
// TEST 5: OpenAI Chat Healer Mode — Orphaned Tool Result in Chat Completions
|
|
171
|
+
// Same test but over /v1/chat/completions: tool role message without prior tool_calls assistant message.
|
|
172
|
+
// OpenAI API strictly throws: 400 'messages with role tool must be a response to a preceding message with tool_calls'
|
|
173
|
+
// LLM Switcher: Converts orphaned tool to user text message -> 200 OK
|
|
174
|
+
// -----------------------------------------------------------------------------
|
|
175
|
+
await runTest('OpenAI Chat Healer: Orphaned role tool without preceding assistant tool_calls', async () => {
|
|
176
|
+
const payload = {
|
|
177
|
+
model: 'ag/gpt-oss-120b-medium',
|
|
178
|
+
max_tokens: 30,
|
|
179
|
+
messages: [
|
|
180
|
+
{ role: 'user', content: 'Turn 1 context' },
|
|
181
|
+
{ role: 'tool', tool_call_id: 'orphaned_chat_call_77', content: 'Tool execution logs: OK' },
|
|
182
|
+
{ role: 'user', content: 'reply pong' }
|
|
183
|
+
],
|
|
184
|
+
stream: false
|
|
185
|
+
};
|
|
186
|
+
|
|
187
|
+
const res = await fetch(`${BASE_URL}/v1/chat/completions`, {
|
|
188
|
+
method: 'POST',
|
|
189
|
+
headers: { 'Content-Type': 'application/json' },
|
|
190
|
+
body: JSON.stringify(payload)
|
|
191
|
+
});
|
|
192
|
+
|
|
193
|
+
if (res.status !== 200) {
|
|
194
|
+
const err = await res.text();
|
|
195
|
+
throw new Error(`Expected HTTP 200, got ${res.status}: ${err}`);
|
|
196
|
+
}
|
|
197
|
+
const json = await res.json();
|
|
198
|
+
return `Chat Healer rescued orphaned tool role. Stop reason: ${json.choices?.[0]?.finish_reason}`;
|
|
199
|
+
});
|
|
200
|
+
|
|
201
|
+
console.log('\n================================================================');
|
|
202
|
+
console.log(` TEST RESULTS: ${passed} PASSED / ${failed} FAILED `);
|
|
203
|
+
console.log('================================================================\n');
|
|
204
|
+
|
|
205
|
+
process.exit(failed ? 1 : 0);
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
import test from 'node:test';
|
|
2
|
+
import assert from 'node:assert/strict';
|
|
3
|
+
import fs from 'node:fs';
|
|
4
|
+
import os from 'node:os';
|
|
5
|
+
import path from 'node:path';
|
|
6
|
+
import { execFileSync } from 'node:child_process';
|
|
7
|
+
import { fileURLToPath } from 'node:url';
|
|
8
|
+
|
|
9
|
+
const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..');
|
|
10
|
+
const HAS_OPENSSL = process.platform !== 'win32' && fs.existsSync('/usr/bin/openssl');
|
|
11
|
+
|
|
12
|
+
// LibreSSL names the -CAcreateserial file after the CA path cut at its first dot, so a
|
|
13
|
+
// directory such as /Users/first.last sent the serial file to /Users/first.srl.
|
|
14
|
+
test('make-certs.sh builds the certificates under a path that contains a dot', { skip: !HAS_OPENSSL && 'posix + openssl' }, (t) => {
|
|
15
|
+
// The first dot of the whole path is in `first.last`, so a stray serial file lands in `dir` as first.srl.
|
|
16
|
+
const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'llmswcerts'));
|
|
17
|
+
t.after(() => fs.rmSync(dir, { recursive: true, force: true }));
|
|
18
|
+
const out = path.join(dir, 'first.last', 'certs');
|
|
19
|
+
execFileSync('bash', [path.join(ROOT, 'blindfold', 'make-certs.sh'), out], { stdio: 'pipe' });
|
|
20
|
+
for (const f of ['ca.pem', 'ca.key', 'leaf.pem', 'leaf.key']) assert.ok(fs.existsSync(path.join(out, f)), `${f} is missing`);
|
|
21
|
+
assert.deepEqual(fs.readdirSync(dir), ['first.last'], 'no serial file outside the output directory');
|
|
22
|
+
});
|
|
@@ -130,6 +130,8 @@ before(async () => {
|
|
|
130
130
|
proxyPort = await freePort();
|
|
131
131
|
blindfoldPort = await freePort();
|
|
132
132
|
tmpDir = fs.mkdtempSync(path.join(os.tmpdir(), 'llmsw-sim-'));
|
|
133
|
+
// Blindfold mode needs a CA. Without its own directory the gateway reads the certificates of the checkout.
|
|
134
|
+
execFileSync('bash', [path.join(ROOT, 'blindfold', 'make-certs.sh'), path.join(tmpDir, 'certs')], { stdio: 'ignore' });
|
|
133
135
|
|
|
134
136
|
const base = `http://127.0.0.1:${upstreamPort}`;
|
|
135
137
|
const models = { opus: 'up-opus', sonnet: 'up-sonnet', haiku: 'up-haiku', fable: 'up-fable' };
|
|
@@ -150,6 +152,7 @@ before(async () => {
|
|
|
150
152
|
...process.env,
|
|
151
153
|
LLM_SWITCHER_CONFIG: path.join(tmpDir, 'config.json'),
|
|
152
154
|
LLM_SWITCHER_STATE_DIR: tmpDir,
|
|
155
|
+
LLM_SWITCHER_BLINDFOLD_CERTS: path.join(tmpDir, 'certs'),
|
|
153
156
|
CLAUDE_CONFIG_DIR: path.join(tmpDir, 'claude'),
|
|
154
157
|
LLM_SWITCHER_PORT: ''
|
|
155
158
|
},
|