llm-switcher 1.2.0 → 1.2.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (48) hide show
  1. package/CHANGELOG.md +23 -0
  2. package/README.md +1 -0
  3. package/blindfold/make-certs.sh +3 -1
  4. package/catalog.mjs +132 -42
  5. package/classifier.mjs +238 -0
  6. package/docs/TOKEN-OPTIMIZER-INTEROP.md +110 -110
  7. package/docs/response-matrix.json +1130 -1130
  8. package/formats.mjs +30 -1
  9. package/icons/antigravity.png +0 -0
  10. package/icons/claude.png +0 -0
  11. package/icons/codex.png +0 -0
  12. package/icons/deepseek.png +0 -0
  13. package/icons/gemini.png +0 -0
  14. package/icons/github.png +0 -0
  15. package/icons/groq.png +0 -0
  16. package/icons/intact.svg +1 -0
  17. package/icons/ollama.png +0 -0
  18. package/icons/openai.png +0 -0
  19. package/icons/openrouter.png +0 -0
  20. package/icons/qwen.png +0 -0
  21. package/icons/vertex.png +0 -0
  22. package/mcp.mjs +2 -2
  23. package/package.json +1 -1
  24. package/proxy.mjs +34 -20
  25. package/scripts/run-tests.mjs +3 -1
  26. package/skills/llm-switcher/SKILL.md +93 -93
  27. package/state.mjs +58 -4
  28. package/switch +0 -0
  29. package/switch.mjs +28 -4
  30. package/tests/blindfold-e2e.test.mjs +2 -2
  31. package/tests/blindfold-task5.test.mjs +1 -1
  32. package/tests/blindfold.task3.test.mjs +2 -2
  33. package/tests/blindfold.wire.test.mjs +3 -1
  34. package/tests/catalog.test.mjs +117 -19
  35. package/tests/classifier.test.mjs +210 -0
  36. package/tests/codex-daemon.test.mjs +92 -0
  37. package/tests/formats.test.mjs +30 -0
  38. package/tests/helpers.mjs +24 -24
  39. package/tests/lifecycle.test.mjs +49 -37
  40. package/tests/live-optimizer-interop.mjs +205 -205
  41. package/tests/make-certs.test.mjs +22 -0
  42. package/tests/real-user-sim.test.mjs +3 -0
  43. package/tests/shim.test.mjs +8 -3
  44. package/tests/switch.test.mjs +30 -24
  45. package/tests/ui.test.mjs +90 -0
  46. package/tests/version.test.mjs +78 -0
  47. package/ui.html +1895 -1616
  48. package/version.mjs +53 -0
@@ -54,16 +54,16 @@ function envFor(ws) {
54
54
  return { ...process.env, HOME: path.join(ws.dir, 'home'), USERPROFILE: path.join(ws.dir, 'home'), LLM_SWITCHER_CONFIG: ws.cfgPath, LLM_SWITCHER_STATE_DIR: ws.dir, LLM_SWITCHER_BLINDFOLD_CERTS: ws.certDir, CLAUDE_CONFIG_DIR: ws.claudeDir, LLM_SWITCHER_PORT: '', PORT: '' };
55
55
  }
56
56
 
57
- function writeConfig(ws, gwPort, bfPort, activeResponses = null) {
57
+ // The 1.2 schema: one pointer per tool and one interceptor port for the whole config. A file in the
58
+ // older schema is migrated on load, which rewrites it and breaks every "changes nothing" assertion.
59
+ function writeConfig(ws, gwPort, bfPort, activeCodex = null) {
58
60
  fs.writeFileSync(ws.cfgPath, JSON.stringify({
59
61
  port: gwPort,
60
- activeProfiles: { anthropic: null, responses: activeResponses, 'openai-chat': null, vertex: null },
62
+ blindfold: { port: bfPort },
63
+ activeProfiles: { claude: null, codex: activeCodex },
61
64
  profiles: {
62
- plain: { name: 'Plain', mode: 'convert', inFormat: 'auto', baseURL: 'http://127.0.0.1:9/v1', apiKey: 'k', defaultModels: { opus: 'o' } },
63
- bf: {
64
- name: 'Blindfold', mode: 'convert', inFormat: 'responses', baseURL: 'http://127.0.0.1:9/v1', apiKey: 'k',
65
- defaultModels: { main: 'm' }, blindfold: true, blindfoldPort: bfPort
66
- }
65
+ plain: { name: 'Plain', mode: 'convert', tool: 'claude', baseURL: 'http://127.0.0.1:9/v1', apiKey: 'k', defaultModels: { opus: 'o' } },
66
+ bf: { name: 'Blindfold', mode: 'convert', tool: 'codex', baseURL: 'http://127.0.0.1:9/v1', apiKey: 'k', defaultModels: { main: 'm' } }
67
67
  }
68
68
  }, null, 2), { mode: 0o600 });
69
69
  }
@@ -121,7 +121,7 @@ test('the gateway proves its identity to a fresh challenge; a replayed /health i
121
121
  const body = await (await fetch(`http://127.0.0.1:${gwPort}/health?challenge=abc`)).json();
122
122
  // The proof binds the role, the port it answers on and its pid, so it cannot be relayed or edited.
123
123
  assert.equal(body.port, gwPort);
124
- assert.equal(body.proof, crypto.createHmac('sha256', token(ws)).update(['gateway', gwPort, body.pid, '', '', '', 'abc'].join('|')).digest('hex'));
124
+ assert.equal(body.proof, crypto.createHmac('sha256', token(ws)).update(['gateway', gwPort, body.pid, '', '', 'abc'].join('|')).digest('hex'));
125
125
  assert.equal(await probe(ws, `s.probeGateway(${gwPort})`), 'ours');
126
126
  assert.equal(await probe(ws, `s.probeGateway(${replayPort})`), 'foreign');
127
127
  assert.equal(await probe(ws, `s.probeGateway(${await freePort()})`), 'free');
@@ -210,18 +210,21 @@ test('the gateway owns the interceptor: dashboard changes, a lost interceptor an
210
210
  let gw = await startGateway(ws, gwPort);
211
211
  const bfState = () => probe(ws, `s.probeBlindfold(${bfPort})`);
212
212
  try {
213
- // 1. the dashboard turns blindfold on for the Codex target
214
- const on = await api(ws, gwPort, '/api/switch', { target: 'responses', profile: 'bf' });
213
+ // 1. the dashboard turns Codex on
214
+ const on = await api(ws, gwPort, '/api/switch', { target: 'codex', profile: 'bf' });
215
215
  assert.equal(on.success, true, JSON.stringify(on));
216
216
  let st = await waitFor(async () => { const s = await bfState(); return s.state === 'ours' && s; });
217
217
  assert.equal(st.gatewayPort, gwPort);
218
- assert.equal(st.prefix, '/backend-api/codex');
218
+ assert.equal(st.activeTools, 'codex');
219
219
 
220
- // 2. the dashboard changes the prefix: the interceptor is respawned with it
221
- const saved = await api(ws, gwPort, '/api/save-profile', { key: 'bf', profile: { blindfoldPrefix: '/backend-api/codex2' } });
222
- assert.notEqual(saved.success, false, JSON.stringify(saved));
223
- st = await waitFor(async () => { const s = await bfState(); return s.prefix === '/backend-api/codex2' && s; });
224
- assert.equal(st.prefix, '/backend-api/codex2');
220
+ // 2. the dashboard turns Claude on as well: the running interceptor takes the new tool set in place
221
+ const both = await api(ws, gwPort, '/api/switch', { target: 'claude', profile: 'plain' });
222
+ assert.notEqual(both.success, false, JSON.stringify(both));
223
+ st = await waitFor(async () => { const s = await bfState(); return s.activeTools === 'claude,codex' && s; });
224
+ assert.equal(st.activeTools, 'claude,codex');
225
+ await api(ws, gwPort, '/api/switch', { target: 'claude', profile: null });
226
+ st = await waitFor(async () => { const s = await bfState(); return s.activeTools === 'codex' && s; });
227
+ assert.equal(st.activeTools, 'codex');
225
228
 
226
229
  // 3. the interceptor dies; a sync brings it back
227
230
  process.kill(st.pid, 'SIGTERM');
@@ -238,31 +241,30 @@ test('the gateway owns the interceptor: dashboard changes, a lost interceptor an
238
241
  st = await waitFor(async () => { const s = await bfState(); return s.state === 'ours' && s; });
239
242
  assert.equal(st.gatewayPort, gwPort, 'the gateway starts the interceptor at boot');
240
243
 
241
- // 5. a foreign process on the new interceptor port: the caller sees the failure
244
+ // 5. config.json names an interceptor port that a foreign process holds: a dashboard change fails
242
245
  const squatter = net.createServer(s => s.end());
243
246
  const squatPort = await new Promise(r => squatter.listen(0, '127.0.0.1', () => r(squatter.address().port)));
247
+ const good = fs.readFileSync(ws.cfgPath);
244
248
  try {
249
+ fs.writeFileSync(ws.cfgPath, JSON.stringify({ ...JSON.parse(good), blindfold: { port: squatPort } }, null, 2), { mode: 0o600 });
250
+ const edited = fs.readFileSync(ws.cfgPath);
245
251
  const badRes = await fetch(`http://127.0.0.1:${gwPort}/api/save-profile`, {
246
252
  method: 'POST', headers: { 'Content-Type': 'application/json', 'x-llm-switcher-token': token(ws) },
247
- body: JSON.stringify({ key: 'bf', profile: { blindfoldPort: squatPort } })
253
+ body: JSON.stringify({ key: 'bf', profile: { name: 'Renamed' } })
248
254
  });
249
255
  assert.equal(badRes.status, 502, 'a failure is not answered with 200');
250
256
  const bad = await badRes.json();
251
257
  assert.equal(bad.success, false);
252
258
  assert.match(bad.error, /held by another process/);
253
- const onDisk = JSON.parse(fs.readFileSync(ws.cfgPath, 'utf8')).profiles.bf.blindfoldPort;
254
- assert.equal(onDisk, bfPort, 'a refused change is not saved, so HTTPS_PROXY never points at the squatter');
259
+ assert.deepEqual(fs.readFileSync(ws.cfgPath), edited, 'a refused change is not saved');
255
260
  assert.equal((await bfState()).state, 'ours', 'the working interceptor stays up');
256
- const status = await fetch(`http://127.0.0.1:${gwPort}/api/status`, { headers: { 'x-llm-switcher-token': token(ws) } }).then(r => r.json());
257
- assert.equal(status.config.profiles.bf.blindfoldPort, bfPort, 'the gateway does not keep the refused change in memory');
258
261
  } finally {
259
262
  squatter.close();
263
+ fs.writeFileSync(ws.cfgPath, good, { mode: 0o600 });
260
264
  }
261
265
 
262
266
  // 6. turning the Codex target off stops the interceptor
263
- await api(ws, gwPort, '/api/save-profile', { key: 'bf', profile: { blindfoldPort: bfPort } });
264
- await waitFor(async () => (await bfState()).state === 'ours');
265
- await api(ws, gwPort, '/api/switch', { target: 'responses', profile: null });
267
+ await api(ws, gwPort, '/api/switch', { target: 'codex', profile: null });
266
268
  assert.equal(await waitFor(async () => (await bfState()).state === 'free'), true);
267
269
  } finally {
268
270
  try { const s = await bfState(); if (s.state === 'ours') process.kill(s.pid, 'SIGTERM'); } catch {}
@@ -276,32 +278,42 @@ test('the gateway owns the interceptor: dashboard changes, a lost interceptor an
276
278
  test('concurrent admin changes: a refused change never reaches disk, and no accepted change is lost', { skip: !HAS_OPENSSL && 'posix + openssl' }, async () => {
277
279
  const ws = makeWorkspace({ certs: true });
278
280
  const gwPort = await freePort();
279
- const bfPort = await freePort();
280
- writeConfig(ws, gwPort, bfPort, 'bf');
281
- const launch = snapshotLaunchFiles(ws.dir);
282
- const gw = await startGateway(ws, gwPort);
281
+ let bfPort;
283
282
  // A squatter that accepts and never answers keeps the identity probe waiting.
284
283
  const silent = net.createServer(() => {});
285
284
  const silentPort = await new Promise(r => silent.listen(0, '127.0.0.1', () => r(silent.address().port)));
285
+ writeConfig(ws, gwPort, silentPort);
286
+ const launch = snapshotLaunchFiles(ws.dir);
287
+ const gw = await startGateway(ws, gwPort);
286
288
  try {
287
- await waitFor(async () => (await probe(ws, `s.probeBlindfold(${bfPort})`)).state === 'ours');
288
- const post = (body) => fetch(`http://127.0.0.1:${gwPort}/api/save-profile`, {
289
+ const post = (p, body) => fetch(`http://127.0.0.1:${gwPort}${p}`, {
289
290
  method: 'POST', headers: { 'Content-Type': 'application/json', 'x-llm-switcher-token': token(ws) }, body: JSON.stringify(body)
290
291
  });
291
- const refused = post({ key: 'bf', profile: { blindfoldPort: silentPort } });
292
+ // Turning Codex on waits for the silent port and is refused; the rename that arrives meanwhile is kept.
293
+ const refused = post('/api/switch', { target: 'codex', profile: 'bf' });
292
294
  await new Promise(r => setTimeout(r, 300));
293
- const other = post({ key: 'plain', profile: { name: 'Renamed' } });
295
+ const other = post('/api/save-profile', { key: 'plain', profile: { name: 'Renamed' } });
294
296
  const [a, b] = await Promise.all([refused, other]);
295
297
  assert.equal(a.status, 502);
296
298
  assert.equal(b.status, 200);
297
- const onDisk = JSON.parse(fs.readFileSync(ws.cfgPath, 'utf8'));
298
- assert.equal(onDisk.profiles.bf.blindfoldPort, bfPort, 'the refused port is not saved by the other request');
299
+ let onDisk = JSON.parse(fs.readFileSync(ws.cfgPath, 'utf8'));
300
+ assert.equal(onDisk.activeProfiles.codex, null, 'the refused switch is not saved by the other request');
299
301
  assert.equal(onDisk.profiles.plain.name, 'Renamed');
300
302
 
303
+ // A usable interceptor port, and Codex on, for the second half. Chosen only now: other test files
304
+ // run at the same time, and a port picked seconds earlier can be taken in between.
305
+ bfPort = await freePort();
306
+ onDisk.blindfold = { port: bfPort };
307
+ fs.writeFileSync(ws.cfgPath, JSON.stringify(onDisk, null, 2), { mode: 0o600 });
308
+ const on = await post('/api/switch', { target: 'codex', profile: 'bf' });
309
+ assert.equal(on.status, 200, await on.text());
310
+ await waitFor(async () => (await probe(ws, `s.probeBlindfold(${bfPort})`)).state === 'ours');
311
+ const postProfile = (body) => post('/api/save-profile', body);
312
+
301
313
  // Two accepted changes to the active profile: the one that saves last keeps the other.
302
314
  const [c, d] = await Promise.all([
303
- post({ key: 'bf', profile: { name: 'First' } }),
304
- post({ key: 'bf', profile: { defaultModels: { main: 'm2' } } })
315
+ postProfile({ key: 'bf', profile: { name: 'First' } }),
316
+ postProfile({ key: 'bf', profile: { defaultModels: { main: 'm2' } } })
305
317
  ]);
306
318
  assert.equal(c.status, 200);
307
319
  assert.equal(d.status, 200);
@@ -1,205 +1,205 @@
1
- // LIVE test: requires a running gateway (switch on) and a real upstream (costs tokens).
2
- // Offline test, no network needed: npm test
3
- const GATEWAY_PORT = Number(process.env.LLM_SWITCHER_PORT) || 3456;
4
- const BASE_URL = `http://127.0.0.1:${GATEWAY_PORT}`;
5
-
6
- console.log('================================================================');
7
- console.log(' LLM SWITCHER — TOKEN OPTIMIZER INTEROPERABILITY TEST SUITE ');
8
- console.log(' Simulating failure modes from Headroom, RTK, and Ponytail ');
9
- console.log('================================================================\n');
10
-
11
- let passed = 0;
12
- let failed = 0;
13
-
14
- async function runTest(name, fn) {
15
- process.stdout.write(`[TEST] ${name}... `);
16
- const t0 = Date.now();
17
- try {
18
- const detail = await fn();
19
- console.log(`PASS (${Date.now() - t0}ms)`);
20
- if (detail) console.log(` ↳ ${detail}`);
21
- passed++;
22
- } catch (err) {
23
- console.log(`FAIL (${Date.now() - t0}ms)`);
24
- console.log(` ↳ ERROR: ${err.message}`);
25
- failed++;
26
- }
27
- }
28
-
29
- // -----------------------------------------------------------------------------
30
- // TEST 1: Headroom Failure Mode — Orphaned tool_result
31
- // When Headroom prunes history to save tokens, it drops the assistant tool_use turn.
32
- // Anthropic strictly throws: 400 invalid_request_error: 'tool_use_id does not correspond to any tool_use'
33
- // LLM Switcher Healer Engine: Repairs orphaned result into context text -> 200 OK
34
- // -----------------------------------------------------------------------------
35
- await runTest('Headroom Simulation: Orphaned tool_result turn', async () => {
36
- const payload = {
37
- model: 'claude-opus-4-6',
38
- max_tokens: 40,
39
- messages: [
40
- { role: 'user', content: 'Context turn before pruning' },
41
- {
42
- role: 'user',
43
- content: [
44
- { type: 'tool_result', tool_use_id: 'headroom_pruned_call_99', content: 'Database query result: 42 rows found' },
45
- { type: 'text', text: 'Reply: healed successfully' }
46
- ]
47
- }
48
- ],
49
- stream: false
50
- };
51
-
52
- const res = await fetch(`${BASE_URL}/v1/messages`, {
53
- method: 'POST',
54
- headers: { 'Content-Type': 'application/json', 'anthropic-version': '2023-06-01' },
55
- body: JSON.stringify(payload)
56
- });
57
-
58
- if (res.status !== 200) {
59
- const err = await res.text();
60
- throw new Error(`Expected HTTP 200, got ${res.status}: ${err}`);
61
- }
62
- const json = await res.json();
63
- const text = json.content?.map(c => c.text).join('') || '';
64
- return `Healed orphaned tool_result. Response HTTP 200: "${text.slice(0, 50).replace(/\n/g, ' ')}..."`;
65
- });
66
-
67
- // -----------------------------------------------------------------------------
68
- // TEST 2: Headroom Failure Mode — Consecutive User turns (Roles must alternate)
69
- // When Headroom collapses history, multiple user turns occur back-to-back.
70
- // Anthropic strictly throws: 400 invalid_request_error: 'roles must alternate'
71
- // LLM Switcher Healer Engine: Merges consecutive same-role turns -> 200 OK
72
- // -----------------------------------------------------------------------------
73
- await runTest('Headroom Simulation: Consecutive User turns (Role alternation violation)', async () => {
74
- const payload = {
75
- model: 'claude-opus-4-6',
76
- max_tokens: 30,
77
- messages: [
78
- { role: 'user', content: 'User message turn 1' },
79
- { role: 'user', content: 'User message turn 2 (no assistant in between)' },
80
- { role: 'user', content: 'User message turn 3: reply pong' }
81
- ],
82
- stream: false
83
- };
84
-
85
- const res = await fetch(`${BASE_URL}/v1/messages`, {
86
- method: 'POST',
87
- headers: { 'Content-Type': 'application/json', 'anthropic-version': '2023-06-01' },
88
- body: JSON.stringify(payload)
89
- });
90
-
91
- if (res.status !== 200) {
92
- const err = await res.text();
93
- throw new Error(`Expected HTTP 200, got ${res.status}: ${err}`);
94
- }
95
- const json = await res.json();
96
- return `Merged consecutive turns seamlessly. Stop reason: ${json.stop_reason}`;
97
- });
98
-
99
- // -----------------------------------------------------------------------------
100
- // TEST 3: Headroom / RTK Failure Mode — Stripped Thinking Parameter
101
- // Optimizer stripped the 'thinking' object to reduce tokens.
102
- // LLM Switcher Thinking Guard: Detects reasoning model (ag/claude-opus-4-6-thinking)
103
- // and restores thinking.budget_tokens automatically -> thinking_delta emitted!
104
- // -----------------------------------------------------------------------------
105
- await runTest('Thinking Guard: Restoring stripped thinking parameter on reasoning models', async () => {
106
- const payload = {
107
- model: 'claude-opus-4-6',
108
- max_tokens: 200,
109
- // Note: NO thinking parameter included (simulating optimizer stripping it)
110
- messages: [
111
- { role: 'user', content: 'Solve step by step: what is 17 * 23? Show reasoning.' }
112
- ],
113
- stream: true
114
- };
115
-
116
- const res = await fetch(`${BASE_URL}/v1/messages`, {
117
- method: 'POST',
118
- headers: { 'Content-Type': 'application/json', 'anthropic-version': '2023-06-01' },
119
- body: JSON.stringify(payload)
120
- });
121
-
122
- if (res.status !== 200) {
123
- throw new Error(`Expected HTTP 200, got ${res.status}`);
124
- }
125
-
126
- const text = await res.text();
127
- const hasThinkingDelta = text.includes('thinking_delta');
128
- const hasSignatureDelta = text.includes('signature_delta');
129
- const hasTextDelta = text.includes('text_delta');
130
-
131
- if (!hasThinkingDelta) {
132
- throw new Error('Thinking Guard failed: thinking_delta was NOT emitted in stream!');
133
- }
134
-
135
- return `Automatically restored thinking: thinking_delta=${hasThinkingDelta}, signature_delta=${hasSignatureDelta}, text_delta=${hasTextDelta}`;
136
- });
137
-
138
- // -----------------------------------------------------------------------------
139
- // TEST 4: RTK & Intermediary Headers Passthrough
140
- // RTK / tracing tools inject headers: x-rtk-version, traceparent, x-request-id.
141
- // LLM Switcher: Transparently preserves all tracking headers without rejection.
142
- // -----------------------------------------------------------------------------
143
- await runTest('RTK Intermediary: Custom headers and traceparent passthrough', async () => {
144
- const payload = {
145
- model: 'claude-opus-4-6',
146
- max_tokens: 20,
147
- messages: [{ role: 'user', content: 'ping' }],
148
- stream: false
149
- };
150
-
151
- const res = await fetch(`${BASE_URL}/v1/messages`, {
152
- method: 'POST',
153
- headers: {
154
- 'Content-Type': 'application/json',
155
- 'anthropic-version': '2023-06-01',
156
- 'x-rtk-version': '0.37.2',
157
- 'x-optimizer-id': 'rtk-cli-hook',
158
- 'traceparent': '00-4bf92f3577b34da6a3ce929d0e0e4736-00f067aa0ba902b7-01'
159
- },
160
- body: JSON.stringify(payload)
161
- });
162
-
163
- if (res.status !== 200) {
164
- throw new Error(`Expected HTTP 200, got ${res.status}`);
165
- }
166
- return `Headers accepted cleanly with HTTP 200 OK`;
167
- });
168
-
169
- // -----------------------------------------------------------------------------
170
- // TEST 5: OpenAI Chat Healer Mode — Orphaned Tool Result in Chat Completions
171
- // Same test but over /v1/chat/completions: tool role message without prior tool_calls assistant message.
172
- // OpenAI API strictly throws: 400 'messages with role tool must be a response to a preceding message with tool_calls'
173
- // LLM Switcher: Converts orphaned tool to user text message -> 200 OK
174
- // -----------------------------------------------------------------------------
175
- await runTest('OpenAI Chat Healer: Orphaned role tool without preceding assistant tool_calls', async () => {
176
- const payload = {
177
- model: 'ag/gpt-oss-120b-medium',
178
- max_tokens: 30,
179
- messages: [
180
- { role: 'user', content: 'Turn 1 context' },
181
- { role: 'tool', tool_call_id: 'orphaned_chat_call_77', content: 'Tool execution logs: OK' },
182
- { role: 'user', content: 'reply pong' }
183
- ],
184
- stream: false
185
- };
186
-
187
- const res = await fetch(`${BASE_URL}/v1/chat/completions`, {
188
- method: 'POST',
189
- headers: { 'Content-Type': 'application/json' },
190
- body: JSON.stringify(payload)
191
- });
192
-
193
- if (res.status !== 200) {
194
- const err = await res.text();
195
- throw new Error(`Expected HTTP 200, got ${res.status}: ${err}`);
196
- }
197
- const json = await res.json();
198
- return `Chat Healer rescued orphaned tool role. Stop reason: ${json.choices?.[0]?.finish_reason}`;
199
- });
200
-
201
- console.log('\n================================================================');
202
- console.log(` TEST RESULTS: ${passed} PASSED / ${failed} FAILED `);
203
- console.log('================================================================\n');
204
-
205
- process.exit(failed ? 1 : 0);
1
+ // LIVE test: requires a running gateway (switch on) and a real upstream (costs tokens).
2
+ // Offline test, no network needed: npm test
3
+ const GATEWAY_PORT = Number(process.env.LLM_SWITCHER_PORT) || 3456;
4
+ const BASE_URL = `http://127.0.0.1:${GATEWAY_PORT}`;
5
+
6
+ console.log('================================================================');
7
+ console.log(' LLM SWITCHER — TOKEN OPTIMIZER INTEROPERABILITY TEST SUITE ');
8
+ console.log(' Simulating failure modes from Headroom, RTK, and Ponytail ');
9
+ console.log('================================================================\n');
10
+
11
+ let passed = 0;
12
+ let failed = 0;
13
+
14
+ async function runTest(name, fn) {
15
+ process.stdout.write(`[TEST] ${name}... `);
16
+ const t0 = Date.now();
17
+ try {
18
+ const detail = await fn();
19
+ console.log(`PASS (${Date.now() - t0}ms)`);
20
+ if (detail) console.log(` ↳ ${detail}`);
21
+ passed++;
22
+ } catch (err) {
23
+ console.log(`FAIL (${Date.now() - t0}ms)`);
24
+ console.log(` ↳ ERROR: ${err.message}`);
25
+ failed++;
26
+ }
27
+ }
28
+
29
+ // -----------------------------------------------------------------------------
30
+ // TEST 1: Headroom Failure Mode — Orphaned tool_result
31
+ // When Headroom prunes history to save tokens, it drops the assistant tool_use turn.
32
+ // Anthropic strictly throws: 400 invalid_request_error: 'tool_use_id does not correspond to any tool_use'
33
+ // LLM Switcher Healer Engine: Repairs orphaned result into context text -> 200 OK
34
+ // -----------------------------------------------------------------------------
35
+ await runTest('Headroom Simulation: Orphaned tool_result turn', async () => {
36
+ const payload = {
37
+ model: 'claude-opus-4-6',
38
+ max_tokens: 40,
39
+ messages: [
40
+ { role: 'user', content: 'Context turn before pruning' },
41
+ {
42
+ role: 'user',
43
+ content: [
44
+ { type: 'tool_result', tool_use_id: 'headroom_pruned_call_99', content: 'Database query result: 42 rows found' },
45
+ { type: 'text', text: 'Reply: healed successfully' }
46
+ ]
47
+ }
48
+ ],
49
+ stream: false
50
+ };
51
+
52
+ const res = await fetch(`${BASE_URL}/v1/messages`, {
53
+ method: 'POST',
54
+ headers: { 'Content-Type': 'application/json', 'anthropic-version': '2023-06-01' },
55
+ body: JSON.stringify(payload)
56
+ });
57
+
58
+ if (res.status !== 200) {
59
+ const err = await res.text();
60
+ throw new Error(`Expected HTTP 200, got ${res.status}: ${err}`);
61
+ }
62
+ const json = await res.json();
63
+ const text = json.content?.map(c => c.text).join('') || '';
64
+ return `Healed orphaned tool_result. Response HTTP 200: "${text.slice(0, 50).replace(/\n/g, ' ')}..."`;
65
+ });
66
+
67
+ // -----------------------------------------------------------------------------
68
+ // TEST 2: Headroom Failure Mode — Consecutive User turns (Roles must alternate)
69
+ // When Headroom collapses history, multiple user turns occur back-to-back.
70
+ // Anthropic strictly throws: 400 invalid_request_error: 'roles must alternate'
71
+ // LLM Switcher Healer Engine: Merges consecutive same-role turns -> 200 OK
72
+ // -----------------------------------------------------------------------------
73
+ await runTest('Headroom Simulation: Consecutive User turns (Role alternation violation)', async () => {
74
+ const payload = {
75
+ model: 'claude-opus-4-6',
76
+ max_tokens: 30,
77
+ messages: [
78
+ { role: 'user', content: 'User message turn 1' },
79
+ { role: 'user', content: 'User message turn 2 (no assistant in between)' },
80
+ { role: 'user', content: 'User message turn 3: reply pong' }
81
+ ],
82
+ stream: false
83
+ };
84
+
85
+ const res = await fetch(`${BASE_URL}/v1/messages`, {
86
+ method: 'POST',
87
+ headers: { 'Content-Type': 'application/json', 'anthropic-version': '2023-06-01' },
88
+ body: JSON.stringify(payload)
89
+ });
90
+
91
+ if (res.status !== 200) {
92
+ const err = await res.text();
93
+ throw new Error(`Expected HTTP 200, got ${res.status}: ${err}`);
94
+ }
95
+ const json = await res.json();
96
+ return `Merged consecutive turns seamlessly. Stop reason: ${json.stop_reason}`;
97
+ });
98
+
99
+ // -----------------------------------------------------------------------------
100
+ // TEST 3: Headroom / RTK Failure Mode — Stripped Thinking Parameter
101
+ // Optimizer stripped the 'thinking' object to reduce tokens.
102
+ // LLM Switcher Thinking Guard: Detects reasoning model (ag/claude-opus-4-6-thinking)
103
+ // and restores thinking.budget_tokens automatically -> thinking_delta emitted!
104
+ // -----------------------------------------------------------------------------
105
+ await runTest('Thinking Guard: Restoring stripped thinking parameter on reasoning models', async () => {
106
+ const payload = {
107
+ model: 'claude-opus-4-6',
108
+ max_tokens: 200,
109
+ // Note: NO thinking parameter included (simulating optimizer stripping it)
110
+ messages: [
111
+ { role: 'user', content: 'Solve step by step: what is 17 * 23? Show reasoning.' }
112
+ ],
113
+ stream: true
114
+ };
115
+
116
+ const res = await fetch(`${BASE_URL}/v1/messages`, {
117
+ method: 'POST',
118
+ headers: { 'Content-Type': 'application/json', 'anthropic-version': '2023-06-01' },
119
+ body: JSON.stringify(payload)
120
+ });
121
+
122
+ if (res.status !== 200) {
123
+ throw new Error(`Expected HTTP 200, got ${res.status}`);
124
+ }
125
+
126
+ const text = await res.text();
127
+ const hasThinkingDelta = text.includes('thinking_delta');
128
+ const hasSignatureDelta = text.includes('signature_delta');
129
+ const hasTextDelta = text.includes('text_delta');
130
+
131
+ if (!hasThinkingDelta) {
132
+ throw new Error('Thinking Guard failed: thinking_delta was NOT emitted in stream!');
133
+ }
134
+
135
+ return `Automatically restored thinking: thinking_delta=${hasThinkingDelta}, signature_delta=${hasSignatureDelta}, text_delta=${hasTextDelta}`;
136
+ });
137
+
138
+ // -----------------------------------------------------------------------------
139
+ // TEST 4: RTK & Intermediary Headers Passthrough
140
+ // RTK / tracing tools inject headers: x-rtk-version, traceparent, x-request-id.
141
+ // LLM Switcher: Transparently preserves all tracking headers without rejection.
142
+ // -----------------------------------------------------------------------------
143
+ await runTest('RTK Intermediary: Custom headers and traceparent passthrough', async () => {
144
+ const payload = {
145
+ model: 'claude-opus-4-6',
146
+ max_tokens: 20,
147
+ messages: [{ role: 'user', content: 'ping' }],
148
+ stream: false
149
+ };
150
+
151
+ const res = await fetch(`${BASE_URL}/v1/messages`, {
152
+ method: 'POST',
153
+ headers: {
154
+ 'Content-Type': 'application/json',
155
+ 'anthropic-version': '2023-06-01',
156
+ 'x-rtk-version': '0.37.2',
157
+ 'x-optimizer-id': 'rtk-cli-hook',
158
+ 'traceparent': '00-4bf92f3577b34da6a3ce929d0e0e4736-00f067aa0ba902b7-01'
159
+ },
160
+ body: JSON.stringify(payload)
161
+ });
162
+
163
+ if (res.status !== 200) {
164
+ throw new Error(`Expected HTTP 200, got ${res.status}`);
165
+ }
166
+ return `Headers accepted cleanly with HTTP 200 OK`;
167
+ });
168
+
169
+ // -----------------------------------------------------------------------------
170
+ // TEST 5: OpenAI Chat Healer Mode — Orphaned Tool Result in Chat Completions
171
+ // Same test but over /v1/chat/completions: tool role message without prior tool_calls assistant message.
172
+ // OpenAI API strictly throws: 400 'messages with role tool must be a response to a preceding message with tool_calls'
173
+ // LLM Switcher: Converts orphaned tool to user text message -> 200 OK
174
+ // -----------------------------------------------------------------------------
175
+ await runTest('OpenAI Chat Healer: Orphaned role tool without preceding assistant tool_calls', async () => {
176
+ const payload = {
177
+ model: 'ag/gpt-oss-120b-medium',
178
+ max_tokens: 30,
179
+ messages: [
180
+ { role: 'user', content: 'Turn 1 context' },
181
+ { role: 'tool', tool_call_id: 'orphaned_chat_call_77', content: 'Tool execution logs: OK' },
182
+ { role: 'user', content: 'reply pong' }
183
+ ],
184
+ stream: false
185
+ };
186
+
187
+ const res = await fetch(`${BASE_URL}/v1/chat/completions`, {
188
+ method: 'POST',
189
+ headers: { 'Content-Type': 'application/json' },
190
+ body: JSON.stringify(payload)
191
+ });
192
+
193
+ if (res.status !== 200) {
194
+ const err = await res.text();
195
+ throw new Error(`Expected HTTP 200, got ${res.status}: ${err}`);
196
+ }
197
+ const json = await res.json();
198
+ return `Chat Healer rescued orphaned tool role. Stop reason: ${json.choices?.[0]?.finish_reason}`;
199
+ });
200
+
201
+ console.log('\n================================================================');
202
+ console.log(` TEST RESULTS: ${passed} PASSED / ${failed} FAILED `);
203
+ console.log('================================================================\n');
204
+
205
+ process.exit(failed ? 1 : 0);
@@ -0,0 +1,22 @@
1
+ import test from 'node:test';
2
+ import assert from 'node:assert/strict';
3
+ import fs from 'node:fs';
4
+ import os from 'node:os';
5
+ import path from 'node:path';
6
+ import { execFileSync } from 'node:child_process';
7
+ import { fileURLToPath } from 'node:url';
8
+
9
+ const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..');
10
+ const HAS_OPENSSL = process.platform !== 'win32' && fs.existsSync('/usr/bin/openssl');
11
+
12
+ // LibreSSL names the -CAcreateserial file after the CA path cut at its first dot, so a
13
+ // directory such as /Users/first.last sent the serial file to /Users/first.srl.
14
+ test('make-certs.sh builds the certificates under a path that contains a dot', { skip: !HAS_OPENSSL && 'posix + openssl' }, (t) => {
15
+ // The first dot of the whole path is in `first.last`, so a stray serial file lands in `dir` as first.srl.
16
+ const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'llmswcerts'));
17
+ t.after(() => fs.rmSync(dir, { recursive: true, force: true }));
18
+ const out = path.join(dir, 'first.last', 'certs');
19
+ execFileSync('bash', [path.join(ROOT, 'blindfold', 'make-certs.sh'), out], { stdio: 'pipe' });
20
+ for (const f of ['ca.pem', 'ca.key', 'leaf.pem', 'leaf.key']) assert.ok(fs.existsSync(path.join(out, f)), `${f} is missing`);
21
+ assert.deepEqual(fs.readdirSync(dir), ['first.last'], 'no serial file outside the output directory');
22
+ });
@@ -130,6 +130,8 @@ before(async () => {
130
130
  proxyPort = await freePort();
131
131
  blindfoldPort = await freePort();
132
132
  tmpDir = fs.mkdtempSync(path.join(os.tmpdir(), 'llmsw-sim-'));
133
+ // Blindfold mode needs a CA. Without its own directory the gateway reads the certificates of the checkout.
134
+ execFileSync('bash', [path.join(ROOT, 'blindfold', 'make-certs.sh'), path.join(tmpDir, 'certs')], { stdio: 'ignore' });
133
135
 
134
136
  const base = `http://127.0.0.1:${upstreamPort}`;
135
137
  const models = { opus: 'up-opus', sonnet: 'up-sonnet', haiku: 'up-haiku', fable: 'up-fable' };
@@ -150,6 +152,7 @@ before(async () => {
150
152
  ...process.env,
151
153
  LLM_SWITCHER_CONFIG: path.join(tmpDir, 'config.json'),
152
154
  LLM_SWITCHER_STATE_DIR: tmpDir,
155
+ LLM_SWITCHER_BLINDFOLD_CERTS: path.join(tmpDir, 'certs'),
153
156
  CLAUDE_CONFIG_DIR: path.join(tmpDir, 'claude'),
154
157
  LLM_SWITCHER_PORT: ''
155
158
  },