llm-switcher 1.2.6 → 1.2.8
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.impeccable/hook.cache.json +1 -0
- package/CHANGELOG.md +57 -0
- package/README.md +74 -4
- package/README.vi.md +75 -4
- package/blindfold/blindfold.mjs +2 -1
- package/docs/TOKEN-OPTIMIZER-INTEROP.md +110 -110
- package/docs/cross-platform.md +3 -1
- package/docs/response-matrix.json +1130 -1130
- package/hook-status.mjs +51 -0
- package/notify-route.ps1 +46 -0
- package/package.json +2 -2
- package/plugin.mjs +204 -0
- package/proxy.mjs +8 -3
- package/service.mjs +1 -1
- package/shim.mjs +28 -2
- package/skills/llm-switcher/SKILL.md +93 -93
- package/state.mjs +41 -5
- package/switch +0 -0
- package/switch.cmd +2 -2
- package/switch.mjs +53 -2
- package/tests/blindfold.test.mjs +1 -1
- package/tests/cdp.mjs +297 -0
- package/tests/dashboard-fixture.mjs +167 -0
- package/tests/dashboard.e2e.test.mjs +786 -0
- package/tests/gateway.e2e.test.mjs +1 -1
- package/tests/helpers.mjs +24 -24
- package/tests/hook-status.test.mjs +119 -0
- package/tests/live-optimizer-interop.mjs +205 -205
- package/tests/plugin.test.mjs +132 -0
- package/tests/real-user-sim.test.mjs +2 -2
- package/tests/service.test.mjs +2 -0
- package/tests/shim.test.mjs +112 -0
- package/tests/state.test.mjs +130 -5
- package/tests/ui.test.mjs +1 -1
- package/ui.html +121 -20
|
@@ -470,7 +470,7 @@ test('save-profile rejects a model name that could act as a command', async () =
|
|
|
470
470
|
assert.equal(ok.status, 200);
|
|
471
471
|
});
|
|
472
472
|
|
|
473
|
-
test('Codex WS transport: upstream 429 becomes response.failed with rate_limit_exceeded', async () => { const ws = new WebSocket(`ws://127.0.0.1:${proxyPort}/v1/responses`);
|
|
473
|
+
test('Codex WS transport: upstream 429 becomes response.failed with rate_limit_exceeded', { skip: typeof WebSocket === 'undefined' && 'global WebSocket needs Node 22+' }, async () => { const ws = new WebSocket(`ws://127.0.0.1:${proxyPort}/v1/responses`);
|
|
474
474
|
await new Promise((resolve, reject) => {
|
|
475
475
|
const t = setTimeout(() => reject(new Error('ws open timeout')), 5000);
|
|
476
476
|
ws.addEventListener('open', () => { clearTimeout(t); resolve(); }, { once: true });
|
package/tests/helpers.mjs
CHANGED
|
@@ -1,24 +1,24 @@
|
|
|
1
|
-
import assert from 'node:assert/strict';
|
|
2
|
-
|
|
3
|
-
// Validate an Anthropic event sequence: indexes increase in start order, deltas only target an open block.
|
|
4
|
-
export function assertValidAnthropicEvents(events) {
|
|
5
|
-
const open = new Set();
|
|
6
|
-
const seen = new Set();
|
|
7
|
-
let expectedNext = 0;
|
|
8
|
-
for (const { event, data } of events) {
|
|
9
|
-
if (event === 'content_block_start') {
|
|
10
|
-
assert.equal(data.index, expectedNext, `block index must be sequential (got ${data.index}, want ${expectedNext})`);
|
|
11
|
-
assert.ok(!seen.has(data.index), `index ${data.index} reused`);
|
|
12
|
-
seen.add(data.index);
|
|
13
|
-
open.add(data.index);
|
|
14
|
-
expectedNext++;
|
|
15
|
-
} else if (event === 'content_block_delta') {
|
|
16
|
-
assert.ok(open.has(data.index), `delta for block ${data.index} which is not open`);
|
|
17
|
-
} else if (event === 'content_block_stop') {
|
|
18
|
-
assert.ok(open.has(data.index), `stop for block ${data.index} which is not open`);
|
|
19
|
-
open.delete(data.index);
|
|
20
|
-
} else if (event === 'message_stop') {
|
|
21
|
-
assert.equal(open.size, 0, 'all blocks must be closed before message_stop');
|
|
22
|
-
}
|
|
23
|
-
}
|
|
24
|
-
}
|
|
1
|
+
import assert from 'node:assert/strict';
|
|
2
|
+
|
|
3
|
+
// Validate an Anthropic event sequence: indexes increase in start order, deltas only target an open block.
|
|
4
|
+
export function assertValidAnthropicEvents(events) {
|
|
5
|
+
const open = new Set();
|
|
6
|
+
const seen = new Set();
|
|
7
|
+
let expectedNext = 0;
|
|
8
|
+
for (const { event, data } of events) {
|
|
9
|
+
if (event === 'content_block_start') {
|
|
10
|
+
assert.equal(data.index, expectedNext, `block index must be sequential (got ${data.index}, want ${expectedNext})`);
|
|
11
|
+
assert.ok(!seen.has(data.index), `index ${data.index} reused`);
|
|
12
|
+
seen.add(data.index);
|
|
13
|
+
open.add(data.index);
|
|
14
|
+
expectedNext++;
|
|
15
|
+
} else if (event === 'content_block_delta') {
|
|
16
|
+
assert.ok(open.has(data.index), `delta for block ${data.index} which is not open`);
|
|
17
|
+
} else if (event === 'content_block_stop') {
|
|
18
|
+
assert.ok(open.has(data.index), `stop for block ${data.index} which is not open`);
|
|
19
|
+
open.delete(data.index);
|
|
20
|
+
} else if (event === 'message_stop') {
|
|
21
|
+
assert.equal(open.size, 0, 'all blocks must be closed before message_stop');
|
|
22
|
+
}
|
|
23
|
+
}
|
|
24
|
+
}
|
|
@@ -0,0 +1,119 @@
|
|
|
1
|
+
// The launch hook: what a coding tool shows the person about the switcher at the start of a session.
|
|
2
|
+
//
|
|
3
|
+
// The shim toast fires only when the shim runs, and it needs the shim directory first on PATH. A
|
|
4
|
+
// hook runs inside the tool itself, so it reports even when the launcher is the person's own script.
|
|
5
|
+
// Both tools read one JSON object from stdout and show `systemMessage` to the person.
|
|
6
|
+
import { test } from 'node:test';
|
|
7
|
+
import assert from 'node:assert/strict';
|
|
8
|
+
import fs from 'node:fs';
|
|
9
|
+
import os from 'node:os';
|
|
10
|
+
import path from 'node:path';
|
|
11
|
+
import http from 'node:http';
|
|
12
|
+
import { execFileSync } from 'node:child_process';
|
|
13
|
+
import { fileURLToPath, pathToFileURL } from 'node:url';
|
|
14
|
+
|
|
15
|
+
const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..');
|
|
16
|
+
const HOOK = path.join(ROOT, 'hook-status.mjs');
|
|
17
|
+
const { identityProof } = await import(pathToFileURL(path.join(ROOT, 'state.mjs')).href);
|
|
18
|
+
|
|
19
|
+
function tmpState(t) {
|
|
20
|
+
const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'llmsw-hook-'));
|
|
21
|
+
t.after(() => fs.rmSync(dir, { recursive: true, force: true }));
|
|
22
|
+
// The proof is an HMAC over the admin token, so the fake gateway needs the same file.
|
|
23
|
+
fs.writeFileSync(path.join(dir, 'admin.token'), 'tok-for-the-hook-test', { mode: 0o600 });
|
|
24
|
+
return dir;
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
function writeConfig(dir, port) {
|
|
28
|
+
fs.writeFileSync(path.join(dir, 'config.json'), JSON.stringify({
|
|
29
|
+
port,
|
|
30
|
+
activeProfiles: { claude: 'cl', codex: null },
|
|
31
|
+
profiles: {
|
|
32
|
+
cl: {
|
|
33
|
+
name: 'Intact Claude', baseURL: 'https://intact.example.io/v1', apiKey: 'sk-secret',
|
|
34
|
+
defaultModels: { opus: 'gemini-3.8-flash' }, model1M: { opus: true }
|
|
35
|
+
}
|
|
36
|
+
}
|
|
37
|
+
}), { mode: 0o600 });
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
// A gateway of this build: it answers /health with a proof over the challenge.
|
|
41
|
+
function fakeGateway(t, port) {
|
|
42
|
+
const srv = http.createServer((req, res) => {
|
|
43
|
+
const nonce = new URL(req.url, 'http://127.0.0.1').searchParams.get('challenge') || '';
|
|
44
|
+
const body = { proxy: 'llm-switcher', port, pid: process.pid };
|
|
45
|
+
body.proof = identityProof(nonce, { role: 'gateway', port, pid: body.pid }, 'tok-for-the-hook-test');
|
|
46
|
+
res.writeHead(200, { 'content-type': 'application/json' });
|
|
47
|
+
res.end(JSON.stringify(body));
|
|
48
|
+
});
|
|
49
|
+
return new Promise((resolve) => srv.listen(port, '127.0.0.1', () => {
|
|
50
|
+
t.after(() => new Promise(r => srv.close(r)));
|
|
51
|
+
resolve(srv);
|
|
52
|
+
}));
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
function freePort() {
|
|
56
|
+
return new Promise((resolve) => {
|
|
57
|
+
const s = http.createServer();
|
|
58
|
+
s.listen(0, '127.0.0.1', () => { const p = s.address().port; s.close(() => resolve(p)); });
|
|
59
|
+
});
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
function runHook(dir, tool) {
|
|
63
|
+
const out = execFileSync(process.execPath, [HOOK, tool], {
|
|
64
|
+
encoding: 'utf8', timeout: 20000,
|
|
65
|
+
env: {
|
|
66
|
+
...process.env,
|
|
67
|
+
LLM_SWITCHER_STATE_DIR: dir,
|
|
68
|
+
LLM_SWITCHER_CONFIG: path.join(dir, 'config.json'),
|
|
69
|
+
LLM_SWITCHER_PORT: ''
|
|
70
|
+
}
|
|
71
|
+
});
|
|
72
|
+
return { raw: out, json: JSON.parse(out.trim() || '{}') };
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
test('a tool that is not routed gets no message at all', async (t) => {
|
|
76
|
+
const dir = tmpState(t);
|
|
77
|
+
writeConfig(dir, await freePort());
|
|
78
|
+
// codex has no profile, so its route file is empty. This is the state after `switch off codex`.
|
|
79
|
+
fs.writeFileSync(path.join(dir, 'active.flag'), 'active');
|
|
80
|
+
fs.writeFileSync(path.join(dir, 'route-codex.txt'), '');
|
|
81
|
+
const r = runHook(dir, 'codex');
|
|
82
|
+
assert.equal(r.json.systemMessage, undefined, 'a tool on its official endpoint says nothing');
|
|
83
|
+
});
|
|
84
|
+
|
|
85
|
+
test('a routed tool with a live gateway reports where its traffic goes', async (t) => {
|
|
86
|
+
const dir = tmpState(t);
|
|
87
|
+
const port = await freePort();
|
|
88
|
+
writeConfig(dir, port);
|
|
89
|
+
await fakeGateway(t, port);
|
|
90
|
+
fs.writeFileSync(path.join(dir, 'active.flag'), 'active');
|
|
91
|
+
fs.writeFileSync(path.join(dir, 'route-claude.txt'), 'claude -> cl | intact.example.io | gemini-3.8-flash | 1M\n');
|
|
92
|
+
const r = runHook(dir, 'claude');
|
|
93
|
+
assert.match(r.json.systemMessage, /LLM Switcher/, 'the message names the switcher');
|
|
94
|
+
assert.match(r.json.systemMessage, /intact\.example\.io/, 'and the host the traffic goes to');
|
|
95
|
+
assert.match(r.json.systemMessage, /gemini-3\.8-flash/, 'and the model');
|
|
96
|
+
assert.doesNotMatch(r.json.systemMessage, /sk-secret/, 'and never the API key');
|
|
97
|
+
});
|
|
98
|
+
|
|
99
|
+
test('a routed tool with a dead gateway is warned that it cannot reach the provider', async (t) => {
|
|
100
|
+
const dir = tmpState(t);
|
|
101
|
+
const port = await freePort(); // nothing listens on it
|
|
102
|
+
writeConfig(dir, port);
|
|
103
|
+
fs.writeFileSync(path.join(dir, 'active.flag'), 'active');
|
|
104
|
+
fs.writeFileSync(path.join(dir, 'route-claude.txt'), 'claude -> cl | intact.example.io | gemini-3.8-flash | 1M\n');
|
|
105
|
+
const r = runHook(dir, 'claude');
|
|
106
|
+
assert.match(r.json.systemMessage, /does not answer|not running/i, 'the message says the gateway is down');
|
|
107
|
+
assert.match(r.json.systemMessage, new RegExp(String(port)), 'and names the port');
|
|
108
|
+
assert.match(r.json.systemMessage, /switch on/, 'and names the command that fixes it');
|
|
109
|
+
});
|
|
110
|
+
|
|
111
|
+
test('a broken state never fails the session: exit 0 and parseable output', async (t) => {
|
|
112
|
+
const dir = tmpState(t);
|
|
113
|
+
fs.writeFileSync(path.join(dir, 'config.json'), '{ this is not json', { mode: 0o600 });
|
|
114
|
+
fs.writeFileSync(path.join(dir, 'active.flag'), 'active');
|
|
115
|
+
const r = runHook(dir, 'claude');
|
|
116
|
+
assert.equal(typeof r.json, 'object', 'the tool still gets one JSON object');
|
|
117
|
+
const bad = runHook(dir, 'not-a-tool');
|
|
118
|
+
assert.equal(bad.json.systemMessage, undefined, 'an unknown tool name says nothing');
|
|
119
|
+
});
|
|
@@ -1,205 +1,205 @@
|
|
|
1
|
-
// LIVE test: requires a running gateway (switch on) and a real upstream (costs tokens).
|
|
2
|
-
// Offline test, no network needed: npm test
|
|
3
|
-
const GATEWAY_PORT = Number(process.env.LLM_SWITCHER_PORT) || 3456;
|
|
4
|
-
const BASE_URL = `http://127.0.0.1:${GATEWAY_PORT}`;
|
|
5
|
-
|
|
6
|
-
console.log('================================================================');
|
|
7
|
-
console.log(' LLM SWITCHER — TOKEN OPTIMIZER INTEROPERABILITY TEST SUITE ');
|
|
8
|
-
console.log(' Simulating failure modes from Headroom, RTK, and Ponytail ');
|
|
9
|
-
console.log('================================================================\n');
|
|
10
|
-
|
|
11
|
-
let passed = 0;
|
|
12
|
-
let failed = 0;
|
|
13
|
-
|
|
14
|
-
async function runTest(name, fn) {
|
|
15
|
-
process.stdout.write(`[TEST] ${name}... `);
|
|
16
|
-
const t0 = Date.now();
|
|
17
|
-
try {
|
|
18
|
-
const detail = await fn();
|
|
19
|
-
console.log(`PASS (${Date.now() - t0}ms)`);
|
|
20
|
-
if (detail) console.log(` ↳ ${detail}`);
|
|
21
|
-
passed++;
|
|
22
|
-
} catch (err) {
|
|
23
|
-
console.log(`FAIL (${Date.now() - t0}ms)`);
|
|
24
|
-
console.log(` ↳ ERROR: ${err.message}`);
|
|
25
|
-
failed++;
|
|
26
|
-
}
|
|
27
|
-
}
|
|
28
|
-
|
|
29
|
-
// -----------------------------------------------------------------------------
|
|
30
|
-
// TEST 1: Headroom Failure Mode — Orphaned tool_result
|
|
31
|
-
// When Headroom prunes history to save tokens, it drops the assistant tool_use turn.
|
|
32
|
-
// Anthropic strictly throws: 400 invalid_request_error: 'tool_use_id does not correspond to any tool_use'
|
|
33
|
-
// LLM Switcher Healer Engine: Repairs orphaned result into context text -> 200 OK
|
|
34
|
-
// -----------------------------------------------------------------------------
|
|
35
|
-
await runTest('Headroom Simulation: Orphaned tool_result turn', async () => {
|
|
36
|
-
const payload = {
|
|
37
|
-
model: 'claude-opus-4-6',
|
|
38
|
-
max_tokens: 40,
|
|
39
|
-
messages: [
|
|
40
|
-
{ role: 'user', content: 'Context turn before pruning' },
|
|
41
|
-
{
|
|
42
|
-
role: 'user',
|
|
43
|
-
content: [
|
|
44
|
-
{ type: 'tool_result', tool_use_id: 'headroom_pruned_call_99', content: 'Database query result: 42 rows found' },
|
|
45
|
-
{ type: 'text', text: 'Reply: healed successfully' }
|
|
46
|
-
]
|
|
47
|
-
}
|
|
48
|
-
],
|
|
49
|
-
stream: false
|
|
50
|
-
};
|
|
51
|
-
|
|
52
|
-
const res = await fetch(`${BASE_URL}/v1/messages`, {
|
|
53
|
-
method: 'POST',
|
|
54
|
-
headers: { 'Content-Type': 'application/json', 'anthropic-version': '2023-06-01' },
|
|
55
|
-
body: JSON.stringify(payload)
|
|
56
|
-
});
|
|
57
|
-
|
|
58
|
-
if (res.status !== 200) {
|
|
59
|
-
const err = await res.text();
|
|
60
|
-
throw new Error(`Expected HTTP 200, got ${res.status}: ${err}`);
|
|
61
|
-
}
|
|
62
|
-
const json = await res.json();
|
|
63
|
-
const text = json.content?.map(c => c.text).join('') || '';
|
|
64
|
-
return `Healed orphaned tool_result. Response HTTP 200: "${text.slice(0, 50).replace(/\n/g, ' ')}..."`;
|
|
65
|
-
});
|
|
66
|
-
|
|
67
|
-
// -----------------------------------------------------------------------------
|
|
68
|
-
// TEST 2: Headroom Failure Mode — Consecutive User turns (Roles must alternate)
|
|
69
|
-
// When Headroom collapses history, multiple user turns occur back-to-back.
|
|
70
|
-
// Anthropic strictly throws: 400 invalid_request_error: 'roles must alternate'
|
|
71
|
-
// LLM Switcher Healer Engine: Merges consecutive same-role turns -> 200 OK
|
|
72
|
-
// -----------------------------------------------------------------------------
|
|
73
|
-
await runTest('Headroom Simulation: Consecutive User turns (Role alternation violation)', async () => {
|
|
74
|
-
const payload = {
|
|
75
|
-
model: 'claude-opus-4-6',
|
|
76
|
-
max_tokens: 30,
|
|
77
|
-
messages: [
|
|
78
|
-
{ role: 'user', content: 'User message turn 1' },
|
|
79
|
-
{ role: 'user', content: 'User message turn 2 (no assistant in between)' },
|
|
80
|
-
{ role: 'user', content: 'User message turn 3: reply pong' }
|
|
81
|
-
],
|
|
82
|
-
stream: false
|
|
83
|
-
};
|
|
84
|
-
|
|
85
|
-
const res = await fetch(`${BASE_URL}/v1/messages`, {
|
|
86
|
-
method: 'POST',
|
|
87
|
-
headers: { 'Content-Type': 'application/json', 'anthropic-version': '2023-06-01' },
|
|
88
|
-
body: JSON.stringify(payload)
|
|
89
|
-
});
|
|
90
|
-
|
|
91
|
-
if (res.status !== 200) {
|
|
92
|
-
const err = await res.text();
|
|
93
|
-
throw new Error(`Expected HTTP 200, got ${res.status}: ${err}`);
|
|
94
|
-
}
|
|
95
|
-
const json = await res.json();
|
|
96
|
-
return `Merged consecutive turns seamlessly. Stop reason: ${json.stop_reason}`;
|
|
97
|
-
});
|
|
98
|
-
|
|
99
|
-
// -----------------------------------------------------------------------------
|
|
100
|
-
// TEST 3: Headroom / RTK Failure Mode — Stripped Thinking Parameter
|
|
101
|
-
// Optimizer stripped the 'thinking' object to reduce tokens.
|
|
102
|
-
// LLM Switcher Thinking Guard: Detects reasoning model (ag/claude-opus-4-6-thinking)
|
|
103
|
-
// and restores thinking.budget_tokens automatically -> thinking_delta emitted!
|
|
104
|
-
// -----------------------------------------------------------------------------
|
|
105
|
-
await runTest('Thinking Guard: Restoring stripped thinking parameter on reasoning models', async () => {
|
|
106
|
-
const payload = {
|
|
107
|
-
model: 'claude-opus-4-6',
|
|
108
|
-
max_tokens: 200,
|
|
109
|
-
// Note: NO thinking parameter included (simulating optimizer stripping it)
|
|
110
|
-
messages: [
|
|
111
|
-
{ role: 'user', content: 'Solve step by step: what is 17 * 23? Show reasoning.' }
|
|
112
|
-
],
|
|
113
|
-
stream: true
|
|
114
|
-
};
|
|
115
|
-
|
|
116
|
-
const res = await fetch(`${BASE_URL}/v1/messages`, {
|
|
117
|
-
method: 'POST',
|
|
118
|
-
headers: { 'Content-Type': 'application/json', 'anthropic-version': '2023-06-01' },
|
|
119
|
-
body: JSON.stringify(payload)
|
|
120
|
-
});
|
|
121
|
-
|
|
122
|
-
if (res.status !== 200) {
|
|
123
|
-
throw new Error(`Expected HTTP 200, got ${res.status}`);
|
|
124
|
-
}
|
|
125
|
-
|
|
126
|
-
const text = await res.text();
|
|
127
|
-
const hasThinkingDelta = text.includes('thinking_delta');
|
|
128
|
-
const hasSignatureDelta = text.includes('signature_delta');
|
|
129
|
-
const hasTextDelta = text.includes('text_delta');
|
|
130
|
-
|
|
131
|
-
if (!hasThinkingDelta) {
|
|
132
|
-
throw new Error('Thinking Guard failed: thinking_delta was NOT emitted in stream!');
|
|
133
|
-
}
|
|
134
|
-
|
|
135
|
-
return `Automatically restored thinking: thinking_delta=${hasThinkingDelta}, signature_delta=${hasSignatureDelta}, text_delta=${hasTextDelta}`;
|
|
136
|
-
});
|
|
137
|
-
|
|
138
|
-
// -----------------------------------------------------------------------------
|
|
139
|
-
// TEST 4: RTK & Intermediary Headers Passthrough
|
|
140
|
-
// RTK / tracing tools inject headers: x-rtk-version, traceparent, x-request-id.
|
|
141
|
-
// LLM Switcher: Transparently preserves all tracking headers without rejection.
|
|
142
|
-
// -----------------------------------------------------------------------------
|
|
143
|
-
await runTest('RTK Intermediary: Custom headers and traceparent passthrough', async () => {
|
|
144
|
-
const payload = {
|
|
145
|
-
model: 'claude-opus-4-6',
|
|
146
|
-
max_tokens: 20,
|
|
147
|
-
messages: [{ role: 'user', content: 'ping' }],
|
|
148
|
-
stream: false
|
|
149
|
-
};
|
|
150
|
-
|
|
151
|
-
const res = await fetch(`${BASE_URL}/v1/messages`, {
|
|
152
|
-
method: 'POST',
|
|
153
|
-
headers: {
|
|
154
|
-
'Content-Type': 'application/json',
|
|
155
|
-
'anthropic-version': '2023-06-01',
|
|
156
|
-
'x-rtk-version': '0.37.2',
|
|
157
|
-
'x-optimizer-id': 'rtk-cli-hook',
|
|
158
|
-
'traceparent': '00-4bf92f3577b34da6a3ce929d0e0e4736-00f067aa0ba902b7-01'
|
|
159
|
-
},
|
|
160
|
-
body: JSON.stringify(payload)
|
|
161
|
-
});
|
|
162
|
-
|
|
163
|
-
if (res.status !== 200) {
|
|
164
|
-
throw new Error(`Expected HTTP 200, got ${res.status}`);
|
|
165
|
-
}
|
|
166
|
-
return `Headers accepted cleanly with HTTP 200 OK`;
|
|
167
|
-
});
|
|
168
|
-
|
|
169
|
-
// -----------------------------------------------------------------------------
|
|
170
|
-
// TEST 5: OpenAI Chat Healer Mode — Orphaned Tool Result in Chat Completions
|
|
171
|
-
// Same test but over /v1/chat/completions: tool role message without prior tool_calls assistant message.
|
|
172
|
-
// OpenAI API strictly throws: 400 'messages with role tool must be a response to a preceding message with tool_calls'
|
|
173
|
-
// LLM Switcher: Converts orphaned tool to user text message -> 200 OK
|
|
174
|
-
// -----------------------------------------------------------------------------
|
|
175
|
-
await runTest('OpenAI Chat Healer: Orphaned role tool without preceding assistant tool_calls', async () => {
|
|
176
|
-
const payload = {
|
|
177
|
-
model: 'ag/gpt-oss-120b-medium',
|
|
178
|
-
max_tokens: 30,
|
|
179
|
-
messages: [
|
|
180
|
-
{ role: 'user', content: 'Turn 1 context' },
|
|
181
|
-
{ role: 'tool', tool_call_id: 'orphaned_chat_call_77', content: 'Tool execution logs: OK' },
|
|
182
|
-
{ role: 'user', content: 'reply pong' }
|
|
183
|
-
],
|
|
184
|
-
stream: false
|
|
185
|
-
};
|
|
186
|
-
|
|
187
|
-
const res = await fetch(`${BASE_URL}/v1/chat/completions`, {
|
|
188
|
-
method: 'POST',
|
|
189
|
-
headers: { 'Content-Type': 'application/json' },
|
|
190
|
-
body: JSON.stringify(payload)
|
|
191
|
-
});
|
|
192
|
-
|
|
193
|
-
if (res.status !== 200) {
|
|
194
|
-
const err = await res.text();
|
|
195
|
-
throw new Error(`Expected HTTP 200, got ${res.status}: ${err}`);
|
|
196
|
-
}
|
|
197
|
-
const json = await res.json();
|
|
198
|
-
return `Chat Healer rescued orphaned tool role. Stop reason: ${json.choices?.[0]?.finish_reason}`;
|
|
199
|
-
});
|
|
200
|
-
|
|
201
|
-
console.log('\n================================================================');
|
|
202
|
-
console.log(` TEST RESULTS: ${passed} PASSED / ${failed} FAILED `);
|
|
203
|
-
console.log('================================================================\n');
|
|
204
|
-
|
|
205
|
-
process.exit(failed ? 1 : 0);
|
|
1
|
+
// LIVE test: requires a running gateway (switch on) and a real upstream (costs tokens).
|
|
2
|
+
// Offline test, no network needed: npm test
|
|
3
|
+
const GATEWAY_PORT = Number(process.env.LLM_SWITCHER_PORT) || 3456;
|
|
4
|
+
const BASE_URL = `http://127.0.0.1:${GATEWAY_PORT}`;
|
|
5
|
+
|
|
6
|
+
console.log('================================================================');
|
|
7
|
+
console.log(' LLM SWITCHER — TOKEN OPTIMIZER INTEROPERABILITY TEST SUITE ');
|
|
8
|
+
console.log(' Simulating failure modes from Headroom, RTK, and Ponytail ');
|
|
9
|
+
console.log('================================================================\n');
|
|
10
|
+
|
|
11
|
+
let passed = 0;
|
|
12
|
+
let failed = 0;
|
|
13
|
+
|
|
14
|
+
async function runTest(name, fn) {
|
|
15
|
+
process.stdout.write(`[TEST] ${name}... `);
|
|
16
|
+
const t0 = Date.now();
|
|
17
|
+
try {
|
|
18
|
+
const detail = await fn();
|
|
19
|
+
console.log(`PASS (${Date.now() - t0}ms)`);
|
|
20
|
+
if (detail) console.log(` ↳ ${detail}`);
|
|
21
|
+
passed++;
|
|
22
|
+
} catch (err) {
|
|
23
|
+
console.log(`FAIL (${Date.now() - t0}ms)`);
|
|
24
|
+
console.log(` ↳ ERROR: ${err.message}`);
|
|
25
|
+
failed++;
|
|
26
|
+
}
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
// -----------------------------------------------------------------------------
|
|
30
|
+
// TEST 1: Headroom Failure Mode — Orphaned tool_result
|
|
31
|
+
// When Headroom prunes history to save tokens, it drops the assistant tool_use turn.
|
|
32
|
+
// Anthropic strictly throws: 400 invalid_request_error: 'tool_use_id does not correspond to any tool_use'
|
|
33
|
+
// LLM Switcher Healer Engine: Repairs orphaned result into context text -> 200 OK
|
|
34
|
+
// -----------------------------------------------------------------------------
|
|
35
|
+
await runTest('Headroom Simulation: Orphaned tool_result turn', async () => {
|
|
36
|
+
const payload = {
|
|
37
|
+
model: 'claude-opus-4-6',
|
|
38
|
+
max_tokens: 40,
|
|
39
|
+
messages: [
|
|
40
|
+
{ role: 'user', content: 'Context turn before pruning' },
|
|
41
|
+
{
|
|
42
|
+
role: 'user',
|
|
43
|
+
content: [
|
|
44
|
+
{ type: 'tool_result', tool_use_id: 'headroom_pruned_call_99', content: 'Database query result: 42 rows found' },
|
|
45
|
+
{ type: 'text', text: 'Reply: healed successfully' }
|
|
46
|
+
]
|
|
47
|
+
}
|
|
48
|
+
],
|
|
49
|
+
stream: false
|
|
50
|
+
};
|
|
51
|
+
|
|
52
|
+
const res = await fetch(`${BASE_URL}/v1/messages`, {
|
|
53
|
+
method: 'POST',
|
|
54
|
+
headers: { 'Content-Type': 'application/json', 'anthropic-version': '2023-06-01' },
|
|
55
|
+
body: JSON.stringify(payload)
|
|
56
|
+
});
|
|
57
|
+
|
|
58
|
+
if (res.status !== 200) {
|
|
59
|
+
const err = await res.text();
|
|
60
|
+
throw new Error(`Expected HTTP 200, got ${res.status}: ${err}`);
|
|
61
|
+
}
|
|
62
|
+
const json = await res.json();
|
|
63
|
+
const text = json.content?.map(c => c.text).join('') || '';
|
|
64
|
+
return `Healed orphaned tool_result. Response HTTP 200: "${text.slice(0, 50).replace(/\n/g, ' ')}..."`;
|
|
65
|
+
});
|
|
66
|
+
|
|
67
|
+
// -----------------------------------------------------------------------------
|
|
68
|
+
// TEST 2: Headroom Failure Mode — Consecutive User turns (Roles must alternate)
|
|
69
|
+
// When Headroom collapses history, multiple user turns occur back-to-back.
|
|
70
|
+
// Anthropic strictly throws: 400 invalid_request_error: 'roles must alternate'
|
|
71
|
+
// LLM Switcher Healer Engine: Merges consecutive same-role turns -> 200 OK
|
|
72
|
+
// -----------------------------------------------------------------------------
|
|
73
|
+
await runTest('Headroom Simulation: Consecutive User turns (Role alternation violation)', async () => {
|
|
74
|
+
const payload = {
|
|
75
|
+
model: 'claude-opus-4-6',
|
|
76
|
+
max_tokens: 30,
|
|
77
|
+
messages: [
|
|
78
|
+
{ role: 'user', content: 'User message turn 1' },
|
|
79
|
+
{ role: 'user', content: 'User message turn 2 (no assistant in between)' },
|
|
80
|
+
{ role: 'user', content: 'User message turn 3: reply pong' }
|
|
81
|
+
],
|
|
82
|
+
stream: false
|
|
83
|
+
};
|
|
84
|
+
|
|
85
|
+
const res = await fetch(`${BASE_URL}/v1/messages`, {
|
|
86
|
+
method: 'POST',
|
|
87
|
+
headers: { 'Content-Type': 'application/json', 'anthropic-version': '2023-06-01' },
|
|
88
|
+
body: JSON.stringify(payload)
|
|
89
|
+
});
|
|
90
|
+
|
|
91
|
+
if (res.status !== 200) {
|
|
92
|
+
const err = await res.text();
|
|
93
|
+
throw new Error(`Expected HTTP 200, got ${res.status}: ${err}`);
|
|
94
|
+
}
|
|
95
|
+
const json = await res.json();
|
|
96
|
+
return `Merged consecutive turns seamlessly. Stop reason: ${json.stop_reason}`;
|
|
97
|
+
});
|
|
98
|
+
|
|
99
|
+
// -----------------------------------------------------------------------------
|
|
100
|
+
// TEST 3: Headroom / RTK Failure Mode — Stripped Thinking Parameter
|
|
101
|
+
// Optimizer stripped the 'thinking' object to reduce tokens.
|
|
102
|
+
// LLM Switcher Thinking Guard: Detects reasoning model (ag/claude-opus-4-6-thinking)
|
|
103
|
+
// and restores thinking.budget_tokens automatically -> thinking_delta emitted!
|
|
104
|
+
// -----------------------------------------------------------------------------
|
|
105
|
+
await runTest('Thinking Guard: Restoring stripped thinking parameter on reasoning models', async () => {
|
|
106
|
+
const payload = {
|
|
107
|
+
model: 'claude-opus-4-6',
|
|
108
|
+
max_tokens: 200,
|
|
109
|
+
// Note: NO thinking parameter included (simulating optimizer stripping it)
|
|
110
|
+
messages: [
|
|
111
|
+
{ role: 'user', content: 'Solve step by step: what is 17 * 23? Show reasoning.' }
|
|
112
|
+
],
|
|
113
|
+
stream: true
|
|
114
|
+
};
|
|
115
|
+
|
|
116
|
+
const res = await fetch(`${BASE_URL}/v1/messages`, {
|
|
117
|
+
method: 'POST',
|
|
118
|
+
headers: { 'Content-Type': 'application/json', 'anthropic-version': '2023-06-01' },
|
|
119
|
+
body: JSON.stringify(payload)
|
|
120
|
+
});
|
|
121
|
+
|
|
122
|
+
if (res.status !== 200) {
|
|
123
|
+
throw new Error(`Expected HTTP 200, got ${res.status}`);
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
const text = await res.text();
|
|
127
|
+
const hasThinkingDelta = text.includes('thinking_delta');
|
|
128
|
+
const hasSignatureDelta = text.includes('signature_delta');
|
|
129
|
+
const hasTextDelta = text.includes('text_delta');
|
|
130
|
+
|
|
131
|
+
if (!hasThinkingDelta) {
|
|
132
|
+
throw new Error('Thinking Guard failed: thinking_delta was NOT emitted in stream!');
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
return `Automatically restored thinking: thinking_delta=${hasThinkingDelta}, signature_delta=${hasSignatureDelta}, text_delta=${hasTextDelta}`;
|
|
136
|
+
});
|
|
137
|
+
|
|
138
|
+
// -----------------------------------------------------------------------------
|
|
139
|
+
// TEST 4: RTK & Intermediary Headers Passthrough
|
|
140
|
+
// RTK / tracing tools inject headers: x-rtk-version, traceparent, x-request-id.
|
|
141
|
+
// LLM Switcher: Transparently preserves all tracking headers without rejection.
|
|
142
|
+
// -----------------------------------------------------------------------------
|
|
143
|
+
await runTest('RTK Intermediary: Custom headers and traceparent passthrough', async () => {
|
|
144
|
+
const payload = {
|
|
145
|
+
model: 'claude-opus-4-6',
|
|
146
|
+
max_tokens: 20,
|
|
147
|
+
messages: [{ role: 'user', content: 'ping' }],
|
|
148
|
+
stream: false
|
|
149
|
+
};
|
|
150
|
+
|
|
151
|
+
const res = await fetch(`${BASE_URL}/v1/messages`, {
|
|
152
|
+
method: 'POST',
|
|
153
|
+
headers: {
|
|
154
|
+
'Content-Type': 'application/json',
|
|
155
|
+
'anthropic-version': '2023-06-01',
|
|
156
|
+
'x-rtk-version': '0.37.2',
|
|
157
|
+
'x-optimizer-id': 'rtk-cli-hook',
|
|
158
|
+
'traceparent': '00-4bf92f3577b34da6a3ce929d0e0e4736-00f067aa0ba902b7-01'
|
|
159
|
+
},
|
|
160
|
+
body: JSON.stringify(payload)
|
|
161
|
+
});
|
|
162
|
+
|
|
163
|
+
if (res.status !== 200) {
|
|
164
|
+
throw new Error(`Expected HTTP 200, got ${res.status}`);
|
|
165
|
+
}
|
|
166
|
+
return `Headers accepted cleanly with HTTP 200 OK`;
|
|
167
|
+
});
|
|
168
|
+
|
|
169
|
+
// -----------------------------------------------------------------------------
|
|
170
|
+
// TEST 5: OpenAI Chat Healer Mode — Orphaned Tool Result in Chat Completions
|
|
171
|
+
// Same test but over /v1/chat/completions: tool role message without prior tool_calls assistant message.
|
|
172
|
+
// OpenAI API strictly throws: 400 'messages with role tool must be a response to a preceding message with tool_calls'
|
|
173
|
+
// LLM Switcher: Converts orphaned tool to user text message -> 200 OK
|
|
174
|
+
// -----------------------------------------------------------------------------
|
|
175
|
+
await runTest('OpenAI Chat Healer: Orphaned role tool without preceding assistant tool_calls', async () => {
|
|
176
|
+
const payload = {
|
|
177
|
+
model: 'ag/gpt-oss-120b-medium',
|
|
178
|
+
max_tokens: 30,
|
|
179
|
+
messages: [
|
|
180
|
+
{ role: 'user', content: 'Turn 1 context' },
|
|
181
|
+
{ role: 'tool', tool_call_id: 'orphaned_chat_call_77', content: 'Tool execution logs: OK' },
|
|
182
|
+
{ role: 'user', content: 'reply pong' }
|
|
183
|
+
],
|
|
184
|
+
stream: false
|
|
185
|
+
};
|
|
186
|
+
|
|
187
|
+
const res = await fetch(`${BASE_URL}/v1/chat/completions`, {
|
|
188
|
+
method: 'POST',
|
|
189
|
+
headers: { 'Content-Type': 'application/json' },
|
|
190
|
+
body: JSON.stringify(payload)
|
|
191
|
+
});
|
|
192
|
+
|
|
193
|
+
if (res.status !== 200) {
|
|
194
|
+
const err = await res.text();
|
|
195
|
+
throw new Error(`Expected HTTP 200, got ${res.status}: ${err}`);
|
|
196
|
+
}
|
|
197
|
+
const json = await res.json();
|
|
198
|
+
return `Chat Healer rescued orphaned tool role. Stop reason: ${json.choices?.[0]?.finish_reason}`;
|
|
199
|
+
});
|
|
200
|
+
|
|
201
|
+
console.log('\n================================================================');
|
|
202
|
+
console.log(` TEST RESULTS: ${passed} PASSED / ${failed} FAILED `);
|
|
203
|
+
console.log('================================================================\n');
|
|
204
|
+
|
|
205
|
+
process.exit(failed ? 1 : 0);
|