llm-switcher 1.2.8 → 1.2.10

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,205 +1,205 @@
1
- // LIVE test: requires a running gateway (switch on) and a real upstream (costs tokens).
2
- // Offline test, no network needed: npm test
3
- const GATEWAY_PORT = Number(process.env.LLM_SWITCHER_PORT) || 3456;
4
- const BASE_URL = `http://127.0.0.1:${GATEWAY_PORT}`;
5
-
6
- console.log('================================================================');
7
- console.log(' LLM SWITCHER — TOKEN OPTIMIZER INTEROPERABILITY TEST SUITE ');
8
- console.log(' Simulating failure modes from Headroom, RTK, and Ponytail ');
9
- console.log('================================================================\n');
10
-
11
- let passed = 0;
12
- let failed = 0;
13
-
14
- async function runTest(name, fn) {
15
- process.stdout.write(`[TEST] ${name}... `);
16
- const t0 = Date.now();
17
- try {
18
- const detail = await fn();
19
- console.log(`PASS (${Date.now() - t0}ms)`);
20
- if (detail) console.log(` ↳ ${detail}`);
21
- passed++;
22
- } catch (err) {
23
- console.log(`FAIL (${Date.now() - t0}ms)`);
24
- console.log(` ↳ ERROR: ${err.message}`);
25
- failed++;
26
- }
27
- }
28
-
29
- // -----------------------------------------------------------------------------
30
- // TEST 1: Headroom Failure Mode — Orphaned tool_result
31
- // When Headroom prunes history to save tokens, it drops the assistant tool_use turn.
32
- // Anthropic strictly throws: 400 invalid_request_error: 'tool_use_id does not correspond to any tool_use'
33
- // LLM Switcher Healer Engine: Repairs orphaned result into context text -> 200 OK
34
- // -----------------------------------------------------------------------------
35
- await runTest('Headroom Simulation: Orphaned tool_result turn', async () => {
36
- const payload = {
37
- model: 'claude-opus-4-6',
38
- max_tokens: 40,
39
- messages: [
40
- { role: 'user', content: 'Context turn before pruning' },
41
- {
42
- role: 'user',
43
- content: [
44
- { type: 'tool_result', tool_use_id: 'headroom_pruned_call_99', content: 'Database query result: 42 rows found' },
45
- { type: 'text', text: 'Reply: healed successfully' }
46
- ]
47
- }
48
- ],
49
- stream: false
50
- };
51
-
52
- const res = await fetch(`${BASE_URL}/v1/messages`, {
53
- method: 'POST',
54
- headers: { 'Content-Type': 'application/json', 'anthropic-version': '2023-06-01' },
55
- body: JSON.stringify(payload)
56
- });
57
-
58
- if (res.status !== 200) {
59
- const err = await res.text();
60
- throw new Error(`Expected HTTP 200, got ${res.status}: ${err}`);
61
- }
62
- const json = await res.json();
63
- const text = json.content?.map(c => c.text).join('') || '';
64
- return `Healed orphaned tool_result. Response HTTP 200: "${text.slice(0, 50).replace(/\n/g, ' ')}..."`;
65
- });
66
-
67
- // -----------------------------------------------------------------------------
68
- // TEST 2: Headroom Failure Mode — Consecutive User turns (Roles must alternate)
69
- // When Headroom collapses history, multiple user turns occur back-to-back.
70
- // Anthropic strictly throws: 400 invalid_request_error: 'roles must alternate'
71
- // LLM Switcher Healer Engine: Merges consecutive same-role turns -> 200 OK
72
- // -----------------------------------------------------------------------------
73
- await runTest('Headroom Simulation: Consecutive User turns (Role alternation violation)', async () => {
74
- const payload = {
75
- model: 'claude-opus-4-6',
76
- max_tokens: 30,
77
- messages: [
78
- { role: 'user', content: 'User message turn 1' },
79
- { role: 'user', content: 'User message turn 2 (no assistant in between)' },
80
- { role: 'user', content: 'User message turn 3: reply pong' }
81
- ],
82
- stream: false
83
- };
84
-
85
- const res = await fetch(`${BASE_URL}/v1/messages`, {
86
- method: 'POST',
87
- headers: { 'Content-Type': 'application/json', 'anthropic-version': '2023-06-01' },
88
- body: JSON.stringify(payload)
89
- });
90
-
91
- if (res.status !== 200) {
92
- const err = await res.text();
93
- throw new Error(`Expected HTTP 200, got ${res.status}: ${err}`);
94
- }
95
- const json = await res.json();
96
- return `Merged consecutive turns seamlessly. Stop reason: ${json.stop_reason}`;
97
- });
98
-
99
- // -----------------------------------------------------------------------------
100
- // TEST 3: Headroom / RTK Failure Mode — Stripped Thinking Parameter
101
- // Optimizer stripped the 'thinking' object to reduce tokens.
102
- // LLM Switcher Thinking Guard: Detects reasoning model (ag/claude-opus-4-6-thinking)
103
- // and restores thinking.budget_tokens automatically -> thinking_delta emitted!
104
- // -----------------------------------------------------------------------------
105
- await runTest('Thinking Guard: Restoring stripped thinking parameter on reasoning models', async () => {
106
- const payload = {
107
- model: 'claude-opus-4-6',
108
- max_tokens: 200,
109
- // Note: NO thinking parameter included (simulating optimizer stripping it)
110
- messages: [
111
- { role: 'user', content: 'Solve step by step: what is 17 * 23? Show reasoning.' }
112
- ],
113
- stream: true
114
- };
115
-
116
- const res = await fetch(`${BASE_URL}/v1/messages`, {
117
- method: 'POST',
118
- headers: { 'Content-Type': 'application/json', 'anthropic-version': '2023-06-01' },
119
- body: JSON.stringify(payload)
120
- });
121
-
122
- if (res.status !== 200) {
123
- throw new Error(`Expected HTTP 200, got ${res.status}`);
124
- }
125
-
126
- const text = await res.text();
127
- const hasThinkingDelta = text.includes('thinking_delta');
128
- const hasSignatureDelta = text.includes('signature_delta');
129
- const hasTextDelta = text.includes('text_delta');
130
-
131
- if (!hasThinkingDelta) {
132
- throw new Error('Thinking Guard failed: thinking_delta was NOT emitted in stream!');
133
- }
134
-
135
- return `Automatically restored thinking: thinking_delta=${hasThinkingDelta}, signature_delta=${hasSignatureDelta}, text_delta=${hasTextDelta}`;
136
- });
137
-
138
- // -----------------------------------------------------------------------------
139
- // TEST 4: RTK & Intermediary Headers Passthrough
140
- // RTK / tracing tools inject headers: x-rtk-version, traceparent, x-request-id.
141
- // LLM Switcher: Transparently preserves all tracking headers without rejection.
142
- // -----------------------------------------------------------------------------
143
- await runTest('RTK Intermediary: Custom headers and traceparent passthrough', async () => {
144
- const payload = {
145
- model: 'claude-opus-4-6',
146
- max_tokens: 20,
147
- messages: [{ role: 'user', content: 'ping' }],
148
- stream: false
149
- };
150
-
151
- const res = await fetch(`${BASE_URL}/v1/messages`, {
152
- method: 'POST',
153
- headers: {
154
- 'Content-Type': 'application/json',
155
- 'anthropic-version': '2023-06-01',
156
- 'x-rtk-version': '0.37.2',
157
- 'x-optimizer-id': 'rtk-cli-hook',
158
- 'traceparent': '00-4bf92f3577b34da6a3ce929d0e0e4736-00f067aa0ba902b7-01'
159
- },
160
- body: JSON.stringify(payload)
161
- });
162
-
163
- if (res.status !== 200) {
164
- throw new Error(`Expected HTTP 200, got ${res.status}`);
165
- }
166
- return `Headers accepted cleanly with HTTP 200 OK`;
167
- });
168
-
169
- // -----------------------------------------------------------------------------
170
- // TEST 5: OpenAI Chat Healer Mode — Orphaned Tool Result in Chat Completions
171
- // Same test but over /v1/chat/completions: tool role message without prior tool_calls assistant message.
172
- // OpenAI API strictly throws: 400 'messages with role tool must be a response to a preceding message with tool_calls'
173
- // LLM Switcher: Converts orphaned tool to user text message -> 200 OK
174
- // -----------------------------------------------------------------------------
175
- await runTest('OpenAI Chat Healer: Orphaned role tool without preceding assistant tool_calls', async () => {
176
- const payload = {
177
- model: 'ag/gpt-oss-120b-medium',
178
- max_tokens: 30,
179
- messages: [
180
- { role: 'user', content: 'Turn 1 context' },
181
- { role: 'tool', tool_call_id: 'orphaned_chat_call_77', content: 'Tool execution logs: OK' },
182
- { role: 'user', content: 'reply pong' }
183
- ],
184
- stream: false
185
- };
186
-
187
- const res = await fetch(`${BASE_URL}/v1/chat/completions`, {
188
- method: 'POST',
189
- headers: { 'Content-Type': 'application/json' },
190
- body: JSON.stringify(payload)
191
- });
192
-
193
- if (res.status !== 200) {
194
- const err = await res.text();
195
- throw new Error(`Expected HTTP 200, got ${res.status}: ${err}`);
196
- }
197
- const json = await res.json();
198
- return `Chat Healer rescued orphaned tool role. Stop reason: ${json.choices?.[0]?.finish_reason}`;
199
- });
200
-
201
- console.log('\n================================================================');
202
- console.log(` TEST RESULTS: ${passed} PASSED / ${failed} FAILED `);
203
- console.log('================================================================\n');
204
-
205
- process.exit(failed ? 1 : 0);
1
+ // LIVE test: requires a running gateway (switch on) and a real upstream (costs tokens).
2
+ // Offline test, no network needed: npm test
3
+ const GATEWAY_PORT = Number(process.env.LLM_SWITCHER_PORT) || 3456;
4
+ const BASE_URL = `http://127.0.0.1:${GATEWAY_PORT}`;
5
+
6
+ console.log('================================================================');
7
+ console.log(' LLM SWITCHER — TOKEN OPTIMIZER INTEROPERABILITY TEST SUITE ');
8
+ console.log(' Simulating failure modes from Headroom, RTK, and Ponytail ');
9
+ console.log('================================================================\n');
10
+
11
+ let passed = 0;
12
+ let failed = 0;
13
+
14
+ async function runTest(name, fn) {
15
+ process.stdout.write(`[TEST] ${name}... `);
16
+ const t0 = Date.now();
17
+ try {
18
+ const detail = await fn();
19
+ console.log(`PASS (${Date.now() - t0}ms)`);
20
+ if (detail) console.log(` ↳ ${detail}`);
21
+ passed++;
22
+ } catch (err) {
23
+ console.log(`FAIL (${Date.now() - t0}ms)`);
24
+ console.log(` ↳ ERROR: ${err.message}`);
25
+ failed++;
26
+ }
27
+ }
28
+
29
+ // -----------------------------------------------------------------------------
30
+ // TEST 1: Headroom Failure Mode — Orphaned tool_result
31
+ // When Headroom prunes history to save tokens, it drops the assistant tool_use turn.
32
+ // Anthropic strictly throws: 400 invalid_request_error: 'tool_use_id does not correspond to any tool_use'
33
+ // LLM Switcher Healer Engine: Repairs orphaned result into context text -> 200 OK
34
+ // -----------------------------------------------------------------------------
35
+ await runTest('Headroom Simulation: Orphaned tool_result turn', async () => {
36
+ const payload = {
37
+ model: 'claude-opus-4-6',
38
+ max_tokens: 40,
39
+ messages: [
40
+ { role: 'user', content: 'Context turn before pruning' },
41
+ {
42
+ role: 'user',
43
+ content: [
44
+ { type: 'tool_result', tool_use_id: 'headroom_pruned_call_99', content: 'Database query result: 42 rows found' },
45
+ { type: 'text', text: 'Reply: healed successfully' }
46
+ ]
47
+ }
48
+ ],
49
+ stream: false
50
+ };
51
+
52
+ const res = await fetch(`${BASE_URL}/v1/messages`, {
53
+ method: 'POST',
54
+ headers: { 'Content-Type': 'application/json', 'anthropic-version': '2023-06-01' },
55
+ body: JSON.stringify(payload)
56
+ });
57
+
58
+ if (res.status !== 200) {
59
+ const err = await res.text();
60
+ throw new Error(`Expected HTTP 200, got ${res.status}: ${err}`);
61
+ }
62
+ const json = await res.json();
63
+ const text = json.content?.map(c => c.text).join('') || '';
64
+ return `Healed orphaned tool_result. Response HTTP 200: "${text.slice(0, 50).replace(/\n/g, ' ')}..."`;
65
+ });
66
+
67
+ // -----------------------------------------------------------------------------
68
+ // TEST 2: Headroom Failure Mode — Consecutive User turns (Roles must alternate)
69
+ // When Headroom collapses history, multiple user turns occur back-to-back.
70
+ // Anthropic strictly throws: 400 invalid_request_error: 'roles must alternate'
71
+ // LLM Switcher Healer Engine: Merges consecutive same-role turns -> 200 OK
72
+ // -----------------------------------------------------------------------------
73
+ await runTest('Headroom Simulation: Consecutive User turns (Role alternation violation)', async () => {
74
+ const payload = {
75
+ model: 'claude-opus-4-6',
76
+ max_tokens: 30,
77
+ messages: [
78
+ { role: 'user', content: 'User message turn 1' },
79
+ { role: 'user', content: 'User message turn 2 (no assistant in between)' },
80
+ { role: 'user', content: 'User message turn 3: reply pong' }
81
+ ],
82
+ stream: false
83
+ };
84
+
85
+ const res = await fetch(`${BASE_URL}/v1/messages`, {
86
+ method: 'POST',
87
+ headers: { 'Content-Type': 'application/json', 'anthropic-version': '2023-06-01' },
88
+ body: JSON.stringify(payload)
89
+ });
90
+
91
+ if (res.status !== 200) {
92
+ const err = await res.text();
93
+ throw new Error(`Expected HTTP 200, got ${res.status}: ${err}`);
94
+ }
95
+ const json = await res.json();
96
+ return `Merged consecutive turns seamlessly. Stop reason: ${json.stop_reason}`;
97
+ });
98
+
99
+ // -----------------------------------------------------------------------------
100
+ // TEST 3: Headroom / RTK Failure Mode — Stripped Thinking Parameter
101
+ // Optimizer stripped the 'thinking' object to reduce tokens.
102
+ // LLM Switcher Thinking Guard: Detects reasoning model (ag/claude-opus-4-6-thinking)
103
+ // and restores thinking.budget_tokens automatically -> thinking_delta emitted!
104
+ // -----------------------------------------------------------------------------
105
+ await runTest('Thinking Guard: Restoring stripped thinking parameter on reasoning models', async () => {
106
+ const payload = {
107
+ model: 'claude-opus-4-6',
108
+ max_tokens: 200,
109
+ // Note: NO thinking parameter included (simulating optimizer stripping it)
110
+ messages: [
111
+ { role: 'user', content: 'Solve step by step: what is 17 * 23? Show reasoning.' }
112
+ ],
113
+ stream: true
114
+ };
115
+
116
+ const res = await fetch(`${BASE_URL}/v1/messages`, {
117
+ method: 'POST',
118
+ headers: { 'Content-Type': 'application/json', 'anthropic-version': '2023-06-01' },
119
+ body: JSON.stringify(payload)
120
+ });
121
+
122
+ if (res.status !== 200) {
123
+ throw new Error(`Expected HTTP 200, got ${res.status}`);
124
+ }
125
+
126
+ const text = await res.text();
127
+ const hasThinkingDelta = text.includes('thinking_delta');
128
+ const hasSignatureDelta = text.includes('signature_delta');
129
+ const hasTextDelta = text.includes('text_delta');
130
+
131
+ if (!hasThinkingDelta) {
132
+ throw new Error('Thinking Guard failed: thinking_delta was NOT emitted in stream!');
133
+ }
134
+
135
+ return `Automatically restored thinking: thinking_delta=${hasThinkingDelta}, signature_delta=${hasSignatureDelta}, text_delta=${hasTextDelta}`;
136
+ });
137
+
138
+ // -----------------------------------------------------------------------------
139
+ // TEST 4: RTK & Intermediary Headers Passthrough
140
+ // RTK / tracing tools inject headers: x-rtk-version, traceparent, x-request-id.
141
+ // LLM Switcher: Transparently preserves all tracking headers without rejection.
142
+ // -----------------------------------------------------------------------------
143
+ await runTest('RTK Intermediary: Custom headers and traceparent passthrough', async () => {
144
+ const payload = {
145
+ model: 'claude-opus-4-6',
146
+ max_tokens: 20,
147
+ messages: [{ role: 'user', content: 'ping' }],
148
+ stream: false
149
+ };
150
+
151
+ const res = await fetch(`${BASE_URL}/v1/messages`, {
152
+ method: 'POST',
153
+ headers: {
154
+ 'Content-Type': 'application/json',
155
+ 'anthropic-version': '2023-06-01',
156
+ 'x-rtk-version': '0.37.2',
157
+ 'x-optimizer-id': 'rtk-cli-hook',
158
+ 'traceparent': '00-4bf92f3577b34da6a3ce929d0e0e4736-00f067aa0ba902b7-01'
159
+ },
160
+ body: JSON.stringify(payload)
161
+ });
162
+
163
+ if (res.status !== 200) {
164
+ throw new Error(`Expected HTTP 200, got ${res.status}`);
165
+ }
166
+ return `Headers accepted cleanly with HTTP 200 OK`;
167
+ });
168
+
169
+ // -----------------------------------------------------------------------------
170
+ // TEST 5: OpenAI Chat Healer Mode — Orphaned Tool Result in Chat Completions
171
+ // Same test but over /v1/chat/completions: tool role message without prior tool_calls assistant message.
172
+ // OpenAI API strictly throws: 400 'messages with role tool must be a response to a preceding message with tool_calls'
173
+ // LLM Switcher: Converts orphaned tool to user text message -> 200 OK
174
+ // -----------------------------------------------------------------------------
175
+ await runTest('OpenAI Chat Healer: Orphaned role tool without preceding assistant tool_calls', async () => {
176
+ const payload = {
177
+ model: 'ag/gpt-oss-120b-medium',
178
+ max_tokens: 30,
179
+ messages: [
180
+ { role: 'user', content: 'Turn 1 context' },
181
+ { role: 'tool', tool_call_id: 'orphaned_chat_call_77', content: 'Tool execution logs: OK' },
182
+ { role: 'user', content: 'reply pong' }
183
+ ],
184
+ stream: false
185
+ };
186
+
187
+ const res = await fetch(`${BASE_URL}/v1/chat/completions`, {
188
+ method: 'POST',
189
+ headers: { 'Content-Type': 'application/json' },
190
+ body: JSON.stringify(payload)
191
+ });
192
+
193
+ if (res.status !== 200) {
194
+ const err = await res.text();
195
+ throw new Error(`Expected HTTP 200, got ${res.status}: ${err}`);
196
+ }
197
+ const json = await res.json();
198
+ return `Chat Healer rescued orphaned tool role. Stop reason: ${json.choices?.[0]?.finish_reason}`;
199
+ });
200
+
201
+ console.log('\n================================================================');
202
+ console.log(` TEST RESULTS: ${passed} PASSED / ${failed} FAILED `);
203
+ console.log('================================================================\n');
204
+
205
+ process.exit(failed ? 1 : 0);
package/ui.html CHANGED
@@ -930,7 +930,7 @@
930
930
  <div class="ph">
931
931
  <div class="title-box">
932
932
  <h1 id="page-title">Routes & Profiles</h1>
933
- <p id="status-sub">Route Claude Code, Codex, and OpenAI clients independently.</p>
933
+ <p id="status-sub">Route Claude Code and Codex independently.</p>
934
934
  </div>
935
935
  <div class="ph-actions" id="ph-actions">
936
936
  <!-- Rendered dynamically per route -->
@@ -1 +0,0 @@
1
- {"version":1,"sessions":{"1d508082-a9d5-487c-8568-c4ba91785a8d":{"updatedAt":1790431452442,"files":{"C:\\Users\\louis\\tools\\llm-switcher\\ui.html":{"editCount":23,"findings":[],"cleanAcked":true}},"footerShown":true},"673c932f-b92d-4af4-8899-3c8e73082f67":{"updatedAt":1790689606133,"files":{"C:\\Users\\louis\\tools\\llm-switcher\\ui.html":{"editCount":7,"findings":["bounce-easing:0:cubic-bezier(0.2, 0.9, 0.3, 1.2)","bounce-easing:0:cubic-bezier(0.3, 1.4, 0.6, 1)"],"cleanAcked":true}},"footerShown":true}}}