llm-switcher 1.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. package/.gitattributes +16 -0
  2. package/LICENSE +21 -0
  3. package/README.md +587 -0
  4. package/README.vi.md +585 -0
  5. package/blindfold/blindfold.mjs +633 -0
  6. package/blindfold/make-certs.sh +88 -0
  7. package/blindfold/wsframe.mjs +176 -0
  8. package/codex-catalog-template.json +1 -0
  9. package/config.example.json +84 -0
  10. package/contract-exclusions.json +41 -0
  11. package/contract.mjs +561 -0
  12. package/docs/LLM-RESPONSE-MATRIX.md +165 -0
  13. package/docs/TOKEN-OPTIMIZER-INTEROP.md +110 -0
  14. package/docs/codex-blindfold.md +214 -0
  15. package/docs/cross-platform.md +136 -0
  16. package/docs/diagrams/blindfold-request-routing.html +14972 -0
  17. package/docs/diagrams/blindfold-request-routing.sequence.json +175 -0
  18. package/docs/diagrams/blindfold-switch-lifecycle.html +14958 -0
  19. package/docs/diagrams/blindfold-switch-lifecycle.lifecycle.json +159 -0
  20. package/docs/diagrams/codex-model-name-resolution.html +15005 -0
  21. package/docs/diagrams/codex-model-name-resolution.workflow.json +71 -0
  22. package/docs/response-matrix.json +1131 -0
  23. package/formats.mjs +2308 -0
  24. package/mcp.mjs +340 -0
  25. package/package.json +36 -0
  26. package/proxy.mjs +1743 -0
  27. package/service.mjs +132 -0
  28. package/shim.mjs +292 -0
  29. package/skills/llm-switcher/SKILL.md +88 -0
  30. package/state.mjs +978 -0
  31. package/switch +5 -0
  32. package/switch.cmd +2 -0
  33. package/switch.mjs +930 -0
  34. package/tests/blindfold.test.mjs +307 -0
  35. package/tests/blindfold.wire.test.mjs +170 -0
  36. package/tests/contract/run.test.mjs +214 -0
  37. package/tests/contract-check.test.mjs +458 -0
  38. package/tests/contract-lab.test.mjs +755 -0
  39. package/tests/datadir.test.mjs +37 -0
  40. package/tests/formats.test.mjs +794 -0
  41. package/tests/gateway.e2e.test.mjs +999 -0
  42. package/tests/helpers.mjs +24 -0
  43. package/tests/lifecycle.test.mjs +416 -0
  44. package/tests/live-optimizer-interop.mjs +205 -0
  45. package/tests/mcp.test.mjs +91 -0
  46. package/tests/service.test.mjs +69 -0
  47. package/tests/shim.test.mjs +228 -0
  48. package/tests/state.test.mjs +675 -0
  49. package/tests/switch.test.mjs +156 -0
  50. package/tests/wsframe.test.mjs +154 -0
  51. package/ui.html +2234 -0
@@ -0,0 +1,999 @@
1
+ // End-to-end tests: spawn proxy.mjs against a mock upstream (offline, no real API key needed).
2
+ // Run: node --test tests/
3
+ import { test, before, after } from 'node:test';
4
+ import assert from 'node:assert/strict';
5
+ import http from 'node:http';
6
+ import net from 'node:net';
7
+ import fs from 'node:fs';
8
+ import crypto from 'node:crypto';
9
+ import os from 'node:os';
10
+ import path from 'node:path';
11
+ import { spawn } from 'node:child_process';
12
+ import zlib from 'node:zlib';
13
+ import { fileURLToPath } from 'node:url';
14
+ import { assertValidAnthropicEvents } from './helpers.mjs';
15
+ import { createFrameReader } from '../blindfold/wsframe.mjs';
16
+
17
+ const MASKED = '__LLM_SWITCHER_KEEP_KEY__';
18
+
19
+ const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..');
20
+
21
+ let upstream, upstreamPort, proxy, proxyPort, tmpDir;
22
+ const received = []; // { url, headers, body }
23
+ const hangState = { closed: false, slowAborted: false };
24
+ const bigState = { finishedAt: 0 };
25
+
26
+ function freePort() {
27
+ return new Promise((resolve, reject) => {
28
+ const s = http.createServer();
29
+ s.listen(0, '127.0.0.1', () => {
30
+ const { port } = s.address();
31
+ s.close(() => resolve(port));
32
+ });
33
+ s.on('error', reject);
34
+ });
35
+ }
36
+
37
+ const sse = (res, objs, { raw = [] } = {}) => {
38
+ res.writeHead(200, { 'Content-Type': 'text/event-stream' });
39
+ for (const o of objs) res.write(`data: ${JSON.stringify(o)}\n\n`);
40
+ for (const r of raw) res.write(r);
41
+ res.end();
42
+ };
43
+
44
+ function startUpstream() {
45
+ upstream = http.createServer((req, res) => {
46
+ let body = '';
47
+ req.on('data', c => { body += c; });
48
+ req.on('end', () => {
49
+ const json = body ? JSON.parse(body) : {};
50
+ received.push({ url: req.url, headers: req.headers, body: json });
51
+ const lastUser = JSON.stringify(json.messages?.at(-1) ?? json.contents?.at(-1) ?? '');
52
+
53
+ if (req.url.startsWith('/chat/v1/chat/completions')) {
54
+ if (lastUser.includes('RATE_LIMIT')) {
55
+ res.writeHead(429, { 'Content-Type': 'application/json', 'retry-after': '7' });
56
+ return res.end(JSON.stringify({ error: { message: 'slow down' } }));
57
+ }
58
+ if (lastUser.includes('APPLY_PATCH')) {
59
+ return sse(res, [
60
+ { choices: [{ index: 0, delta: { tool_calls: [{ index: 0, id: 'call_patch', type: 'function', function: { name: 'apply_patch', arguments: '{"input":"*** Begin Patch\\n' } }] } }] },
61
+ { choices: [{ index: 0, delta: { tool_calls: [{ index: 0, function: { arguments: '*** End Patch"}' } }] } }] },
62
+ { choices: [{ index: 0, delta: {}, finish_reason: 'tool_calls' }] }
63
+ ]);
64
+ }
65
+ if (lastUser.includes('ERROR_THEN_HANG')) {
66
+ res.writeHead(200, { 'Content-Type': 'text/event-stream' });
67
+ res.write(`data: ${JSON.stringify({ error: { message: 'in-band failure' } })}\n\n`);
68
+ res.on('close', () => { hangState.closed = true; });
69
+ return;
70
+ }
71
+ if (lastUser.includes('SLOW_TURN')) {
72
+ res.on('close', () => { if (!res.writableEnded) hangState.slowAborted = true; });
73
+ res.writeHead(200, { 'Content-Type': 'text/event-stream' });
74
+ res.write(`data: ${JSON.stringify({ choices: [{ index: 0, delta: { content: 'slow ' } }] })}\n\n`);
75
+ return setTimeout(() => {
76
+ res.write(`data: ${JSON.stringify({ choices: [{ index: 0, delta: { content: 'done' }, finish_reason: 'stop' }] })}\n\n`);
77
+ res.end();
78
+ }, 300);
79
+ }
80
+ if (lastUser.includes('MID_STREAM_ERROR')) {
81
+ return sse(res, [
82
+ { choices: [{ index: 0, delta: { content: 'partial' } }] },
83
+ { error: { message: 'upstream exploded' } }
84
+ ]);
85
+ }
86
+ if (!json.stream) {
87
+ return res.end(JSON.stringify({
88
+ choices: [{ index: 0, message: { role: 'assistant', content: '<think>quiet plan</think>Final answer' }, finish_reason: 'stop' }],
89
+ usage: { prompt_tokens: 50, completion_tokens: 4 }
90
+ }));
91
+ }
92
+ return sse(res, [
93
+ { choices: [{ index: 0, delta: { role: 'assistant', reasoning_content: 'Let me think' } }] },
94
+ { choices: [{ index: 0, delta: { content: 'Running tool' } }] },
95
+ { choices: [{ index: 0, delta: { tool_calls: [{ index: 0, id: 'call_A', type: 'function', function: { name: 'read_file', arguments: '{"pa' } }] } }] },
96
+ { choices: [{ index: 0, delta: { tool_calls: [{ index: 0, function: { arguments: 'th":"a.txt"}' } }] } }] },
97
+ { choices: [{ index: 0, delta: {}, finish_reason: 'tool_calls' }] },
98
+ { choices: [], usage: { prompt_tokens: 1234, completion_tokens: 56, prompt_tokens_details: { cached_tokens: 1000 } } }
99
+ ], { raw: ['data: [DONE]\n\n'] });
100
+ }
101
+
102
+ if (req.url.startsWith('/vtx/models/')) {
103
+ return sse(res, [
104
+ { candidates: [{ content: { role: 'model', parts: [{ text: 'thinking about it', thought: true }] } }] },
105
+ { candidates: [{ content: { role: 'model', parts: [{ functionCall: { name: 'get_weather', args: { city: 'Hanoi' } }, thoughtSignature: 'SIG_HANOI' }] } }] },
106
+ { candidates: [{ content: { role: 'model', parts: [{ functionCall: { name: 'get_weather', args: { city: 'Saigon' } } }] }, finishReason: 'STOP' }], usageMetadata: { promptTokenCount: 20, candidatesTokenCount: 9 } }
107
+ ]);
108
+ }
109
+
110
+ if (req.url.startsWith('/ant/messages/count_tokens')) {
111
+ res.writeHead(200, { 'Content-Type': 'application/json' });
112
+ return res.end(JSON.stringify({ input_tokens: 4242 }));
113
+ }
114
+
115
+ const antSse = (events) => events.map(([e, d]) => `event: ${e}\ndata: ${JSON.stringify(d)}\n\n`).join('');
116
+ const antStart = ['message_start', { type: 'message_start', message: { id: 'm', type: 'message', role: 'assistant', model: json.model, content: [], usage: { input_tokens: 11, output_tokens: 1 } } }];
117
+ const antEnd = [['message_delta', { type: 'message_delta', delta: { stop_reason: 'end_turn' }, usage: { output_tokens: 7 } }], ['message_stop', { type: 'message_stop' }]];
118
+ if (req.url.startsWith('/ant/messages') && lastUser.includes('DIRECT_STREAM')) {
119
+ res.writeHead(200, { 'Content-Type': 'text/event-stream' });
120
+ return res.end(antSse([antStart, ...antEnd]));
121
+ }
122
+ if (req.url.startsWith('/ant/messages') && lastUser.includes('DIRECT_BREAK')) {
123
+ res.writeHead(200, { 'Content-Type': 'text/event-stream' });
124
+ res.write(antSse([antStart]));
125
+ return setTimeout(() => res.destroy(), 50);
126
+ }
127
+ if (req.url.startsWith('/ant/messages') && lastUser.includes('DIRECT_SLOW')) {
128
+ res.writeHead(200, { 'Content-Type': 'text/event-stream' });
129
+ res.write(antSse([antStart]));
130
+ return setTimeout(() => res.end(antSse(antEnd)), 600);
131
+ }
132
+ if (req.url.startsWith('/ant/messages') && lastUser.includes('DIRECT_BIG')) {
133
+ // 32 MB, written with this server's own backpressure; bigState records when the last byte left.
134
+ res.writeHead(200, { 'Content-Type': 'text/event-stream' });
135
+ const chunk = Buffer.alloc(64 * 1024, 'a');
136
+ let left = 512;
137
+ const pump = () => {
138
+ while (left > 0) {
139
+ left--;
140
+ if (!res.write(chunk)) return res.once('drain', pump);
141
+ }
142
+ res.end(() => { bigState.finishedAt = Date.now(); });
143
+ };
144
+ return pump();
145
+ }
146
+ if (req.url.startsWith('/ant/messages')) {
147
+ res.writeHead(200, { 'Content-Type': 'application/json', 'transfer-encoding': 'chunked', connection: 'keep-alive' });
148
+ return res.end(JSON.stringify({ id: 'msg_1', type: 'message', role: 'assistant', model: json.model, content: [{ type: 'text', text: 'direct ok' }], stop_reason: 'end_turn', usage: { input_tokens: 3, output_tokens: 2 } }));
149
+ }
150
+
151
+ res.writeHead(404);
152
+ res.end('{}');
153
+ });
154
+ });
155
+ return new Promise(r => upstream.listen(0, '127.0.0.1', () => { upstreamPort = upstream.address().port; r(); }));
156
+ }
157
+
158
+ before(async () => {
159
+ await startUpstream();
160
+ proxyPort = await freePort();
161
+ tmpDir = fs.mkdtempSync(path.join(os.tmpdir(), 'llm-switcher-test-'));
162
+ const base = `http://127.0.0.1:${upstreamPort}`;
163
+ const models = { opus: 'up-opus', sonnet: 'up-sonnet', haiku: 'up-haiku', fable: 'up-fable' };
164
+ const cfg = {
165
+ port: proxyPort,
166
+ activeProfile: 'chat',
167
+ activeProfiles: { anthropic: 'chat', responses: 'chat', 'openai-chat': 'chat', vertex: 'chat' },
168
+ profiles: {
169
+ chat: { name: 'Mock Chat', mode: 'convert', inFormat: 'auto', outFormat: 'openai-chat', baseURL: `${base}/chat/v1`, apiKey: 'sk-secret-chat', defaultModels: models },
170
+ vtx: { name: 'Mock Vertex', mode: 'convert', inFormat: 'auto', outFormat: 'vertex', baseURL: `${base}/vtx`, apiKey: 'sk-secret-vtx', defaultModels: models },
171
+ agmock: { name: 'Mock AG via chat', mode: 'convert', inFormat: 'responses', baseURL: `${base}/chat/v1`, apiKey: 'sk-secret-ag', defaultModels: { main: 'ag/mock-flash', review: 'ag/mock-review', subagent: 'ag/mock-low' } },
172
+ pub: { name: 'Mock Public', mode: 'convert', inFormat: 'responses', baseURL: `${base}/chat/v1`, apiKey: 'sk-secret-pub', publicModels: ['gpt-5.6-sol', 'gpt-5.2'], defaultModels: { main: 'ag/mock-flash' }, model1M: { main: true } },
173
+ roles: { name: 'Mock Public Roles', mode: 'convert', inFormat: 'responses', baseURL: `${base}/chat/v1`, apiKey: 'sk-secret-roles', publicModels: ['gpt-5.6-sol', 'gpt-5.6-terra', 'gpt-5.6-luna'], defaultModels: { main: 'ag/mock-flash', review: 'ag/mock-review', subagent: 'ag/mock-low' } },
174
+ native: { name: 'Mock Strict OpenAI', mode: 'convert', inFormat: 'auto', outFormat: 'openai-chat', thinkingMode: 'native', baseURL: `${base}/chat/v1`, apiKey: 'sk-secret-native', defaultModels: models },
175
+ ant: { name: 'Mock Anthropic', mode: 'direct', inFormat: 'auto', outFormat: 'anthropic', baseURL: `${base}/ant`, apiKey: 'sk-secret-ant', defaultModels: models }
176
+ }
177
+ };
178
+ fs.writeFileSync(path.join(tmpDir, 'config.json'), JSON.stringify(cfg, null, 2));
179
+ proxy = spawn(process.execPath, [path.join(ROOT, 'proxy.mjs'), '--port', String(proxyPort)], {
180
+ env: { ...process.env, LLM_SWITCHER_CONFIG: path.join(tmpDir, 'config.json'), LLM_SWITCHER_STATE_DIR: tmpDir, CLAUDE_CONFIG_DIR: path.join(tmpDir, 'claude'), LLM_SWITCHER_PORT: '' },
181
+ stdio: ['ignore', 'pipe', 'pipe']
182
+ });
183
+ let log = '';
184
+ proxy.stdout.on('data', d => { log += d; });
185
+ proxy.stderr.on('data', d => { log += d; });
186
+ for (let i = 0; i < 50; i++) {
187
+ try {
188
+ const r = await fetch(`http://127.0.0.1:${proxyPort}/health`);
189
+ if (r.ok) return;
190
+ } catch {}
191
+ await new Promise(r => setTimeout(r, 100));
192
+ }
193
+ throw new Error(`proxy did not start:\n${log}`);
194
+ });
195
+
196
+ after(() => {
197
+ proxy?.kill();
198
+ upstream?.close();
199
+ if (tmpDir) fs.rmSync(tmpDir, { recursive: true, force: true });
200
+ });
201
+
202
+ const url = (p) => `http://127.0.0.1:${proxyPort}${p}`;
203
+ // The admin API requires the per-install token that the gateway writes next to config.json.
204
+ const adminToken = () => fs.readFileSync(path.join(tmpDir, 'admin.token'), 'utf8').trim();
205
+ const withToken = (p, headers) => (p.startsWith('/api/') ? { 'x-llm-switcher-token': adminToken(), ...headers } : headers);
206
+ const post = (p, body, headers = {}) => fetch(url(p), { method: 'POST', headers: withToken(p, { 'Content-Type': 'application/json', ...headers }), body: JSON.stringify(body) });
207
+
208
+ function parseSSE(text) {
209
+ return text.split(/\n\n/).filter(Boolean).map(block => {
210
+ let event = null;
211
+ let data = '';
212
+ for (const line of block.split('\n')) {
213
+ if (line.startsWith('event:')) event = line.slice(6).trim();
214
+ else if (line.startsWith('data:')) data += line.slice(5).trim();
215
+ }
216
+ return { event, raw: data, data: data && data !== '[DONE]' ? JSON.parse(data) : null };
217
+ });
218
+ }
219
+
220
+ function rawRequest({ path: p, method = 'GET', headers = {} }) {
221
+ return new Promise((resolve, reject) => {
222
+ const req = http.request({ host: '127.0.0.1', port: proxyPort, path: p, method, headers }, res => {
223
+ let data = '';
224
+ res.on('data', c => { data += c; });
225
+ res.on('end', () => resolve({ status: res.statusCode, body: data }));
226
+ });
227
+ req.on('error', reject);
228
+ req.end();
229
+ });
230
+ }
231
+
232
+ test('Claude Code stream via OpenAI upstream: valid Anthropic events, full tool args, real usage', async () => {
233
+ const res = await post('/v1/messages', {
234
+ model: 'claude-opus-4-6', max_tokens: 4096, stream: true,
235
+ messages: [{ role: 'user', content: 'read a.txt' }],
236
+ tools: [{ name: 'read_file', description: 'r', input_schema: { type: 'object', properties: { path: { type: 'string' } } } }]
237
+ });
238
+ assert.equal(res.status, 200);
239
+ const events = parseSSE(await res.text()).map(e => ({ event: e.event, data: e.data }));
240
+ assertValidAnthropicEvents(events);
241
+ const starts = events.filter(e => e.event === 'content_block_start').map(e => e.data.content_block.type);
242
+ assert.deepEqual(starts, ['thinking', 'text', 'tool_use']);
243
+ const toolIdx = events.find(e => e.data?.content_block?.type === 'tool_use').data.index;
244
+ const args = events.filter(e => e.event === 'content_block_delta' && e.data.index === toolIdx).map(e => e.data.delta.partial_json).join('');
245
+ assert.deepEqual(JSON.parse(args), { path: 'a.txt' });
246
+ const md = events.find(e => e.event === 'message_delta').data;
247
+ assert.equal(md.delta.stop_reason, 'tool_use');
248
+ assert.equal(md.usage.output_tokens, 56);
249
+ assert.equal(md.usage.input_tokens + md.usage.cache_read_input_tokens, 1234);
250
+ assert.equal(received.at(-1).body.model, 'up-opus');
251
+ });
252
+
253
+ test('Anthropic request without "stream" gets JSON (not SSE); <think> tags become a thinking block', async () => {
254
+ const res = await post('/v1/messages', { model: 'claude-haiku-4-5', max_tokens: 100, messages: [{ role: 'user', content: 'hi' }] });
255
+ assert.match(res.headers.get('content-type'), /application\/json/);
256
+ const json = await res.json();
257
+ assert.equal(received.at(-1).body.stream, false);
258
+ assert.deepEqual(json.content.map(b => b.type), ['thinking', 'text']);
259
+ assert.equal(json.content[0].thinking, 'quiet plan');
260
+ assert.equal(json.content[1].text, 'Final answer');
261
+ });
262
+
263
+ test('Codex (Responses) stream: output_item.done carries function_call; no [DONE] line', async () => {
264
+ const res = await post('/v1/responses', { model: 'gpt-5-codex', stream: true, input: 'go', tools: [{ type: 'function', name: 'read_file', parameters: { type: 'object' } }] });
265
+ const text = await res.text();
266
+ assert.ok(!text.includes('[DONE]'));
267
+ const events = parseSSE(text);
268
+ assert.ok(events.every(e => e.event === e.data.type), 'event: line matches data.type');
269
+ const done = events.filter(e => e.event === 'response.output_item.done').map(e => e.data.item);
270
+ const fc = done.find(i => i.type === 'function_call');
271
+ assert.equal(fc.call_id, 'call_A');
272
+ assert.deepEqual(JSON.parse(fc.arguments), { path: 'a.txt' });
273
+ assert.equal(events.at(-1).event, 'response.completed');
274
+ assert.equal(events.at(-1).data.response.usage.input_tokens, 1234);
275
+ });
276
+
277
+ test('Codex via ag/* target: Gemini-hostile tool schemas are rewritten before upstream', async () => {
278
+ const res = await post('/v1/responses', {
279
+ model: 'main', stream: false,
280
+ input: 'clean my tools',
281
+ tools: [{ type: 'function', name: 'gmail_x', parameters: {
282
+ type: 'object',
283
+ properties: {
284
+ ids: { anyOf: [{ type: 'array', items: { type: 'string' } }, { type: 'null' }] },
285
+ part: { $ref: '#/$defs/P' },
286
+ raw: 'object'
287
+ },
288
+ required: ['ids', 'nope'],
289
+ $defs: { P: { type: 'object', properties: { t: { type: 'string' } } } }
290
+ } }]
291
+ }, { 'x-llm-profile': 'agmock' });
292
+ assert.equal(res.status, 200);
293
+ const up = received.at(-1);
294
+ assert.equal(up.body.model, 'ag/mock-flash');
295
+ const params = up.body.tools[0].function.parameters;
296
+ assert.deepEqual(params.properties.ids, { type: 'array', items: { type: 'string' }, nullable: true });
297
+ assert.deepEqual(params.properties.part, { type: 'object', properties: { t: { type: 'string' } } });
298
+ assert.deepEqual(params.properties.raw, { type: 'object', properties: {} });
299
+ assert.deepEqual(params.required, ['ids']);
300
+ assert.ok(!JSON.stringify(params).includes('$ref'), 'no $ref survives');
301
+ assert.equal(params.$defs, undefined);
302
+ });
303
+
304
+ test('Codex bare OpenAI model IDs fail closed to the main slot (no 404 passthrough)', async () => {
305
+ const res = await post('/v1/responses', { model: 'gpt-5.6-sol', stream: false, input: 'RATE_LIMIT probe' }, { 'x-llm-profile': 'agmock' });
306
+ assert.equal(received.at(-1).body.model, 'ag/mock-flash');
307
+ assert.equal(res.status, 429);
308
+ assert.equal((await res.json()).error.message, 'slow down');
309
+ });
310
+
311
+ // Once Codex is told the official names, those names must still reach the right
312
+ // slot. Without this the blanket gpt-* fail-closed rule sends review and subagent
313
+ // traffic to the main model.
314
+ test('Official public names resolve to their own slot, not to the main fail-closed slot', async () => {
315
+ await post('/v1/responses', { model: 'gpt-5.6-terra', stream: false, input: 'go' }, { 'x-llm-profile': 'roles' });
316
+ assert.equal(received.at(-1).body.model, 'ag/mock-review');
317
+ await post('/v1/responses', { model: 'gpt-5.6-luna', stream: false, input: 'go' }, { 'x-llm-profile': 'roles' });
318
+ assert.equal(received.at(-1).body.model, 'ag/mock-low');
319
+ // An official name the profile does not publish still fails closed to main.
320
+ await post('/v1/responses', { model: 'gpt-5.1-codex-max', stream: false, input: 'go' }, { 'x-llm-profile': 'roles' });
321
+ assert.equal(received.at(-1).body.model, 'ag/mock-flash');
322
+ });
323
+
324
+ // The upgrade handler is a second entrance to the gateway. Node emits 'upgrade', not
325
+ // 'request', so route() and its checkRequestOrigin never run there. Without this guard
326
+ // any web page can open ws://127.0.0.1:<port>/v1/responses and spend the profile key:
327
+ // browsers do not apply same-origin to WebSocket.
328
+ function rawHandshake(headers) {
329
+ return new Promise((resolve, reject) => {
330
+ const socket = net.connect(proxyPort, '127.0.0.1', () => {
331
+ const lines = ['GET /v1/responses HTTP/1.1', ...headers,
332
+ 'Upgrade: websocket', 'Connection: Upgrade', 'Sec-WebSocket-Key: dGhlIHNhbXBsZSBub25jZQ==',
333
+ 'Sec-WebSocket-Version: 13'];
334
+ socket.write(lines.join('\r\n') + '\r\n\r\n');
335
+ });
336
+ let data = '';
337
+ socket.setTimeout(5000, () => { socket.destroy(); resolve(data); });
338
+ socket.on('data', c => {
339
+ data += c;
340
+ if (data.includes('\r\n\r\n')) { socket.destroy(); resolve(data); }
341
+ });
342
+ socket.on('error', reject);
343
+ socket.on('close', () => resolve(data));
344
+ });
345
+ }
346
+
347
+ test('WS upgrade refuses a foreign Origin and a foreign Host, and still accepts loopback', async () => {
348
+ const evilOrigin = await rawHandshake([`Host: 127.0.0.1:${proxyPort}`, 'Origin: https://evil.example']);
349
+ assert.ok(!evilOrigin.includes('101'), `foreign Origin must not get a 101: ${evilOrigin.slice(0, 80)}`);
350
+
351
+ const evilHost = await rawHandshake([`Host: evil.example:${proxyPort}`]);
352
+ assert.ok(!evilHost.includes('101'), `foreign Host must not get a 101: ${evilHost.slice(0, 80)}`);
353
+
354
+ // An absent Origin stays allowed: Codex sends none, and blindfold deletes it.
355
+ const ok = await rawHandshake([`Host: 127.0.0.1:${proxyPort}`]);
356
+ assert.match(ok, /HTTP\/1\.1 101/, 'a loopback handshake must still succeed');
357
+ });
358
+
359
+ // The dashboard writes these keys, and a hand-edited config reaches the same sink.
360
+ // state.mjs already drops an unsafe name before it can reach env.cmd; rejecting it
361
+ // here as well tells the user why, instead of losing the value in silence.
362
+ test('save-profile rejects a model name that could act as a command', async () => {
363
+ const bad = await post('/api/save-profile', {
364
+ key: 'probe', profile: { name: 'p', baseURL: 'http://127.0.0.1:1/v1', publicModels: ['a & echo pwned'] }
365
+ });
366
+ assert.equal(bad.status, 400);
367
+ assert.match((await bad.json()).error, /publicModels/);
368
+
369
+ const badRole = await post('/api/save-profile', {
370
+ key: 'probe', profile: { name: 'p', baseURL: 'http://127.0.0.1:1/v1', codexRoles: { review: 'x"&y' } }
371
+ });
372
+ assert.equal(badRole.status, 400);
373
+ assert.match((await badRole.json()).error, /codexRoles/);
374
+
375
+ const badPort = await post('/api/save-profile', {
376
+ key: 'probe', profile: { name: 'p', baseURL: 'http://127.0.0.1:1/v1', blindfoldPort: 70000 }
377
+ });
378
+ assert.equal(badPort.status, 400);
379
+ assert.match((await badPort.json()).error, /blindfoldPort/);
380
+
381
+ const badHost = await post('/api/save-profile', {
382
+ key: 'probe', profile: { name: 'p', baseURL: 'http://127.0.0.1:1/v1', blindfoldHost: 'not a host/' }
383
+ });
384
+ assert.equal(badHost.status, 400);
385
+ assert.match((await badHost.json()).error, /blindfoldHost/);
386
+
387
+ // The shapes the dashboard actually sends stay valid, empty strings included.
388
+ const ok = await post('/api/save-profile', {
389
+ key: 'probe', profile: {
390
+ name: 'p', baseURL: 'http://127.0.0.1:1/v1', inFormat: 'responses',
391
+ publicModels: ['gpt-5.6-sol', 'ag/mock-flash'],
392
+ codexRoles: { main: 'gpt-5.6-sol', review: '', subagent: '' },
393
+ blindfold: true, blindfoldHost: 'chatgpt.com', blindfoldPort: 3457, blindfoldPrefix: '/backend-api/codex'
394
+ }
395
+ });
396
+ assert.equal(ok.status, 200);
397
+ });
398
+
399
+ test('Codex WS transport: upstream 429 becomes response.failed with rate_limit_exceeded', async () => { const ws = new WebSocket(`ws://127.0.0.1:${proxyPort}/v1/responses`);
400
+ await new Promise((resolve, reject) => {
401
+ const t = setTimeout(() => reject(new Error('ws open timeout')), 5000);
402
+ ws.addEventListener('open', () => { clearTimeout(t); resolve(); }, { once: true });
403
+ ws.addEventListener('error', () => { clearTimeout(t); reject(new Error('ws open error')); }, { once: true });
404
+ });
405
+ const seen = [];
406
+ const failedP = new Promise((resolve, reject) => {
407
+ const t = setTimeout(() => reject(new Error('no response.failed, got: ' + JSON.stringify(seen.map(s => s.type)))), 15000);
408
+ ws.addEventListener('message', (ev) => {
409
+ const msg = JSON.parse(String(ev.data));
410
+ seen.push(msg);
411
+ if (msg.type === 'response.failed') { clearTimeout(t); resolve(msg); }
412
+ });
413
+ });
414
+ ws.send(JSON.stringify({ type: 'response.create', model: 'main', input: 'RATE_LIMIT over ws' }));
415
+ const failed = await failedP;
416
+ ws.close();
417
+ assert.deepEqual(seen.slice(0, 2).map(s => s.type), ['response.created', 'response.in_progress']);
418
+ assert.equal(failed.response.status, 'failed');
419
+ assert.equal(failed.response.error.code, 'rate_limit_exceeded');
420
+ assert.match(failed.response.error.message, /slow down/);
421
+ });
422
+
423
+ test('Public catalog: /v1/models serves official names with no switcher branding', async () => {
424
+ const res = await fetch(url('/v1/models'), { headers: { 'x-llm-profile': 'pub' } });
425
+ assert.equal(res.status, 200);
426
+ const text = await res.text();
427
+ assert.ok(!text.includes('llm-switcher'), 'no switcher branding leaks to the client');
428
+ assert.ok(!text.includes('ag/'), 'no upstream IDs leak to the client');
429
+ const json = JSON.parse(text);
430
+ assert.deepEqual(json.data.map(m => m.id), ['gpt-5.6-sol', 'gpt-5.2']);
431
+ assert.ok(json.models.every(m => m.slug && m.display_name));
432
+ });
433
+
434
+ test('Response events echo the requested model, never the mapped upstream ID', async () => {
435
+ const res = await post('/v1/responses', { model: 'main', stream: true, input: 'go' }, { 'x-llm-profile': 'agmock' });
436
+ assert.equal(received.at(-1).body.model, 'ag/mock-flash');
437
+ const events = parseSSE(await res.text());
438
+ const created = events.find(e => e.event === 'response.created');
439
+ assert.equal(created.data.response.model, 'main');
440
+ assert.equal(events.at(-1).data.response.model, 'main');
441
+ });
442
+
443
+ test('Model list exposes only real mapped IDs, no slot aliases', async () => {
444
+ const res = await fetch(url('/v1/models'));
445
+ assert.equal(res.status, 200);
446
+ const json = await res.json();
447
+ const ids = json.data.map(m => m.id);
448
+ assert.deepEqual([...ids].sort(), ['up-fable', 'up-haiku', 'up-opus', 'up-sonnet']);
449
+ assert.ok(!ids.some(id => ['main', 'review', 'subagent', 'opus', 'sonnet', 'haiku', 'fable'].includes(id)), 'no slot aliases, got: ' + ids.join(','));
450
+ assert.ok(json.models.every(m => m.slug && m.display_name));
451
+ });
452
+
453
+ test('Vertex upstream: multiple functionCall chunks become separate tool_use blocks', async () => {
454
+ const res = await post('/v1/messages', { model: 'claude-sonnet-4-6', max_tokens: 2048, stream: true, messages: [{ role: 'user', content: 'weather x2' }] }, { 'x-llm-profile': 'vtx' });
455
+ const events = parseSSE(await res.text()).map(e => ({ event: e.event, data: e.data }));
456
+ assertValidAnthropicEvents(events);
457
+ const tools = events.filter(e => e.data?.content_block?.type === 'tool_use');
458
+ assert.equal(tools.length, 2);
459
+ const argsOf = (idx) => JSON.parse(events.filter(e => e.event === 'content_block_delta' && e.data.index === idx).map(e => e.data.delta.partial_json).join(''));
460
+ assert.deepEqual(argsOf(tools[0].data.index), { city: 'Hanoi' });
461
+ assert.deepEqual(argsOf(tools[1].data.index), { city: 'Saigon' });
462
+ assert.match(received.at(-1).url, /:streamGenerateContent\?alt=sse$/);
463
+ });
464
+
465
+ test('Gemini SDK route: model comes from the URL path', async () => {
466
+ const res = await post('/v1beta/models/claude-opus-like:generateContent', { contents: [{ role: 'user', parts: [{ text: 'hi' }] }] });
467
+ assert.equal(res.status, 200);
468
+ const json = await res.json();
469
+ assert.equal(received.at(-1).body.model, 'up-opus');
470
+ assert.ok(json.candidates[0].content.parts.some(p => p.text === 'Final answer'));
471
+ });
472
+
473
+ test('Upstream 429 is returned in Anthropic error shape with retry-after', async () => {
474
+ const res = await post('/v1/messages', { model: 'claude-opus-4-6', max_tokens: 10, messages: [{ role: 'user', content: 'RATE_LIMIT' }] });
475
+ assert.equal(res.status, 429);
476
+ assert.equal(res.headers.get('retry-after'), '7');
477
+ const json = await res.json();
478
+ assert.deepEqual(json, { type: 'error', error: { type: 'rate_limit_error', message: 'slow down' } });
479
+ });
480
+
481
+ test('Mid-stream upstream error surfaces as an error event instead of a fake end_turn', async () => {
482
+ const res = await post('/v1/messages', { model: 'claude-opus-4-6', max_tokens: 10, stream: true, messages: [{ role: 'user', content: 'MID_STREAM_ERROR' }] });
483
+ const events = parseSSE(await res.text());
484
+ assert.equal(events.at(-1).event, 'error');
485
+ assert.match(events.at(-1).data.error.message, /exploded/);
486
+ assert.ok(!events.some(e => e.event === 'message_stop'));
487
+ });
488
+
489
+ test('Direct Anthropic passthrough strips hop-by-hop headers and blocks client credentials', async () => {
490
+ const res = await post('/v1/messages', { model: 'claude-opus-4-6', max_tokens: 10, messages: [{ role: 'user', content: 'hi' }] },
491
+ { 'x-llm-profile': 'ant', 'x-goog-api-key': 'client-google-key', 'x-request-id': 'trace-1' });
492
+ assert.equal(res.status, 200);
493
+ assert.equal((await res.json()).content[0].text, 'direct ok');
494
+ const up = received.at(-1);
495
+ assert.equal(up.headers['x-api-key'], 'sk-secret-ant');
496
+ assert.equal(up.headers['x-goog-api-key'], undefined);
497
+ assert.equal(up.headers['x-request-id'], 'trace-1');
498
+ assert.equal(up.body.model, 'up-opus');
499
+ });
500
+
501
+ test('Healer: orphaned tool_result and missing tool_result produce a valid chat history', async () => {
502
+ await post('/v1/messages', {
503
+ model: 'claude-opus-4-6', max_tokens: 50, messages: [
504
+ { role: 'user', content: [{ type: 'tool_result', tool_use_id: 'pruned', content: 'old' }, { type: 'text', text: 'hi' }] },
505
+ { role: 'assistant', content: [{ type: 'tool_use', id: 'kept', name: 'f', input: {} }] },
506
+ { role: 'user', content: 'result was pruned' }
507
+ ]
508
+ });
509
+ const msgs = received.at(-1).body.messages;
510
+ msgs.forEach((m, i) => {
511
+ if (m.tool_calls) {
512
+ const ids = m.tool_calls.map(t => t.id);
513
+ const following = msgs.slice(i + 1, i + 1 + ids.length);
514
+ assert.deepEqual(following.map(f => f.tool_call_id), ids);
515
+ }
516
+ if (m.role === 'tool') assert.ok(msgs[i - 1].tool_calls || msgs[i - 1].role === 'tool');
517
+ });
518
+ });
519
+
520
+ test('Security: foreign Host / Origin are rejected (DNS rebinding & CSRF)', async () => {
521
+ const rebinding = await rawRequest({ path: '/api/status', headers: { host: `evil.example:${proxyPort}` } });
522
+ assert.equal(rebinding.status, 403);
523
+ const csrf = await rawRequest({ path: '/api/status', headers: { origin: 'https://evil.example' } });
524
+ assert.equal(csrf.status, 403);
525
+ const otherLocalApp = await rawRequest({ path: '/api/status', headers: { origin: 'http://localhost:5173' } });
526
+ assert.equal(otherLocalApp.status, 403);
527
+ const ok = await rawRequest({ path: '/api/status', headers: { origin: `http://127.0.0.1:${proxyPort}`, 'x-llm-switcher-token': adminToken() } });
528
+ assert.equal(ok.status, 200);
529
+ });
530
+
531
+ // Any local process can reach loopback. Without a token it must get nothing from /api/*,
532
+ // and a masked key must never be resolved for a baseURL the profile does not have.
533
+ test('Security: the admin API refuses a caller without the token and changes nothing', async () => {
534
+ const hits = [];
535
+ const sink = http.createServer((req, res) => { hits.push(req.headers); res.writeHead(200, { 'content-type': 'application/json' }); res.end('{"data":[]}'); });
536
+ await new Promise(r => sink.listen(0, '127.0.0.1', r));
537
+ const sinkURL = `http://127.0.0.1:${sink.address().port}`;
538
+ try {
539
+ const before = fs.readFileSync(path.join(tmpDir, 'config.json'), 'utf8');
540
+ const calls = [
541
+ ['GET', '/api/status'], ['GET', '/api/logs'], ['POST', '/api/logs/clear', {}],
542
+ ['POST', '/api/switch', { profile: 'chat' }], ['POST', '/api/toggle', { enabled: false }],
543
+ ['POST', '/api/save-profile', { key: 'chat', profile: { baseURL: sinkURL, apiKey: MASKED } }],
544
+ ['POST', '/api/delete-profile', { key: 'chat' }],
545
+ ['POST', '/api/test-upstream', { key: 'chat', apiKey: MASKED, baseURL: sinkURL }],
546
+ ['POST', '/api/fetch-models', { key: 'chat', apiKey: MASKED, baseURL: sinkURL }]
547
+ ];
548
+ for (const [method, p, body] of calls) {
549
+ for (const token of [undefined, 'wrong-token']) {
550
+ const headers = { 'Content-Type': 'application/json', ...(token ? { 'x-llm-switcher-token': token } : {}) };
551
+ const r = await fetch(url(p), { method, headers, body: body ? JSON.stringify(body) : undefined });
552
+ assert.equal(r.status, 401, `${method} ${p} token=${token}`);
553
+ }
554
+ }
555
+ assert.equal(fs.readFileSync(path.join(tmpDir, 'config.json'), 'utf8'), before, 'config.json is unchanged');
556
+ assert.equal(hits.length, 0, 'no request reached the sink');
557
+
558
+ // With the token, a masked key still stays home when the baseURL is not the stored one.
559
+ await post('/api/fetch-models', { key: 'chat', apiKey: MASKED, baseURL: sinkURL });
560
+ assert.equal(hits.length, 1);
561
+ assert.ok(!JSON.stringify(hits[0]).includes('sk-secret-chat'), 'the stored key is not sent to a foreign baseURL');
562
+
563
+ for (const p of ['/', '/ui']) {
564
+ const page = await (await fetch(url(p))).text();
565
+ assert.ok(!page.includes(adminToken()), `${p} must not embed the token`);
566
+ }
567
+ assert.equal((await fetch(url('/health'))).status, 200, '/health needs no token');
568
+ assert.equal((fs.statSync(path.join(tmpDir, 'admin.token')).mode & 0o777).toString(8), '600');
569
+ } finally {
570
+ sink.close();
571
+ }
572
+ });
573
+
574
+ test('Admin API: keys are redacted and invalid switch input is rejected', async () => {
575
+ const status = await (await fetch(url('/api/status'), { headers: withToken('/api/status', {}) })).json();
576
+ assert.ok(!JSON.stringify(status).includes('sk-secret'));
577
+ assert.equal(status.config.profiles.chat.hasApiKey, true);
578
+
579
+ for (const body of [{ target: '__proto__', profile: 'chat' }, { target: 'anthropic', profile: 'constructor' }, { profile: 'toString' }]) {
580
+ const r = await post('/api/switch', body);
581
+ assert.equal(r.status, 400, JSON.stringify(body));
582
+ }
583
+ const badKey = await post('/api/save-profile', { key: '__proto__', profile: { name: 'x', baseURL: 'http://a' } });
584
+ assert.equal(badKey.status, 400);
585
+ });
586
+
587
+ test('Direct Anthropic path runs the native healer (orphan result, placeholder thinking)', async () => {
588
+ const res = await post('/v1/messages', {
589
+ model: 'claude-opus-4-6', max_tokens: 4096, thinking: { type: 'enabled', budget_tokens: 2048 }, messages: [
590
+ { role: 'user', content: 'go' },
591
+ { role: 'assistant', content: [{ type: 'thinking', thinking: 'converted', signature: 'reasoning-sig' }, { type: 'tool_use', id: 't1', name: 'f', input: {} }] },
592
+ { role: 'user', content: [{ type: 'text', text: 'pruned result' }] }
593
+ ]
594
+ }, { 'x-llm-profile': 'ant' });
595
+ assert.equal(res.status, 200);
596
+ await res.text();
597
+ const body = received.at(-1).body;
598
+ assert.ok(!JSON.stringify(body).includes('reasoning-sig'));
599
+ assert.equal(body.messages[2].content[0].type, 'tool_result');
600
+ assert.equal(body.thinking, undefined);
601
+ });
602
+
603
+ test('count_tokens: forwarded to native Anthropic upstream, estimated otherwise', async () => {
604
+ const payload = { model: 'claude-opus-4-6', messages: [{ role: 'user', content: 'x'.repeat(4000) }] };
605
+ const real = await (await post('/v1/messages/count_tokens', payload, { 'x-llm-profile': 'ant' })).json();
606
+ assert.equal(real.input_tokens, 4242);
607
+ assert.equal(received.at(-1).body.model, 'up-opus');
608
+ const est = await (await post('/v1/messages/count_tokens', payload)).json();
609
+ assert.ok(est.input_tokens >= 1000 && est.input_tokens < 1010, String(est.input_tokens));
610
+ });
611
+
612
+ test('thinkingMode native profile: strict OpenAI body', async () => {
613
+ const res = await post('/v1/messages', { model: 'claude-opus-4-6', max_tokens: 5000, system: 'SYS', thinking: { type: 'enabled', budget_tokens: 3000 }, messages: [{ role: 'user', content: 'hi' }] }, { 'x-llm-profile': 'native' });
614
+ assert.equal(res.status, 200);
615
+ await res.json();
616
+ const body = received.at(-1).body;
617
+ assert.equal(body.thinking, undefined);
618
+ assert.equal(body.reasoning_effort, 'medium');
619
+ assert.equal(body.max_completion_tokens, 5000);
620
+ assert.equal(body.messages[0].content, 'SYS');
621
+ });
622
+
623
+ test('Codex freeform apply_patch round-trips as custom_tool_call', async () => {
624
+ const res = await post('/v1/responses', {
625
+ model: 'gpt-5-codex', stream: true, input: 'APPLY_PATCH please',
626
+ tools: [{ type: 'custom', name: 'apply_patch', description: 'Edit files', format: { type: 'grammar', syntax: 'lark', definition: 'start: begin_patch hunk+ end_patch' } }]
627
+ });
628
+ const events = parseSSE(await res.text());
629
+ const upTool = received.at(-1).body.tools[0].function;
630
+ assert.equal(upTool.name, 'apply_patch');
631
+ assert.deepEqual(upTool.parameters.required, ['input']);
632
+ const item = events.find(e => e.event === 'response.output_item.done').data.item;
633
+ assert.equal(item.type, 'custom_tool_call');
634
+ assert.equal(item.call_id, 'call_patch');
635
+ assert.equal(item.input, '*** Begin Patch\n*** End Patch');
636
+ });
637
+
638
+ test('Gemini thought signature survives a Claude Code round-trip through the gateway', async () => {
639
+ const first = await post('/v1/messages', { model: 'claude-sonnet-4-6', max_tokens: 2048, stream: true, messages: [{ role: 'user', content: 'weather x2' }] }, { 'x-llm-profile': 'vtx' });
640
+ const events = parseSSE(await first.text());
641
+ const toolUses = events.filter(e => e.data?.content_block?.type === 'tool_use').map(e => e.data.content_block);
642
+ assert.equal(toolUses.length, 2);
643
+
644
+ await (await post('/v1/messages', {
645
+ model: 'claude-sonnet-4-6', max_tokens: 2048, stream: true, messages: [
646
+ { role: 'user', content: 'weather x2' },
647
+ { role: 'assistant', content: toolUses.map(t => ({ type: 'tool_use', id: t.id, name: t.name, input: {} })) },
648
+ { role: 'user', content: toolUses.map(t => ({ type: 'tool_result', tool_use_id: t.id, content: '{"temp":30}' })) }
649
+ ]
650
+ }, { 'x-llm-profile': 'vtx' })).text();
651
+ const body = received.at(-1).body;
652
+ assert.equal(body.contents[1].parts[0].thoughtSignature, 'SIG_HANOI');
653
+ assert.equal(body.contents[2].role, 'user');
654
+ assert.deepEqual(body.contents[2].parts.map(p => p.functionResponse.name), ['get_weather', 'get_weather']);
655
+ });
656
+
657
+ // A hand edit with a syntax error must not be overwritten by the gateway's cached copy.
658
+ test('Admin API refuses to write while config.json does not parse, and keeps the hand edit after repair', async () => {
659
+ const cfgPath = path.join(tmpDir, 'config.json');
660
+ const good = fs.readFileSync(cfgPath, 'utf8');
661
+ const sha = (t) => crypto.createHash('sha256').update(t).digest('hex');
662
+ try {
663
+ await new Promise(r => setTimeout(r, 20));
664
+ fs.writeFileSync(cfgPath, '{ invalid');
665
+ const broken = fs.readFileSync(cfgPath, 'utf8');
666
+ for (const [p, body] of [['/api/switch', { profile: 'chat' }], ['/api/save-profile', { key: 'chat', profile: { name: 'x' } }]]) {
667
+ const r = await post(p, body);
668
+ assert.ok(r.status >= 400, `${p} must refuse, got ${r.status}`);
669
+ assert.match((await r.json()).error, /config\.json/);
670
+ }
671
+ assert.equal(sha(fs.readFileSync(cfgPath, 'utf8')), sha(broken), 'the broken file is not overwritten');
672
+ const status = await fetch(url('/api/status'), { headers: withToken('/api/status', {}) });
673
+ assert.ok(status.status >= 400);
674
+ assert.match((await status.json()).error, /config\.json/);
675
+
676
+ const edited = JSON.parse(good);
677
+ edited.profiles.chat.name = 'Hand edited';
678
+ await new Promise(r => setTimeout(r, 20));
679
+ fs.writeFileSync(cfgPath, JSON.stringify(edited, null, 2));
680
+ const r = await post('/api/switch', { target: 'anthropic', profile: 'chat' });
681
+ assert.equal(r.status, 200);
682
+ assert.equal(JSON.parse(fs.readFileSync(cfgPath, 'utf8')).profiles.chat.name, 'Hand edited', 'the repaired hand edit survives the next mutation');
683
+ } finally {
684
+ await new Promise(r => setTimeout(r, 20));
685
+ fs.writeFileSync(cfgPath, good);
686
+ }
687
+ });
688
+
689
+ // Codex role handling follows the protocol the client speaks, not the profile's inFormat:
690
+ // an `auto` profile serves /v1/responses too, and a bare OpenAI id has no credentials upstream.
691
+ test('an auto profile maps Codex names by client protocol and leaves Claude mapping unchanged', async () => {
692
+ const upstreamModel = async (p, body) => {
693
+ const before = received.length;
694
+ const r = await post(p, body, { 'x-llm-profile': 'chat' });
695
+ await r.text();
696
+ return received.slice(before).at(-1)?.body?.model;
697
+ };
698
+ assert.equal(await upstreamModel('/v1/responses', { model: 'gpt-5.5', stream: false, input: 'hi' }), 'up-opus');
699
+ assert.equal(await upstreamModel('/v1/responses', { model: 'main', stream: false, input: 'hi' }), 'up-opus');
700
+ assert.equal(await upstreamModel('/v1/messages', { model: 'claude-opus-4-6', max_tokens: 16, messages: [{ role: 'user', content: 'hi' }] }), 'up-opus');
701
+ assert.equal(await upstreamModel('/v1/chat/completions', { model: 'default', messages: [{ role: 'user', content: 'hi' }] }), 'up-sonnet');
702
+ });
703
+
704
+ // ---- Request input limits and the Codex WS transport (audit F04, F05, F20, F37, F38, F51) ----
705
+
706
+ test('readBody: gzip, deflate and br bodies decode; a decompression bomb gets 413 and the gateway stays up', async () => {
707
+ const body = Buffer.from('{}');
708
+ for (const [enc, pack] of [['gzip', zlib.gzipSync], ['deflate', zlib.deflateSync], ['br', zlib.brotliCompressSync]]) {
709
+ const r = await fetch(url('/api/logs/clear'), { method: 'POST', headers: withToken('/api/logs/clear', { 'Content-Type': 'application/json', 'Content-Encoding': enc }), body: pack(body) });
710
+ assert.equal(r.status, 200, enc);
711
+ }
712
+ // 8 MB of zeros packs into a few KB, far under the 1 MB raw cap of /api/*.
713
+ const bomb = zlib.gzipSync(Buffer.alloc(8 * 1024 * 1024));
714
+ assert.ok(bomb.length < 64 * 1024);
715
+ const r = await fetch(url('/api/logs/clear'), { method: 'POST', headers: withToken('/api/logs/clear', { 'Content-Type': 'application/json', 'Content-Encoding': 'gzip' }), body: bomb });
716
+ assert.equal(r.status, 413);
717
+ assert.match((await r.json()).error, /after decompression/);
718
+ assert.equal((await fetch(url('/health'))).status, 200);
719
+ });
720
+
721
+ function clientFrame(opcode, payload, { fin = true, length } = {}) {
722
+ const data = Buffer.from(payload);
723
+ const len = length ?? data.length;
724
+ const head = len < 126 ? Buffer.from([(fin ? 0x80 : 0) | opcode, 0x80 | len])
725
+ : len < 65536 ? Buffer.from([(fin ? 0x80 : 0) | opcode, 0x80 | 126, len >> 8, len & 0xff])
726
+ : Buffer.concat([Buffer.from([(fin ? 0x80 : 0) | opcode, 0x80 | 127]), (() => { const b = Buffer.alloc(8); b.writeBigUInt64BE(BigInt(len)); return b; })()]);
727
+ const mask = crypto.randomBytes(4);
728
+ const masked = Buffer.from(data.map((b, i) => b ^ mask[i & 3]));
729
+ return Buffer.concat([head, mask, masked]);
730
+ }
731
+
732
+ // A raw client: the tests need fragments and oversized headers that WebSocket cannot send.
733
+ function rawWs(headers = {}) {
734
+ return new Promise((resolve, reject) => {
735
+ const socket = net.connect(proxyPort, '127.0.0.1');
736
+ const extra = Object.entries(headers).map(([k, v]) => `${k}: ${v}\r\n`).join('');
737
+ socket.write(`GET /v1/responses HTTP/1.1\r\nHost: 127.0.0.1:${proxyPort}\r\nUpgrade: websocket\r\nConnection: Upgrade\r\nSec-WebSocket-Version: 13\r\nSec-WebSocket-Key: ${crypto.randomBytes(16).toString('base64')}\r\n${extra}\r\n`);
738
+ let head = Buffer.alloc(0);
739
+ const read = createFrameReader();
740
+ const messages = [];
741
+ let closed = false;
742
+ const onData = (chunk) => {
743
+ head = Buffer.concat([head, chunk]);
744
+ const end = head.indexOf('\r\n\r\n');
745
+ if (end < 0) return;
746
+ socket.off('data', onData);
747
+ const rest = head.subarray(end + 4);
748
+ const collect = (c) => { for (const f of read(c)) messages.push(f.type === 'text' ? JSON.parse(f.payload.toString()) : f); };
749
+ socket.on('data', collect);
750
+ if (rest.length) collect(rest);
751
+ resolve({ socket, reply: head.subarray(0, end).toString(), messages, closed: () => closed });
752
+ };
753
+ socket.on('data', onData);
754
+ socket.on('close', () => { closed = true; });
755
+ socket.on('error', reject);
756
+ });
757
+ }
758
+
759
+ async function until(check, ms = 5000) {
760
+ const stop = Date.now() + ms;
761
+ while (Date.now() < stop) {
762
+ if (check()) return true;
763
+ await new Promise(r => setTimeout(r, 20));
764
+ }
765
+ return false;
766
+ }
767
+
768
+ test('Codex WS: a fragmented message is reassembled', async () => {
769
+ const ws = await rawWs();
770
+ const text = JSON.stringify({ type: 'session.update', session: { tag: 'fragmented' } });
771
+ ws.socket.write(clientFrame(1, text.slice(0, 10), { fin: false }));
772
+ ws.socket.write(clientFrame(0, text.slice(10), { fin: true }));
773
+ assert.ok(await until(() => ws.messages.some(m => m.type === 'session.updated')), 'no session.updated');
774
+ assert.equal(ws.messages.find(m => m.type === 'session.updated').session.tag, 'fragmented');
775
+ ws.socket.destroy();
776
+ });
777
+
778
+ test('Codex WS: a frame larger than the body cap closes the socket before it is buffered', async () => {
779
+ const ws = await rawWs();
780
+ ws.socket.write(clientFrame(1, 'x', { length: 2 ** 40 }).subarray(0, 14));
781
+ assert.ok(await until(() => ws.closed()), 'socket stays open');
782
+ const close = ws.messages.find(m => m.type === 'close');
783
+ assert.ok(close, 'a close frame is sent');
784
+ assert.equal(close.payload.readUInt16BE(0), 1009);
785
+ });
786
+
787
+ test('Codex WS: 101 reply names the public main model or the slot, never the upstream id', async () => {
788
+ const hidden = await rawWs({ 'x-llm-profile': 'agmock' });
789
+ assert.match(hidden.reply, /\r\nOpenAI-Model: main\r\n/i);
790
+ assert.ok(!hidden.reply.includes('ag/mock-flash'));
791
+ hidden.socket.destroy();
792
+ const published = await rawWs({ 'x-llm-profile': 'pub' });
793
+ assert.match(published.reply, /\r\nOpenAI-Model: gpt-5\.6-sol\r\n/i);
794
+ published.socket.destroy();
795
+ });
796
+
797
+ test('Codex WS: overlapping response.create turns run one after the other', async () => {
798
+ const ws = await rawWs();
799
+ const create = (input) => clientFrame(1, JSON.stringify({ type: 'response.create', model: 'main', input }));
800
+ ws.socket.write(Buffer.concat([create('SLOW_TURN first'), create('SLOW_TURN second')]));
801
+ const ends = () => ws.messages.filter(m => m.type === 'response.completed' || m.type === 'response.failed').length;
802
+ assert.ok(await until(() => ends() === 2, 10000), 'both turns end');
803
+ const order = ws.messages.filter(m => /^response\.(created|completed|failed)$/.test(m.type)).map(m => m.type);
804
+ assert.deepEqual(order, ['response.created', 'response.completed', 'response.created', 'response.completed']);
805
+ ws.socket.destroy();
806
+ });
807
+
808
+ test('Codex WS: a mid-stream error is logged with its text', async () => {
809
+ const ws = await rawWs();
810
+ ws.socket.write(clientFrame(1, JSON.stringify({ type: 'response.create', model: 'main', input: 'MID_STREAM_ERROR over ws' })));
811
+ assert.ok(await until(() => ws.messages.some(m => m.type === 'response.failed')), 'no response.failed');
812
+ ws.socket.destroy();
813
+ const { logs } = await (await fetch(url('/api/logs'), { headers: withToken('/api/logs', {}) })).json();
814
+ const entry = logs.find(l => l.clientFormat === 'responses-ws' && l.status === 502);
815
+ assert.ok(entry, 'a 502 responses-ws entry');
816
+ assert.match(entry.error || '', /upstream exploded/);
817
+ });
818
+
819
+ test('Codex WS: a turn that throws before its own error handling still ends in response.failed', async () => {
820
+ const ws = await rawWs();
821
+ ws.socket.write(clientFrame(1, JSON.stringify({ type: 'response.create', model: 12345, input: 'numeric model' })));
822
+ assert.ok(await until(() => ws.messages.some(m => m.type === 'response.failed')), `no response.failed: ${JSON.stringify(ws.messages.map(m => m.type))}`);
823
+ ws.socket.destroy();
824
+ });
825
+
826
+ test('An in-band upstream error closes the upstream connection instead of leaving it open', async () => {
827
+ const r = await post('/v1/chat/completions', { model: 'main', stream: true, messages: [{ role: 'user', content: 'ERROR_THEN_HANG' }] });
828
+ await r.text();
829
+ assert.ok(await until(() => hangState.closed, 3000), 'the upstream response is still open');
830
+ });
831
+
832
+ // ---- Request path: passthrough logging, backpressure, WS half-close, model windows (F21, F36, F51, G09, racer-M4) ----
833
+
834
+ const logsNow = async () => (await (await fetch(url('/api/logs'), { headers: withToken('/api/logs', {}) })).json()).logs;
835
+ const ant = (content) => post('/v1/messages', { model: 'claude-opus-4-6', max_tokens: 10, stream: true, messages: [{ role: 'user', content }] }, { 'x-llm-profile': 'ant' });
836
+
837
+ test('Direct passthrough logs the real token counts', async () => {
838
+ await (await ant('DIRECT_STREAM')).text();
839
+ const entry = (await logsNow()).find(l => l.profile === 'ant');
840
+ assert.deepEqual(entry.tokens, { prompt: 11, completion: 7 });
841
+ const plain = await post('/v1/messages', { model: 'claude-opus-4-6', max_tokens: 10, messages: [{ role: 'user', content: 'hi' }] }, { 'x-llm-profile': 'ant' });
842
+ await plain.json();
843
+ assert.deepEqual((await logsNow()).find(l => l.profile === 'ant').tokens, { prompt: 3, completion: 2 });
844
+ });
845
+
846
+ test('Direct passthrough logs a mid-stream failure as 502 and a client abort as 499', async () => {
847
+ await (await ant('DIRECT_BREAK')).text().catch(() => {});
848
+ await new Promise(r => setTimeout(r, 100));
849
+ const broken = (await logsNow()).find(l => l.profile === 'ant');
850
+ assert.equal(broken.status, 502);
851
+ assert.ok(broken.error, 'the failure is named');
852
+
853
+ const ac = new AbortController();
854
+ const r = await fetch(url('/v1/messages'), { method: 'POST', signal: ac.signal, headers: { 'Content-Type': 'application/json', 'x-llm-profile': 'ant' },
855
+ body: JSON.stringify({ model: 'claude-opus-4-6', max_tokens: 10, stream: true, messages: [{ role: 'user', content: 'DIRECT_SLOW' }] }) });
856
+ const reader = r.body.getReader();
857
+ await reader.read();
858
+ ac.abort();
859
+ await new Promise(res => setTimeout(res, 800));
860
+ assert.equal((await logsNow()).find(l => l.profile === 'ant').status, 499);
861
+ });
862
+
863
+ test('Direct passthrough waits for a slow client instead of buffering the whole upstream stream', async () => {
864
+ bigState.finishedAt = 0;
865
+ const resumedAt = await new Promise((resolve, reject) => {
866
+ const req = http.request({ host: '127.0.0.1', port: proxyPort, path: '/v1/messages', method: 'POST', headers: { 'Content-Type': 'application/json', 'x-llm-profile': 'ant' } }, res => {
867
+ res.pause();
868
+ setTimeout(() => {
869
+ const at = Date.now();
870
+ res.resume();
871
+ res.on('end', () => resolve(at));
872
+ }, 1500);
873
+ });
874
+ req.on('error', reject);
875
+ req.end(JSON.stringify({ model: 'claude-opus-4-6', max_tokens: 10, stream: true, messages: [{ role: 'user', content: 'DIRECT_BIG' }] }));
876
+ });
877
+ assert.ok(bigState.finishedAt >= resumedAt, `upstream finished ${resumedAt - bigState.finishedAt} ms before the client read anything`);
878
+ });
879
+
880
+ test('Codex WS: a client half-close aborts the running turn', async () => {
881
+ hangState.slowAborted = false;
882
+ const ws = await rawWs();
883
+ ws.socket.write(clientFrame(1, JSON.stringify({ type: 'response.create', model: 'main', input: 'SLOW_TURN half close' })));
884
+ await new Promise(r => setTimeout(r, 100));
885
+ ws.socket.end();
886
+ assert.ok(await until(() => hangState.slowAborted, 2000), 'the upstream turn kept running');
887
+ });
888
+
889
+ test('/v1/models windows follow model1M and the entry comes from codex-catalog-template.json', async () => {
890
+ const template = JSON.parse(fs.readFileSync(path.join(ROOT, 'codex-catalog-template.json'), 'utf8'));
891
+ const pub = await (await fetch(url('/v1/models'), { headers: { 'x-llm-profile': 'pub' } })).json();
892
+ const win = Object.fromEntries(pub.models.map(m => [m.slug, m.context_window]));
893
+ assert.deepEqual(win, { 'gpt-5.6-sol': 1000000, 'gpt-5.2': template.context_window });
894
+ for (const m of pub.models) assert.equal(m.description, template.description);
895
+ const one = await (await fetch(url('/v1/models/gpt-5.2'), { headers: { 'x-llm-profile': 'pub' } })).json();
896
+ assert.equal(one.context_window, template.context_window);
897
+ assert.equal(one.id, 'gpt-5.2');
898
+ const main = await (await fetch(url('/v1/models/gpt-5.6-sol'), { headers: { 'x-llm-profile': 'pub' } })).json();
899
+ assert.equal(main.context_window, 1000000);
900
+ });
901
+
902
+ // ---- Admin API: stale dashboard writes, control characters, key destination (F06, F33, keeper next-time) ----
903
+
904
+ test('a dashboard change based on a stale revision is refused with 409 and changes nothing', async () => {
905
+ const { revision } = await (await fetch(url('/api/status'), { headers: withToken('/api/status', {}) })).json();
906
+ assert.match(revision, /^[0-9a-f]{16}$/);
907
+ const first = await post('/api/save-profile', { key: 'rev1', revision, profile: { name: 'Rev', mode: 'convert', inFormat: 'auto', baseURL: 'http://127.0.0.1:9/v1', apiKey: 'k' } });
908
+ assert.equal(first.status, 200);
909
+ const next = (await first.json()).revision;
910
+ assert.notEqual(next, revision);
911
+ const before = fs.readFileSync(path.join(tmpDir, 'config.json'), 'utf8');
912
+ const stale = await post('/api/save-profile', { key: 'rev1', revision, profile: { name: 'Stale' } });
913
+ assert.equal(stale.status, 409);
914
+ assert.equal(fs.readFileSync(path.join(tmpDir, 'config.json'), 'utf8'), before);
915
+ // A caller that sends no revision (the MCP server) is not checked.
916
+ assert.equal((await post('/api/delete-profile', { key: 'rev1' })).status, 200);
917
+ });
918
+
919
+ test('save-profile refuses control characters in the name', async () => {
920
+ const r = await post('/api/save-profile', { key: 'ctl', profile: { name: 'bad\u001b[2J', mode: 'convert', inFormat: 'auto', baseURL: 'http://127.0.0.1:9/v1', apiKey: 'k' } });
921
+ assert.equal(r.status, 400);
922
+ });
923
+
924
+ test('a kept API key is never pointed at a new baseURL or new endpoints', async () => {
925
+ const create = await post('/api/save-profile', { key: 'keydest', profile: { name: 'K', mode: 'convert', inFormat: 'auto', baseURL: 'http://127.0.0.1:9/v1', apiKey: 'sk-keydest' } });
926
+ assert.equal(create.status, 200);
927
+ try {
928
+ for (const change of [{ endpoints: { 'openai-chat': 'http://evil.test/v1/chat/completions' } }, { baseURL: 'http://evil.test/v1' }, { baseURL: 'http://evil.test/v1', apiKey: MASKED }]) {
929
+ const r = await post('/api/save-profile', { key: 'keydest', profile: change });
930
+ assert.equal(r.status, 400, JSON.stringify(change));
931
+ }
932
+ const stored = JSON.parse(fs.readFileSync(path.join(tmpDir, 'config.json'), 'utf8')).profiles.keydest;
933
+ assert.equal(stored.baseURL, 'http://127.0.0.1:9/v1');
934
+ assert.equal(stored.endpoints, undefined);
935
+ assert.equal(stored.apiKey, 'sk-keydest');
936
+ // A new key typed together with the new URL is accepted.
937
+ assert.equal((await post('/api/save-profile', { key: 'keydest', profile: { baseURL: 'http://127.0.0.1:8/v1', apiKey: 'sk-new' } })).status, 200);
938
+ // The same destination keeps the key without retyping it.
939
+ assert.equal((await post('/api/save-profile', { key: 'keydest', profile: { name: 'K2', apiKey: MASKED } })).status, 200);
940
+ assert.equal(JSON.parse(fs.readFileSync(path.join(tmpDir, 'config.json'), 'utf8')).profiles.keydest.apiKey, 'sk-new');
941
+ } finally {
942
+ await post('/api/delete-profile', { key: 'keydest' });
943
+ }
944
+ });
945
+
946
+ // ---- Routes no test reached before (audit F46) ----
947
+
948
+ test('OPTIONS answers a loopback preflight and refuses a foreign origin', async () => {
949
+ const ok = await fetch(url('/v1/messages'), { method: 'OPTIONS', headers: { Origin: `http://127.0.0.1:${proxyPort}` } });
950
+ assert.equal(ok.status, 204);
951
+ assert.equal(ok.headers.get('access-control-allow-origin'), `http://127.0.0.1:${proxyPort}`);
952
+ const foreign = await fetch(url('/v1/messages'), { method: 'OPTIONS', headers: { Origin: 'https://evil.example' } });
953
+ assert.equal(foreign.status, 403);
954
+ });
955
+
956
+ test('/ui serves the dashboard with anti-framing headers', async () => {
957
+ const r = await fetch(url('/ui'));
958
+ assert.equal(r.status, 200);
959
+ assert.match(r.headers.get('content-type'), /text\/html/);
960
+ assert.equal(r.headers.get('x-frame-options'), 'DENY');
961
+ assert.match(await r.text(), /LLM Switcher/);
962
+ });
963
+
964
+ test('Vertex routes: streaming action, full resource path, and an unknown action', async () => {
965
+ const stream = await post('/v1beta/models/claude-opus-like:streamGenerateContent?alt=sse', { contents: [{ role: 'user', parts: [{ text: 'hi' }] }] });
966
+ assert.equal(stream.status, 200);
967
+ assert.match(stream.headers.get('content-type'), /text\/event-stream/);
968
+ const frames = parseSSE(await stream.text()).filter(f => f.data);
969
+ assert.ok(frames.some(f => f.data.candidates?.[0]?.content?.parts?.some(p => p.text)), 'a text part is streamed');
970
+ const full = await post('/v1/projects/p1/locations/us-central1/publishers/google/models/claude-opus-like:generateContent', { contents: [{ role: 'user', parts: [{ text: 'hi' }] }] });
971
+ assert.equal(full.status, 200);
972
+ assert.equal(received.at(-1).body.model, 'up-opus');
973
+ const bad = await post('/v1beta/models/x:explode', { contents: [] });
974
+ assert.equal(bad.status, 404);
975
+ });
976
+
977
+ test('Codex WS: conversation.item.create is echoed and response.cancel stops the running turn', async () => {
978
+ hangState.slowAborted = false;
979
+ const ws = await rawWs();
980
+ ws.socket.write(clientFrame(1, JSON.stringify({ type: 'conversation.item.create', item: { id: 'i1' } })));
981
+ assert.ok(await until(() => ws.messages.some(m => m.type === 'conversation.item.created')));
982
+ assert.equal(ws.messages.find(m => m.type === 'conversation.item.created').item.id, 'i1');
983
+ ws.socket.write(clientFrame(1, JSON.stringify({ type: 'response.create', model: 'main', input: 'SLOW_TURN cancel me' })));
984
+ assert.ok(await until(() => ws.messages.some(m => m.type === 'response.created')));
985
+ ws.socket.write(clientFrame(1, JSON.stringify({ type: 'response.cancel' })));
986
+ assert.ok(await until(() => hangState.slowAborted, 2000), 'the upstream turn kept running');
987
+ assert.ok(!ws.messages.some(m => m.type === 'response.completed'));
988
+ ws.socket.destroy();
989
+ });
990
+
991
+ test('Admin test-upstream reports latency and a sample from the upstream', async () => {
992
+ const r = await post('/api/test-upstream', { baseURL: `http://127.0.0.1:${upstreamPort}/chat/v1`, apiKey: 'k', model: 'm', mode: 'convert' });
993
+ assert.equal(r.status, 200);
994
+ const json = await r.json();
995
+ assert.equal(json.ok, true);
996
+ assert.equal(json.outFormat, 'openai-chat');
997
+ assert.match(json.sample, /Final answer/);
998
+ assert.equal(typeof json.latency, 'number');
999
+ });