llm-switcher 1.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. package/.gitattributes +16 -0
  2. package/LICENSE +21 -0
  3. package/README.md +587 -0
  4. package/README.vi.md +585 -0
  5. package/blindfold/blindfold.mjs +633 -0
  6. package/blindfold/make-certs.sh +88 -0
  7. package/blindfold/wsframe.mjs +176 -0
  8. package/codex-catalog-template.json +1 -0
  9. package/config.example.json +84 -0
  10. package/contract-exclusions.json +41 -0
  11. package/contract.mjs +561 -0
  12. package/docs/LLM-RESPONSE-MATRIX.md +165 -0
  13. package/docs/TOKEN-OPTIMIZER-INTEROP.md +110 -0
  14. package/docs/codex-blindfold.md +214 -0
  15. package/docs/cross-platform.md +136 -0
  16. package/docs/diagrams/blindfold-request-routing.html +14972 -0
  17. package/docs/diagrams/blindfold-request-routing.sequence.json +175 -0
  18. package/docs/diagrams/blindfold-switch-lifecycle.html +14958 -0
  19. package/docs/diagrams/blindfold-switch-lifecycle.lifecycle.json +159 -0
  20. package/docs/diagrams/codex-model-name-resolution.html +15005 -0
  21. package/docs/diagrams/codex-model-name-resolution.workflow.json +71 -0
  22. package/docs/response-matrix.json +1131 -0
  23. package/formats.mjs +2308 -0
  24. package/mcp.mjs +340 -0
  25. package/package.json +36 -0
  26. package/proxy.mjs +1743 -0
  27. package/service.mjs +132 -0
  28. package/shim.mjs +292 -0
  29. package/skills/llm-switcher/SKILL.md +88 -0
  30. package/state.mjs +978 -0
  31. package/switch +5 -0
  32. package/switch.cmd +2 -0
  33. package/switch.mjs +930 -0
  34. package/tests/blindfold.test.mjs +307 -0
  35. package/tests/blindfold.wire.test.mjs +170 -0
  36. package/tests/contract/run.test.mjs +214 -0
  37. package/tests/contract-check.test.mjs +458 -0
  38. package/tests/contract-lab.test.mjs +755 -0
  39. package/tests/datadir.test.mjs +37 -0
  40. package/tests/formats.test.mjs +794 -0
  41. package/tests/gateway.e2e.test.mjs +999 -0
  42. package/tests/helpers.mjs +24 -0
  43. package/tests/lifecycle.test.mjs +416 -0
  44. package/tests/live-optimizer-interop.mjs +205 -0
  45. package/tests/mcp.test.mjs +91 -0
  46. package/tests/service.test.mjs +69 -0
  47. package/tests/shim.test.mjs +228 -0
  48. package/tests/state.test.mjs +675 -0
  49. package/tests/switch.test.mjs +156 -0
  50. package/tests/wsframe.test.mjs +154 -0
  51. package/ui.html +2234 -0
package/proxy.mjs ADDED
@@ -0,0 +1,1743 @@
1
+ import http from 'node:http';
2
+ import fs from 'node:fs';
3
+ import path from 'node:path';
4
+ import zlib from 'node:zlib';
5
+ import crypto from 'node:crypto';
6
+ import { promisify } from 'node:util';
7
+ import { fileURLToPath } from 'node:url';
8
+ import { createFrameReader } from './blindfold/wsframe.mjs';
9
+ import {
10
+ OUT_FORMATS, IN_FORMATS, parseToIR, emitUpstreamBody, createUpstreamNormalizer, createCollector,
11
+ createThinkTagSplitter, splitThinkTags, healAnthropicPayload, estimateTokens, THINKING_MODES,
12
+ toGeminiSchema, isAntigravityModel,
13
+ createAnthropicStream, createChatStream, createResponsesStream, createVertexStream,
14
+ buildAnthropicMessage, buildChatMessage, buildResponsesMessage, buildVertexMessage
15
+ } from './formats.mjs';
16
+ import {
17
+ TARGETS, configPath, loadConfig, getConfigLoadError, saveConfig, resolvePort, hasProfile, isValidProfileKey,
18
+ getActiveMap, setTargetProfile, activateProfile, deactivateProfile, deactivateAll, deleteProfile,
19
+ isProfileActive, profileAcceptsTarget, applyLaunchState, readLaunchFlags, redactConfig, MASKED_KEY,
20
+ modelForSlot, primaryModel, codexPublicModel, isSafeModelName, parsePort, CODEX_MODEL_SLOTS,
21
+ ensureAdminToken, identityProof, reconcileBlindfold, checkBlindfoldTarget,
22
+ codexModelEntry, smallestWindows, publicModelWindows, model1MForSlot, computeLaunchState,
23
+ contractLabSettings
24
+ } from './state.mjs';
25
+ import { createContractLab, createHalfTap, tapClientWrites, capText, capJson, toolVersionFromUA, finishHalf, PROBE_HEADER, TRACE_ID_RE } from './contract.mjs';
26
+
27
+ const __dirname = path.dirname(fileURLToPath(import.meta.url));
28
+ const uiHtmlPath = path.join(__dirname, 'ui.html');
29
+
30
+ const MAX_BODY_SIZE = 50 * 1024 * 1024; // 50MB
31
+ const MAX_API_BODY_SIZE = 1024 * 1024; // 1MB for /api/*
32
+
33
+ // Avoid network conflicts: keep localhost / 127.0.0.1 out of external proxies (RTK, Headroom, VPN)
34
+ const currentNoProxy = process.env.NO_PROXY || process.env.no_proxy || '';
35
+ const localHosts = ['127.0.0.1', 'localhost'];
36
+ const existingNoProxy = currentNoProxy.split(',').map(s => s.trim().toLowerCase());
37
+ const missingNoProxy = localHosts.filter(h => !existingNoProxy.includes(h));
38
+ if (missingNoProxy.length > 0) {
39
+ process.env.NO_PROXY = currentNoProxy ? `${currentNoProxy},${missingNoProxy.join(',')}` : missingNoProxy.join(',');
40
+ process.env.no_proxy = process.env.NO_PROXY;
41
+ }
42
+
43
+ // Fixed port for the process lifetime: changing "port" in config.json at runtime has no effect
44
+ // env.cmd / flags would point at a port the server is not listening on.
45
+ const PORT = resolvePort();
46
+
47
+ // In-memory Request / Response Inspector Ring Buffer (up to 40 most recent requests)
48
+ const requestLogs = [];
49
+ const MAX_LOGS = 40;
50
+ function logInspection(entry) {
51
+ requestLogs.push({
52
+ id: `req_${Date.now()}_${Math.random().toString(36).slice(2, 6)}`,
53
+ timestamp: new Date().toLocaleTimeString(),
54
+ ...entry
55
+ });
56
+ if (requestLogs.length > MAX_LOGS) {
57
+ requestLogs.shift();
58
+ }
59
+ }
60
+
61
+ function debugLog(...args) {
62
+ if (loadConfig()?.debug) {
63
+ console.log('[DEBUG]', ...args);
64
+ }
65
+ }
66
+
67
+ function sendJson(res, status, obj, headers = {}) {
68
+ if (res.headersSent) {
69
+ try { res.end(); } catch {}
70
+ return;
71
+ }
72
+ res.writeHead(status, { 'Content-Type': 'application/json', ...headers });
73
+ res.end(JSON.stringify(obj));
74
+ }
75
+
76
+ // ----------------------------------------------------
77
+ // Request guards & body reading
78
+ // ----------------------------------------------------
79
+ const LOOPBACK_HOSTS = new Set(['127.0.0.1', 'localhost', '[::1]', '::1']);
80
+
81
+ function hostnameOf(hostHeader) {
82
+ const h = String(hostHeader || '').trim().toLowerCase();
83
+ if (h.startsWith('[')) return h.slice(0, h.indexOf(']') + 1);
84
+ return h.replace(/:\d+$/, '');
85
+ }
86
+
87
+ // Block DNS rebinding (unknown Host pointing at 127.0.0.1) and CSRF from other sites (unknown Origin).
88
+ // Without the Host check, a malicious page could burn tokens via /v1/*.
89
+ function checkRequestOrigin(req) {
90
+ if (req.headers.host && !LOOPBACK_HOSTS.has(hostnameOf(req.headers.host))) {
91
+ return 'Forbidden: untrusted Host header';
92
+ }
93
+ const origin = req.headers.origin;
94
+ if (origin !== undefined) {
95
+ try {
96
+ const u = new URL(origin);
97
+ const port = u.port || (u.protocol === 'https:' ? '443' : '80');
98
+ if (!LOOPBACK_HOSTS.has(u.hostname.toLowerCase()) || port !== String(PORT)) {
99
+ return 'Forbidden: untrusted origin';
100
+ }
101
+ } catch {
102
+ return 'Forbidden: invalid origin';
103
+ }
104
+ }
105
+ return null;
106
+ }
107
+
108
+ // The Host/Origin guard stops browser pages only. A local process that is not the owner must
109
+ // also present the token from admin.token (mode 0600) to use /api/*.
110
+ const ADMIN_TOKEN = Buffer.from(ensureAdminToken());
111
+
112
+ // `switch contract-probe` names the trace id of the exchange it drives, so it can print the id
113
+ // before intact holds it. Only a process that can read admin.token is believed, and the marker
114
+ // header is on BLOCKED_PASSTHROUGH: it never leaves this gateway.
115
+ function probeTraceId(req) {
116
+ const given = String(req.headers[PROBE_HEADER] || '');
117
+ if (!given || !TRACE_ID_RE.test(given)) return null;
118
+ return isAdminRequest(req) ? given : null;
119
+ }
120
+
121
+ function isAdminRequest(req) {
122
+ const given = Buffer.from(String(req.headers['x-llm-switcher-token'] || ''));
123
+ return given.length === ADMIN_TOKEN.length && crypto.timingSafeEqual(given, ADMIN_TOKEN);
124
+ }
125
+
126
+ function readBody(req, limit) {
127
+ return new Promise((resolve, reject) => {
128
+ let size = 0;
129
+ let tooLarge = false;
130
+ const chunks = [];
131
+ req.on('data', chunk => {
132
+ if (tooLarge) return;
133
+ size += chunk.length;
134
+ if (size > limit) {
135
+ tooLarge = true;
136
+ const err = new Error(`Payload Too Large (max ${Math.round(limit / 1024 / 1024)}MB)`);
137
+ err.status = 413;
138
+ reject(err);
139
+ req.resume();
140
+ return;
141
+ }
142
+ chunks.push(chunk);
143
+ });
144
+ req.on('end', () => {
145
+ if (tooLarge) return;
146
+ decodeBody(Buffer.concat(chunks), req.headers['content-encoding'], limit).then(resolve, reject);
147
+ });
148
+ req.on('error', err => {
149
+ err.status = 400;
150
+ reject(err);
151
+ });
152
+ });
153
+ }
154
+
155
+ const INFLATERS = {
156
+ gzip: promisify(zlib.gunzip),
157
+ deflate: promisify(zlib.inflate),
158
+ br: promisify(zlib.brotliDecompress),
159
+ ...(typeof zlib.zstdDecompress === 'function' ? { zstd: promisify(zlib.zstdDecompress) } : {})
160
+ };
161
+
162
+ // Decompression runs off the event loop and stops at the raw-body limit: a few KB of gzip can
163
+ // expand to gigabytes.
164
+ async function decodeBody(raw, contentEncoding, limit) {
165
+ let encoding = String(contentEncoding || '').toLowerCase().trim();
166
+ if (!['gzip', 'deflate', 'br', 'zstd'].includes(encoding)) {
167
+ // Magic bytes: some clients compress without saying so.
168
+ if (raw.length >= 4 && raw[0] === 0x28 && raw[1] === 0xb5 && raw[2] === 0x2f && raw[3] === 0xfd && INFLATERS.zstd) encoding = 'zstd';
169
+ else if (raw.length >= 2 && raw[0] === 0x1f && raw[1] === 0x8b) encoding = 'gzip';
170
+ else return raw;
171
+ }
172
+ const inflate = INFLATERS[encoding];
173
+ if (!inflate) throw Object.assign(new Error('zstd decompression not supported in this Node.js version'), { status: 415 });
174
+ try {
175
+ return await inflate(raw, { maxOutputLength: limit });
176
+ } catch (err) {
177
+ if (err.code === 'ERR_BUFFER_TOO_LARGE') {
178
+ throw Object.assign(new Error(`Payload Too Large after decompression (max ${Math.round(limit / 1024 / 1024)}MB)`), { status: 413 });
179
+ }
180
+ err.status = 400;
181
+ throw err;
182
+ }
183
+ }
184
+
185
+ async function readJsonBody(req, limit = MAX_API_BODY_SIZE) {
186
+ const buf = await readBody(req, limit);
187
+ try {
188
+ const v = JSON.parse(buf.toString('utf8') || '{}');
189
+ if (!v || typeof v !== 'object' || Array.isArray(v)) throw new Error('JSON body must be an object');
190
+ return v;
191
+ } catch (e) {
192
+ const err = new Error(`Invalid JSON body: ${e.message}`);
193
+ err.status = 400;
194
+ throw err;
195
+ }
196
+ }
197
+
198
+ // ----------------------------------------------------
199
+ // Profile & model resolution
200
+ // ----------------------------------------------------
201
+ function getActiveProfile(clientFormat, req) {
202
+ const cfg = loadConfig();
203
+ if (!cfg) return { cfg: null, profileKey: null, profile: null };
204
+
205
+ let reqProfile = null;
206
+ if (req) {
207
+ reqProfile = req.headers['x-llm-profile'] || req.headers['x-profile'] || null;
208
+ if (!reqProfile && req.url) {
209
+ try { reqProfile = new URL(req.url, 'http://localhost').searchParams.get('profile'); } catch {}
210
+ }
211
+ }
212
+ if (reqProfile) {
213
+ if (hasProfile(cfg, reqProfile)) return { cfg, profileKey: reqProfile, profile: cfg.profiles[reqProfile] };
214
+ return { cfg, profileKey: reqProfile, profile: null, error: `Profile "${reqProfile}" requested via x-llm-profile/?profile= does not exist` };
215
+ }
216
+
217
+ // Look up the active profile by CLI target (clientFormat: anthropic | responses | openai-chat | vertex).
218
+ // Deleted profile / disabled target counts as OFF, never silently fall through to another profile (different API key!).
219
+ const key = clientFormat ? getActiveMap(cfg)[clientFormat] : (cfg.activeProfile || null);
220
+ if (!key || !hasProfile(cfg, key)) return { cfg, profileKey: key || null, profile: null };
221
+ return { cfg, profileKey: key, profile: cfg.profiles[key] };
222
+ }
223
+
224
+ // For endpoints not tied to a specific client format (/health, /v1/models).
225
+ function getFirstActiveProfile(preferred, req) {
226
+ if (req) {
227
+ const r = getActiveProfile(null, req);
228
+ if (r.profile) return r;
229
+ }
230
+ for (const t of preferred) {
231
+ const r = getActiveProfile(t);
232
+ if (r.profile) return r;
233
+ }
234
+ return { cfg: loadConfig(), profileKey: null, profile: null };
235
+ }
236
+
237
+ // clientFormat is the protocol the request arrived in. An `auto` profile serves every protocol,
238
+ // so the profile's inFormat cannot tell a Codex request from a Claude one.
239
+ function mapModel(requestedModel, profile, clientFormat) {
240
+ if (!requestedModel) return primaryModel(profile);
241
+ const clean = requestedModel.replace(/\[1m\]/gi, '').trim();
242
+ // If the client specified a model with a provider prefix (e.g. ag/..., gh/..., cf/...), keep it as-is
243
+ if (clean.includes('/') && !clean.startsWith('anthropic/')) {
244
+ return clean;
245
+ }
246
+ const m = clean.toLowerCase();
247
+ if (clientFormat === 'responses') {
248
+ // Aliases for the real Codex roles in the docs (model / review_model /
249
+ // agents.default_subagent_model). Unknown names pass through unchanged.
250
+ const aliases = {
251
+ main: ['main', 'default', 'codex-main', 'codex-default'],
252
+ review: ['review', 'codex-review'],
253
+ subagent: ['subagent', 'codex-subagent']
254
+ };
255
+ for (const [slot, names] of Object.entries(aliases)) {
256
+ if (names.includes(m)) return modelForSlot(profile, slot) || clean;
257
+ }
258
+ // The official names the CLI was given (publicModels) carry the role. Resolve
259
+ // them before the fail-closed rule below, otherwise review and subagent traffic
260
+ // collapses onto the main slot.
261
+ for (const slot of CODEX_MODEL_SLOTS) {
262
+ const publicName = codexPublicModel(profile, slot);
263
+ if (publicName && publicName.toLowerCase() === m) return modelForSlot(profile, slot) || clean;
264
+ }
265
+ // Codex model IDs change frequently. Bare OpenAI IDs (config leftovers like gpt-5.6-sol,
266
+ // retired gpt-5.3-codex) have no credentials behind 9Router -> fail closed to the main slot
267
+ // instead of passing through to an upstream 404. Provider-prefixed names (ag/..., cf/...)
268
+ // are preserved by the early return above.
269
+ if (/^(gpt|o\d|codex)([-/]|$)/i.test(clean)) return modelForSlot(profile, 'main') || clean;
270
+ return clean || requestedModel;
271
+ }
272
+ if (clientFormat === 'openai-chat' || clientFormat === 'vertex') {
273
+ if (m === 'default' || m === 'main' || !clean) return modelForSlot(profile, 'default') || clean;
274
+ }
275
+ if (m.includes('fable')) return modelForSlot(profile, 'fable') || clean;
276
+ if (m.includes('opus')) return modelForSlot(profile, 'opus') || clean;
277
+ if (m.includes('haiku')) return modelForSlot(profile, 'haiku') || clean;
278
+ if (m.includes('sonnet')) return modelForSlot(profile, 'sonnet') || clean;
279
+ return clean || requestedModel;
280
+ }
281
+
282
+ // Wait until a slow client has taken the buffered bytes, so the gateway does not read the whole
283
+ // upstream stream into memory. A closed client, or an aborted turn, also ends the wait.
284
+ function drained(stream, signal) {
285
+ if (!stream.writableNeedDrain || stream.destroyed || signal?.aborted) return null;
286
+ return new Promise(resolve => {
287
+ const done = () => {
288
+ stream.off('drain', done);
289
+ stream.off('close', done);
290
+ signal?.removeEventListener('abort', done);
291
+ resolve();
292
+ };
293
+ stream.on('drain', done);
294
+ stream.on('close', done);
295
+ signal?.addEventListener('abort', done);
296
+ });
297
+ }
298
+
299
+ function sendSSE(res, event, data) {
300
+ // event=null -> raw `data:` line (OpenAI Chat / Vertex clients).
301
+ if (event) res.write(`event: ${event}\ndata: ${JSON.stringify(data)}\n\n`);
302
+ else res.write(`data: ${JSON.stringify(data)}\n\n`);
303
+ }
304
+
305
+ // ----------------------------------------------------
306
+ // Client-shaped errors
307
+ // ----------------------------------------------------
308
+ function anthropicErrorType(status) {
309
+ switch (status) {
310
+ case 400: return 'invalid_request_error';
311
+ case 401: return 'authentication_error';
312
+ case 403: return 'permission_error';
313
+ case 404: return 'not_found_error';
314
+ case 413: return 'request_too_large';
315
+ case 429: return 'rate_limit_error';
316
+ case 529: return 'overloaded_error';
317
+ default: return status >= 500 ? 'api_error' : 'invalid_request_error';
318
+ }
319
+ }
320
+
321
+ function vertexStatus(status) {
322
+ switch (status) {
323
+ case 400: return 'INVALID_ARGUMENT';
324
+ case 401: return 'UNAUTHENTICATED';
325
+ case 403: return 'PERMISSION_DENIED';
326
+ case 404: return 'NOT_FOUND';
327
+ case 429: return 'RESOURCE_EXHAUSTED';
328
+ case 503: return 'UNAVAILABLE';
329
+ default: return status >= 500 ? 'INTERNAL' : 'FAILED_PRECONDITION';
330
+ }
331
+ }
332
+
333
+ // Each SDK parses errors in its own shape; Claude Code relies on error.type + retry-after to decide retries.
334
+ function sendClientError(res, clientFormat, status, message, headers = {}) {
335
+ let body;
336
+ if (clientFormat === 'anthropic') {
337
+ body = { type: 'error', error: { type: anthropicErrorType(status), message } };
338
+ } else if (clientFormat === 'vertex') {
339
+ body = { error: { code: status, message, status: vertexStatus(status) } };
340
+ } else {
341
+ body = { error: { message, type: status >= 500 ? 'server_error' : 'invalid_request_error', code: String(status) } };
342
+ }
343
+ sendJson(res, status, body, headers);
344
+ }
345
+
346
+ function extractUpstreamMessage(text) {
347
+ try {
348
+ const j = JSON.parse(text);
349
+ const e = Array.isArray(j) ? j[0]?.error : j.error;
350
+ if (typeof e === 'string') return e;
351
+ if (e?.message) return e.message;
352
+ if (j.message) return j.message;
353
+ } catch {}
354
+ return text;
355
+ }
356
+
357
+ const RETRY_HEADERS = ['retry-after', 'retry-after-ms', 'x-should-retry'];
358
+ function pickRetryHeaders(upstreamRes) {
359
+ const h = {};
360
+ for (const k of RETRY_HEADERS) {
361
+ const v = upstreamRes.headers.get(k);
362
+ if (v) h[k] = v;
363
+ }
364
+ return h;
365
+ }
366
+
367
+ // ----------------------------------------------------
368
+ // Upstream endpoint & headers
369
+ // ----------------------------------------------------
370
+ function resolveOutFormat(profile, mappedModel) {
371
+ if (profile.outFormat && OUT_FORMATS.includes(profile.outFormat)) return profile.outFormat;
372
+ const mode = profile.mode || 'hybrid';
373
+ if (mode === 'direct') return 'anthropic';
374
+ if (mode === 'convert') return 'openai-chat';
375
+ return String(mappedModel || '').toLowerCase().startsWith('claude-') ? 'anthropic' : 'openai-chat';
376
+ }
377
+
378
+ // Client headers that must NOT be forwarded upstream: client credentials (e.g. the Gemini SDK's x-goog-api-key
379
+ // would leak to a third-party upstream), switcher control headers, and hop-by-hop / network identity headers.
380
+ // x-intact-trace is on this list because a client must never choose the id that joins the two
381
+ // halves of a captured exchange. Only the contract lab of this gateway writes it.
382
+ const BLOCKED_PASSTHROUGH = new Set([
383
+ 'x-api-key', 'x-goog-api-key', 'x-goog-user-project', 'x-profile', 'x-llm-profile',
384
+ 'x-real-ip', 'x-forwarded-for', 'x-forwarded-host', 'x-forwarded-proto', 'x-forwarded-port',
385
+ 'x-intact-trace', PROBE_HEADER, 'x-llm-switcher-token'
386
+ ]);
387
+
388
+ function upstreamEndpoint(profile, outFormat, model, stream, req) {
389
+ const base = String(profile.baseURL || '').replace(/\/+$/, '');
390
+ const ov = profile.endpoints || {};
391
+ const key = profile.apiKey || '';
392
+ const headers = { 'Content-Type': 'application/json' };
393
+
394
+ // Safely pass through client headers from intermediate tools (x-request-id, traceparent, x-...)
395
+ if (req?.headers) {
396
+ for (const [k, v] of Object.entries(req.headers)) {
397
+ const lk = k.toLowerCase();
398
+ if ((lk.startsWith('x-') || lk === 'traceparent' || lk === 'tracestate') && !BLOCKED_PASSTHROUGH.has(lk)) {
399
+ headers[lk] = v;
400
+ }
401
+ }
402
+ }
403
+
404
+ if (outFormat === 'anthropic') {
405
+ headers['x-api-key'] = key;
406
+ headers['anthropic-version'] = req?.headers?.['anthropic-version'] || '2023-06-01';
407
+ if (req?.headers?.['anthropic-beta']) headers['anthropic-beta'] = req.headers['anthropic-beta'];
408
+ } else {
409
+ headers['authorization'] = `Bearer ${key}`;
410
+ }
411
+
412
+ if (outFormat === 'anthropic') {
413
+ return { url: ov.anthropic || `${base}/messages`, headers };
414
+ }
415
+ if (outFormat === 'vertex') {
416
+ const action = stream ? 'streamGenerateContent' : 'generateContent';
417
+ const url = ov.vertex
418
+ ? ov.vertex.replace('{model}', encodeURIComponent(model)).replace('{action}', action)
419
+ : `${base}/models/${encodeURIComponent(model)}:${action}`;
420
+ return { url: stream && !/[?&]alt=sse/.test(url) ? `${url}${url.includes('?') ? '&' : '?'}alt=sse` : url, headers };
421
+ }
422
+ return { url: ov['openai-chat'] || `${base}/chat/completions`, headers };
423
+ }
424
+
425
+ // Read the upstream stream: accept both SSE `data:` and raw JSON lines (Vertex framing).
426
+ async function* readUpstreamPayloads(upstreamRes) {
427
+ const reader = upstreamRes.body.getReader();
428
+ const decoder = new TextDecoder('utf8');
429
+ let buffer = '';
430
+ let finished = false;
431
+ try {
432
+ while (true) {
433
+ const { done, value } = await reader.read();
434
+ if (done) {
435
+ finished = true;
436
+ break;
437
+ }
438
+ buffer += decoder.decode(value, { stream: true });
439
+ const lines = buffer.split('\n');
440
+ buffer = lines.pop();
441
+ for (const line of lines) {
442
+ const parsed = parsePayloadLine(line);
443
+ if (parsed) yield parsed;
444
+ }
445
+ }
446
+ buffer += decoder.decode();
447
+ const tail = parsePayloadLine(buffer);
448
+ if (tail) yield tail;
449
+ } finally {
450
+ // A consumer that stops early (an in-band error) must close the upstream connection too.
451
+ if (!finished) await reader.cancel().catch(() => {});
452
+ try { reader.releaseLock(); } catch {}
453
+ }
454
+ }
455
+
456
+ function parsePayloadLine(line) {
457
+ let s = line.trim();
458
+ if (!s) return null;
459
+ if (s.startsWith('data:')) s = s.slice(5).trim();
460
+ else if (s.startsWith('event:') || s.startsWith(':') || s.startsWith('id:') || s.startsWith('retry:')) return null;
461
+ if (!s || s === '[DONE]' || !s.startsWith('{')) return null;
462
+ try { return JSON.parse(s); } catch { return null; }
463
+ }
464
+
465
+ function clientRenderer(clientFormat, res, model, opts = {}) {
466
+ if (clientFormat === 'anthropic') return createAnthropicStream((e, d) => sendSSE(res, e, d), model);
467
+ if (clientFormat === 'responses') return createResponsesStream((e, d) => sendSSE(res, e, d), model, opts);
468
+ if (clientFormat === 'vertex') return createVertexStream((e, d) => sendSSE(res, null, d), model);
469
+ return createChatStream((e, d) => sendSSE(res, null, d), model);
470
+ }
471
+
472
+ function clientMessage(clientFormat, args) {
473
+ if (clientFormat === 'anthropic') return buildAnthropicMessage(args);
474
+ if (clientFormat === 'responses') return buildResponsesMessage(args);
475
+ if (clientFormat === 'vertex') return buildVertexMessage(args);
476
+ return buildChatMessage(args);
477
+ }
478
+
479
+ function previewOf(ir) {
480
+ const lastUser = (ir.messages || []).slice().reverse().find(m => m.role === 'user');
481
+ if (!lastUser) return '';
482
+ if (typeof lastUser.content === 'string') return lastUser.content.slice(0, 300);
483
+ if (Array.isArray(lastUser.content)) {
484
+ return lastUser.content.map(p => (p.type === 'text' ? p.text : `[${p.type}]`)).join(' ').slice(0, 300);
485
+ }
486
+ return '';
487
+ }
488
+
489
+ // ----------------------------------------------------
490
+ // Mode 1: DIRECT PASS-THROUGH (Native Anthropic to Native Anthropic)
491
+ // ----------------------------------------------------
492
+ const HOP_BY_HOP = new Set(['content-length', 'content-encoding', 'transfer-encoding', 'connection', 'keep-alive']);
493
+
494
+ async function forwardAnthropicDirect(res, payload, bodyBuffer, url, headers, mappedModel, signal, profile) {
495
+ debugLog('Direct forward to native Anthropic endpoint:', url);
496
+
497
+ // Only re-serialize when a fix is actually needed; otherwise forward the client's original bytes.
498
+ let json = payload;
499
+ let modified = false;
500
+ if (mappedModel && json.model !== mappedModel) {
501
+ json = { ...json, model: mappedModel };
502
+ modified = true;
503
+ }
504
+ const healed = healAnthropicPayload(json);
505
+ if (healed.changed) {
506
+ json = healed.payload;
507
+ modified = true;
508
+ debugLog('Healer (direct):', healed.notes.join('; '));
509
+ }
510
+ if (profile?.thinkingMode === 'off' && json.thinking) {
511
+ json = { ...json };
512
+ delete json.thinking;
513
+ modified = true;
514
+ }
515
+ const body = modified ? Buffer.from(JSON.stringify(json), 'utf8') : bodyBuffer;
516
+
517
+ const upstreamRes = await fetch(url, { method: 'POST', headers, body, signal });
518
+ const resHeaders = {};
519
+ for (const [k, v] of upstreamRes.headers.entries()) {
520
+ if (!HOP_BY_HOP.has(k.toLowerCase())) resHeaders[k] = v;
521
+ }
522
+ res.writeHead(upstreamRes.status, resHeaders);
523
+
524
+ let errorPreview = '';
525
+ const usage = createUsageTap();
526
+ const reader = upstreamRes.body.getReader();
527
+ let status = upstreamRes.status;
528
+ let streamError = null;
529
+ try {
530
+ while (true) {
531
+ const { done, value } = await reader.read();
532
+ if (done) break;
533
+ if (!upstreamRes.ok && errorPreview.length < 300) errorPreview += Buffer.from(value).toString('utf8');
534
+ usage.push(value);
535
+ res.write(value);
536
+ await drained(res);
537
+ }
538
+ } catch (streamErr) {
539
+ // The status line is already sent, so only the log can say that the stream broke.
540
+ if (signal.aborted) {
541
+ status = 499;
542
+ streamError = 'client disconnected mid-stream';
543
+ } else {
544
+ console.error('[DirectForward] Stream error:', streamErr.message);
545
+ status = 502;
546
+ streamError = `stream interrupted: ${streamErr.cause?.message || streamErr.message}`;
547
+ }
548
+ } finally {
549
+ res.end();
550
+ }
551
+ const error = streamError || (upstreamRes.ok ? null : errorPreview.slice(0, 300));
552
+ return { status, error, healed: healed.notes, tokens: usage.tokens() };
553
+ }
554
+
555
+ // Reads the token counts out of the passthrough bytes without changing them: message_start and
556
+ // message_delta in a stream, `usage` in a JSON body.
557
+ function createUsageTap(limit = 4 * 1024 * 1024) {
558
+ const decoder = new TextDecoder('utf8');
559
+ let text = '';
560
+ let seen = 0;
561
+ const tokens = { prompt: 0, completion: 0 };
562
+ const take = (u) => {
563
+ if (!u || typeof u !== 'object') return;
564
+ if (typeof u.input_tokens === 'number') tokens.prompt = u.input_tokens + (u.cache_read_input_tokens || 0) + (u.cache_creation_input_tokens || 0);
565
+ if (typeof u.output_tokens === 'number') tokens.completion = u.output_tokens;
566
+ };
567
+ const line = (l) => {
568
+ const t = l.startsWith('data:') ? l.slice(5).trim() : '';
569
+ if (!t.includes('"usage"')) return;
570
+ try {
571
+ const d = JSON.parse(t);
572
+ take(d.message?.usage || d.usage);
573
+ } catch {}
574
+ };
575
+ return {
576
+ push(chunk) {
577
+ if (seen > limit) return;
578
+ seen += chunk.length;
579
+ text += decoder.decode(chunk, { stream: true });
580
+ const lines = text.split('\n');
581
+ text = lines.pop();
582
+ for (const l of lines) line(l);
583
+ },
584
+ tokens() {
585
+ text += decoder.decode();
586
+ if (text.trim().startsWith('{')) {
587
+ try { take(JSON.parse(text).usage); } catch {}
588
+ } else if (text) line(text);
589
+ return tokens;
590
+ }
591
+ };
592
+ }
593
+
594
+ function cleanSchemaDeep(obj) {
595
+ if (!obj || typeof obj !== 'object') return obj;
596
+ if (Array.isArray(obj)) return obj.map(cleanSchemaDeep);
597
+ const res = {};
598
+ for (const [k, v] of Object.entries(obj)) {
599
+ if (k === 'encrypted' || k === '$schema' || k === 'cache_control') continue;
600
+ if (k === 'properties' && v && typeof v === 'object') {
601
+ const cleanProps = {};
602
+ for (const [pk, pv] of Object.entries(v)) {
603
+ if (typeof pv === 'string') {
604
+ cleanProps[pk] = { type: pv === 'object' ? 'object' : pv, ...(pv === 'object' ? { properties: {} } : {}) };
605
+ } else {
606
+ cleanProps[pk] = cleanSchemaDeep(pv);
607
+ }
608
+ }
609
+ res[k] = cleanProps;
610
+ } else {
611
+ res[k] = cleanSchemaDeep(v);
612
+ }
613
+ }
614
+ if (!res.type && res.properties) res.type = 'object';
615
+ if (res.type === 'object' && !res.properties) res.properties = {};
616
+ return res;
617
+ }
618
+
619
+ // The upstream request of the HTTP and the WS transport: the IR as an upstream body, with tool schemas
620
+ // made acceptable to the target:
621
+ // 1. Strip disallowed keywords ('encrypted', '$schema', 'cache_control')
622
+ // 2. Fix invalid schema values where a property has a string value "object" instead of a valid schema object
623
+ // 3. For ag/* targets (Gemini behind 9Router): rewrite to the strict Schema subset
624
+ function buildUpstreamRequest(profile, outFormat, ir, mappedModel, req) {
625
+ const upBody = emitUpstreamBody(outFormat, ir, mappedModel, { thinkingMode: profile.thinkingMode });
626
+ if (upBody?.tools && Array.isArray(upBody.tools)) {
627
+ upBody.tools = cleanSchemaDeep(upBody.tools);
628
+ if (outFormat === 'openai-chat' && isAntigravityModel(mappedModel)) {
629
+ upBody.tools = geminiSafeTools(upBody.tools);
630
+ }
631
+ }
632
+ const { url, headers } = upstreamEndpoint(profile, outFormat, mappedModel, ir.stream, req);
633
+ return { url, headers, upBody };
634
+ }
635
+
636
+ // Feeds upstream stream events to a client renderer until the stream ends, fails or is aborted.
637
+ // Returns the stream error, or null. `sink` is the client stream whose backpressure it waits for.
638
+ async function pumpStream(upstreamRes, { normalize, col, renderer, splitter, signal, sink, tag }) {
639
+ let streamError = null;
640
+ let events = 0;
641
+ try {
642
+ for await (const parsed of readUpstreamPayloads(upstreamRes)) {
643
+ events++;
644
+ const ev = normalize(parsed);
645
+ col.add(ev);
646
+ if (ev.error) {
647
+ streamError = ev.error;
648
+ break;
649
+ }
650
+ for (const t of ev.think) renderer.think(t.text, t.sig);
651
+ if (ev.sig) renderer.think('', ev.sig);
652
+ for (const t of ev.text) splitter.push(t);
653
+ if (ev.tools.length) {
654
+ splitter.flush();
655
+ for (const tc of ev.tools) renderer.tool(tc);
656
+ }
657
+ await drained(sink, signal);
658
+ }
659
+ if (!streamError && events === 0) streamError = 'Upstream returned an empty stream';
660
+ } catch (streamErr) {
661
+ if (!signal.aborted) {
662
+ console.error(`[${tag}] Stream error:`, streamErr.message);
663
+ streamError = streamErr.message || 'stream interrupted';
664
+ }
665
+ }
666
+ return streamError;
667
+ }
668
+
669
+ // Off unless config.json carries a contractLab block with enabled: true. Every call returns at
670
+ // once, so a slow or absent intact cannot reach the answer the client is waiting for.
671
+ const contractLab = createContractLab({ settings: () => contractLabSettings(loadConfig()) });
672
+
673
+ // ----------------------------------------------------
674
+ // LLM Switcher generic pipeline: client --parse--> IR --emit--> upstream
675
+ // Client (input) formats : anthropic | openai-chat | responses (Codex) | vertex
676
+ // Upstream (output) : profile.outFormat or legacy mode mapping
677
+ // direct -> anthropic | convert -> openai-chat
678
+ // hybrid -> claude-* via native anthropic, others via openai-chat
679
+ // ----------------------------------------------------
680
+ async function handleConvert(clientFormat, req, res, bodyBuffer, opts = {}) {
681
+ const { profileKey, profile, error: profileError } = getActiveProfile(clientFormat, req);
682
+ if (!loadConfig()) {
683
+ sendClientError(res, clientFormat, 500, `LLM Switcher config not loaded (${configPath}): ${getConfigLoadError()?.message || 'missing file'}`);
684
+ return;
685
+ }
686
+ if (!profile) {
687
+ sendClientError(res, clientFormat, profileError ? 400 : 503, profileError ||
688
+ `Proxy is currently OFF for ${clientFormat}. Set an active profile for this target in Web UI or via switch command.`);
689
+ return;
690
+ }
691
+
692
+ let payload;
693
+ try {
694
+ payload = JSON.parse(bodyBuffer.toString('utf8'));
695
+ } catch {
696
+ sendClientError(res, clientFormat, 400, 'Invalid JSON body');
697
+ return;
698
+ }
699
+
700
+ const wantIn = profile.inFormat || 'auto';
701
+ if (wantIn !== 'auto' && wantIn !== clientFormat) {
702
+ sendClientError(res, clientFormat, 400, `Profile "${profileKey}" expects "${wantIn}" input, got "${clientFormat}"`);
703
+ return;
704
+ }
705
+
706
+ let ir;
707
+ try {
708
+ ir = parseToIR(clientFormat, payload);
709
+ } catch (e) {
710
+ sendClientError(res, clientFormat, 400, `Cannot parse ${clientFormat} request: ${e.message}`);
711
+ return;
712
+ }
713
+ if (clientFormat === 'vertex') {
714
+ ir.stream = Boolean(opts.vertexStream);
715
+ if (!ir.model && opts.vertexModel) ir.model = opts.vertexModel;
716
+ }
717
+
718
+ const reqStartTime = Date.now();
719
+ const requestPreview = previewOf(ir);
720
+ const requestedModel = ir.model || payload.model || '';
721
+ const mappedModel = mapModel(requestedModel, profile, clientFormat);
722
+ const outFormat = resolveOutFormat(profile, mappedModel);
723
+ console.log(`[llm-switcher] ${clientFormat} -> ${outFormat} "${requestedModel}" -> "${mappedModel}" [${profile.name || profileKey}]`);
724
+
725
+ const logBase = { clientFormat, outFormat, profile: profileKey, model: mappedModel, stream: ir.stream, requestPreview };
726
+ const log = (extra) => logInspection({ ...logBase, duration: Date.now() - reqStartTime, tokens: { prompt: 0, completion: 0 }, ...extra });
727
+
728
+ // Contract lab: a sampled request carries a trace id to intact, and the bytes this gateway
729
+ // writes back are copied for the upload that follows the answer.
730
+ const traceId = probeTraceId(req) || contractLab.traceFor(mappedModel);
731
+ const halfTap = traceId ? createHalfTap() : null;
732
+ if (halfTap) tapClientWrites(res, halfTap);
733
+
734
+ // AbortController to cancel the upstream fetch as soon as the client disconnects (saves tokens)
735
+ const ac = new AbortController();
736
+ const onClientClose = () => {
737
+ if (!res.writableEnded) {
738
+ debugLog(`[${profileKey}] Client connection closed before response ended, aborting upstream request`);
739
+ ac.abort();
740
+ }
741
+ };
742
+ res.on('close', onClientClose);
743
+ let answered = false; // set only after a complete 2xx answer; the half upload depends on it
744
+
745
+ try {
746
+ // Fast path: anthropic in/out goes straight through, preserving original bytes (including thinking signatures).
747
+ // Note: this branch skips the Healer Engine because it bypasses the IR.
748
+ if (clientFormat === 'anthropic' && outFormat === 'anthropic') {
749
+ const { url, headers } = upstreamEndpoint(profile, 'anthropic', mappedModel, ir.stream, req);
750
+ if (traceId) headers['x-intact-trace'] = traceId;
751
+ try {
752
+ const r = await forwardAnthropicDirect(res, payload, bodyBuffer, url, headers, mappedModel, ac.signal, profile);
753
+ answered = !r.error && r.status >= 200 && r.status < 300;
754
+ log({ status: r.status, tokens: r.tokens, responsePreview: r.healed.length ? `(direct forward, healed: ${r.healed.join('; ')})` : '(direct forward)', error: r.error || undefined });
755
+ } catch (err) {
756
+ if (ac.signal.aborted) return log({ status: 499, error: 'client disconnected' });
757
+ console.error(`[${profileKey}] Direct forward error:`, err.message);
758
+ sendClientError(res, clientFormat, 502, `Direct forward error: ${err.message}`);
759
+ log({ status: 502, error: err.message });
760
+ }
761
+ return;
762
+ }
763
+
764
+ const { url, headers, upBody } = buildUpstreamRequest(profile, outFormat, ir, mappedModel, req);
765
+ if (traceId) headers['x-intact-trace'] = traceId;
766
+ debugLog(`[${profileKey}] ${clientFormat} -> ${outFormat} ${url} ::`, JSON.stringify(upBody).slice(0, 500));
767
+
768
+ let upstreamRes;
769
+ try {
770
+ upstreamRes = await fetch(url, { method: 'POST', headers, body: JSON.stringify(upBody), signal: ac.signal });
771
+ } catch (fetchErr) {
772
+ if (ac.signal.aborted) return log({ status: 499, error: 'client disconnected' });
773
+ console.error(`[${profileKey}] Network error:`, fetchErr.message);
774
+ sendClientError(res, clientFormat, 502, `Failed to connect to upstream: ${fetchErr.cause?.message || fetchErr.message}`);
775
+ return log({ status: 502, error: fetchErr.message });
776
+ }
777
+
778
+ if (!upstreamRes.ok) {
779
+ const errText = await upstreamRes.text().catch(() => '');
780
+ console.error(`[${profileKey}] Error HTTP ${upstreamRes.status}:`, errText.slice(0, 500));
781
+ sendClientError(res, clientFormat, upstreamRes.status, extractUpstreamMessage(errText) || `Upstream HTTP ${upstreamRes.status}`, pickRetryHeaders(upstreamRes));
782
+ return log({ status: upstreamRes.status, error: errText.slice(0, 300) });
783
+ }
784
+
785
+ const normalize = createUpstreamNormalizer(outFormat);
786
+ const col = createCollector();
787
+
788
+ // ---- non-stream ----
789
+ if (!ir.stream) {
790
+ let json;
791
+ try {
792
+ json = await upstreamRes.json();
793
+ } catch (e) {
794
+ if (ac.signal.aborted) return log({ status: 499, error: 'client disconnected' });
795
+ sendClientError(res, clientFormat, 502, `Upstream returned non-JSON: ${e.message}`);
796
+ return log({ status: 502, error: e.message });
797
+ }
798
+ col.add(normalize(json));
799
+ if (col.error) {
800
+ sendClientError(res, clientFormat, 502, `Upstream error: ${col.error}`);
801
+ return log({ status: 502, error: String(col.error).slice(0, 300) });
802
+ }
803
+ const split = splitThinkTags(col.text.join(''));
804
+ const think = [...col.think, split.think].filter(Boolean);
805
+ const tools = [...col.tools.values()].sort((a, b) => a.index - b.index);
806
+ const completion = col.completion();
807
+ const out = clientMessage(clientFormat, {
808
+ model: requestedModel || mappedModel, think, text: [split.text], tools, toolMeta: ir.toolMeta,
809
+ finish: col.finish, prompt: col.prompt, completion, cached: col.cached,
810
+ reasoning: col.reasoning, sig: col.sig
811
+ });
812
+ sendJson(res, 200, out);
813
+ answered = true;
814
+ return log({
815
+ status: 200, tokens: { prompt: col.prompt, completion },
816
+ thinkingChars: think.join('').length, responsePreview: split.text.slice(0, 300)
817
+ });
818
+ }
819
+
820
+ // ---- stream ----
821
+ res.writeHead(200, {
822
+ 'Content-Type': 'text/event-stream; charset=utf-8',
823
+ 'Cache-Control': 'no-cache',
824
+ 'Connection': 'keep-alive',
825
+ 'X-Accel-Buffering': 'no'
826
+ });
827
+ const renderer = clientRenderer(clientFormat, res, requestedModel || mappedModel, { toolMeta: ir.toolMeta });
828
+ renderer.start();
829
+ const splitter = createThinkTagSplitter(t => renderer.think(t), t => renderer.text(t));
830
+ const streamError = await pumpStream(upstreamRes, { normalize, col, renderer, splitter, signal: ac.signal, sink: res, tag: profileKey });
831
+
832
+ if (ac.signal.aborted) {
833
+ return log({ status: 499, error: 'client disconnected mid-stream', responsePreview: col.text.join('').slice(0, 300) });
834
+ }
835
+ splitter.flush();
836
+ const completion = col.completion();
837
+ if (streamError) {
838
+ // Report the error clearly instead of a fake "end_turn" ending -> the client knows the response was cut off and can retry.
839
+ renderer.error(streamError);
840
+ } else {
841
+ renderer.finish(col.finish, { completion, prompt: col.prompt, cached: col.cached, reasoning: col.reasoning, hasTools: col.tools.size > 0 });
842
+ }
843
+ // OpenAI Chat clients expect a terminal [DONE] line (Responses API does not use [DONE]).
844
+ if (clientFormat === 'openai-chat') res.write('data: [DONE]\n\n');
845
+ res.end();
846
+ answered = !streamError;
847
+ log({
848
+ status: streamError ? 502 : 200, stream: true,
849
+ tokens: { prompt: col.prompt, completion },
850
+ thinkingChars: col.think.join('').length,
851
+ responsePreview: col.text.join('').slice(0, 300),
852
+ error: streamError ? String(streamError).slice(0, 300) : undefined
853
+ });
854
+ } finally {
855
+ res.off('close', onClientClose);
856
+ // A half is uploaded only for a complete 2xx answer: anything else would diff as a loss the
857
+ // converter never made.
858
+ if (traceId) {
859
+ finishHalf(contractLab, traceId, ac.signal.aborted || !answered, {
860
+ toolRequest: capText(bodyBuffer),
861
+ toolResponse: halfTap?.text() || '',
862
+ toolVersion: toolVersionFromUA(req.headers['user-agent']),
863
+ inFormat: clientFormat,
864
+ outFormat
865
+ });
866
+ }
867
+ }
868
+ }
869
+
870
+ // Claude Code calls /v1/messages/count_tokens to measure context. Native Anthropic upstream -> ask for the real count;
871
+ // other upstreams have no equivalent endpoint -> estimate (skip image base64, add a fixed cost per image).
872
+ async function handleCountTokens(req, res, buf) {
873
+ let payload;
874
+ try {
875
+ payload = JSON.parse(buf.toString('utf8'));
876
+ } catch {
877
+ return sendClientError(res, 'anthropic', 400, 'Invalid JSON body');
878
+ }
879
+ const { profile } = getActiveProfile('anthropic', req);
880
+ if (profile) {
881
+ const mappedModel = mapModel(payload.model || '', profile, 'anthropic');
882
+ if (resolveOutFormat(profile, mappedModel) === 'anthropic') {
883
+ const { url, headers } = upstreamEndpoint(profile, 'anthropic', mappedModel, false, req);
884
+ const countUrl = profile.endpoints?.countTokens || url.replace(/\/messages$/, '/messages/count_tokens');
885
+ try {
886
+ const body = healAnthropicPayload({ ...payload, model: mappedModel }).payload;
887
+ const r = await fetch(countUrl, { method: 'POST', headers, body: JSON.stringify(body), signal: AbortSignal.timeout(15000) });
888
+ if (r.ok) {
889
+ const j = await r.json();
890
+ if (typeof j.input_tokens === 'number') return sendJson(res, 200, { input_tokens: j.input_tokens });
891
+ } else {
892
+ debugLog(`count_tokens upstream HTTP ${r.status}, falling back to estimate`);
893
+ }
894
+ } catch (err) {
895
+ debugLog('count_tokens upstream failed, falling back to estimate:', err.message);
896
+ }
897
+ }
898
+ }
899
+ return sendJson(res, 200, { input_tokens: estimateTokens(payload) });
900
+ }
901
+
902
+ // ----------------------------------------------------
903
+ // Admin API helpers
904
+ // ----------------------------------------------------
905
+ // loadConfig keeps serving the last good copy when config.json stops parsing. The admin API must
906
+ // not act on that copy: a save would replace the user's hand edit with stale data.
907
+ function requireConfig(res) {
908
+ const cfg = loadConfig();
909
+ const loadError = getConfigLoadError();
910
+ if (cfg && loadError && fs.existsSync(configPath)) {
911
+ sendJson(res, 409, { error: `config.json (${configPath}) does not parse: ${loadError.message}. Fix the file; the gateway does not overwrite it until it parses.` });
912
+ return null;
913
+ }
914
+ if (!cfg) {
915
+ sendJson(res, 500, { error: `Config not loaded (${configPath}): ${loadError?.message || 'missing file'}. Copy config.example.json to config.json.` });
916
+ }
917
+ return cfg;
918
+ }
919
+
920
+ // Config changes and interceptor reconciles run one at a time, from read to save. A change that
921
+ // waits for its interceptor check must not be saved by a concurrent one, and two must not both
922
+ // start an interceptor.
923
+ let adminChain = Promise.resolve();
924
+ function serialized(fn) {
925
+ const run = adminChain.then(fn);
926
+ adminChain = run.catch(() => {});
927
+ return run;
928
+ }
929
+ const CONFIG_CHANGES = new Set(['/api/switch', '/api/toggle', '/api/save-profile', '/api/delete-profile', '/api/blindfold/sync']);
930
+
931
+ // Callers run it inside serialized().
932
+ function reconcile(cfg) {
933
+ return reconcileBlindfold(cfg, PORT).then(r => {
934
+ if (!r.ok) console.error(`[llm-switcher:blindfold] ${r.error}`);
935
+ else if (r.action === 'started') console.log('[llm-switcher:blindfold] interceptor started');
936
+ return r;
937
+ });
938
+ }
939
+
940
+ // Returns what the caller must show: removed settings.json values, and a blindfold failure.
941
+ // `base` is the revision the change started from.
942
+ async function commit(cfg, base) {
943
+ // Refuse before saving: a saved interceptor port that a squatter holds would route Codex through it.
944
+ const problem = await checkBlindfoldTarget(cfg, PORT);
945
+ if (problem) return { success: false, error: `Not saved: ${problem}` };
946
+ // The admin lock covers this process only. The CLI can save config.json during the check above,
947
+ // and a save now would overwrite it.
948
+ const current = loadConfig();
949
+ if (!current || configRevision(current) !== base) {
950
+ return { success: false, status: 409, error: 'Not saved: config.json changed while this change was checked. Reload and try again.' };
951
+ }
952
+ saveConfig(cfg);
953
+ const revision = configRevision(cfg);
954
+ const st = applyLaunchState(cfg, PORT);
955
+ const removed = st.settings?.removed || [];
956
+ if (removed.length) console.log(`[llm-switcher] settings.json: removed switcher-written values: ${removed.join(', ')}`);
957
+ const bf = await reconcile(cfg);
958
+ return {
959
+ revision,
960
+ settingsRemoved: removed,
961
+ ...(st.envWriteError ? { envWriteError: st.envWriteError } : {}),
962
+ ...(bf.ok ? {} : { success: false, error: `Saved, but the blindfold interceptor is not in line: ${bf.error}` })
963
+ };
964
+ }
965
+
966
+ const VALID_MODES = ['hybrid', 'convert', 'direct'];
967
+
968
+ const CONTROL_CHARS = /[\u0000-\u001f\u007f-\u009f]/;
969
+
970
+ function validateProfileInput(p) {
971
+ if (!p || typeof p !== 'object' || Array.isArray(p)) return 'Profile must be an object';
972
+ // These strings reach the terminal through `switch status`; a control character could rewrite it.
973
+ for (const k of ['name', 'baseURL', 'optimizerURL']) {
974
+ if (typeof p[k] === 'string' && CONTROL_CHARS.test(p[k])) return `${k} must not contain control characters`;
975
+ }
976
+ if (p.inFormat && p.inFormat !== 'auto' && !IN_FORMATS.includes(p.inFormat)) return `Invalid inFormat "${p.inFormat}"`;
977
+ if (p.outFormat && !OUT_FORMATS.includes(p.outFormat)) return `Invalid outFormat "${p.outFormat}"`;
978
+ if (p.mode && !VALID_MODES.includes(p.mode)) return `Invalid mode "${p.mode}"`;
979
+ if (p.thinkingMode && !THINKING_MODES.includes(p.thinkingMode)) return `Invalid thinkingMode "${p.thinkingMode}"`;
980
+ if (p.baseURL !== undefined) {
981
+ try {
982
+ const u = new URL(p.baseURL);
983
+ if (!['http:', 'https:'].includes(u.protocol)) return 'baseURL must be http(s)';
984
+ } catch {
985
+ return `Invalid baseURL "${p.baseURL}"`;
986
+ }
987
+ }
988
+ // Names the CLI receives end up on a command line. state.mjs drops an unsafe one
989
+ // before the write; refusing it here says why instead of losing it in silence.
990
+ if (p.publicModels !== undefined) {
991
+ if (!Array.isArray(p.publicModels)) return 'publicModels must be an array';
992
+ for (const name of p.publicModels) {
993
+ if (name !== '' && !isSafeModelName(name)) return `Invalid publicModels entry "${name}"`;
994
+ }
995
+ }
996
+ if (p.codexRoles !== undefined) {
997
+ if (!p.codexRoles || typeof p.codexRoles !== 'object' || Array.isArray(p.codexRoles)) {
998
+ return 'codexRoles must be an object';
999
+ }
1000
+ for (const [slot, name] of Object.entries(p.codexRoles)) {
1001
+ if (name !== '' && !isSafeModelName(name)) return `Invalid codexRoles.${slot} value "${name}"`;
1002
+ }
1003
+ }
1004
+ if (p.blindfoldPort !== undefined && p.blindfoldPort !== '' && !parsePort(p.blindfoldPort)) {
1005
+ return `Invalid blindfoldPort "${p.blindfoldPort}"`;
1006
+ }
1007
+ if (p.blindfoldHost !== undefined && p.blindfoldHost !== '' && !/^[A-Za-z0-9.-]{1,253}$/.test(p.blindfoldHost)) {
1008
+ return `Invalid blindfoldHost "${p.blindfoldHost}"`;
1009
+ }
1010
+ if (p.blindfoldPrefix !== undefined && p.blindfoldPrefix !== '' && !/^\/[A-Za-z0-9._~/-]{0,200}$/.test(p.blindfoldPrefix)) {
1011
+ return `Invalid blindfoldPrefix "${p.blindfoldPrefix}"`;
1012
+ }
1013
+ return null;
1014
+ }
1015
+
1016
+ // API keys are masked with MASKED_KEY in the UI; if the client sends back the masked value, reuse the stored real key.
1017
+ // Only for the stored baseURL: otherwise the masked value would send the real key to any host the caller names.
1018
+ const sameBaseURL = (a, b) => String(a || '').replace(/\/+$/, '') === String(b || '').replace(/\/+$/, '');
1019
+ // endpoints override baseURL per format, so they are part of where the key goes.
1020
+ const sameDestination = (a, b) => sameBaseURL(a.baseURL, b.baseURL) && JSON.stringify(a.endpoints || {}) === JSON.stringify(b.endpoints || {});
1021
+
1022
+ // Changes when config.json changes. The dashboard sends it back, so a change made from a stale
1023
+ // page is refused instead of overwriting what another tab or the CLI saved.
1024
+ function configRevision(cfg) {
1025
+ return crypto.createHash('sha256').update(JSON.stringify(cfg)).digest('hex').slice(0, 16);
1026
+ }
1027
+
1028
+ function resolveApiKey(cfg, profileKey, apiKey, baseURL) {
1029
+ if (apiKey !== MASKED_KEY) return apiKey || '';
1030
+ if (!hasProfile(cfg, profileKey)) return '';
1031
+ const stored = cfg.profiles[profileKey];
1032
+ if (baseURL !== undefined && !sameBaseURL(baseURL, stored.baseURL)) return '';
1033
+ return stored.apiKey || '';
1034
+ }
1035
+
1036
+ function upstreamTimeout(ms) {
1037
+ return AbortSignal.timeout(ms);
1038
+ }
1039
+
1040
+ async function testUpstream(body, cfg) {
1041
+ const baseURL = String(body.baseURL || '').replace(/\/+$/, '');
1042
+ if (!baseURL) return { status: 400, json: { ok: false, error: 'Missing baseURL' } };
1043
+ const apiKey = resolveApiKey(cfg, body.key, body.apiKey, baseURL);
1044
+ const model = body.model || 'default';
1045
+ const profile = { baseURL, apiKey, mode: body.mode, outFormat: body.outFormat || undefined };
1046
+ const outFormat = resolveOutFormat(profile, model);
1047
+ const { url, headers } = upstreamEndpoint(profile, outFormat, model, false, null);
1048
+ const ir = { model, system: '', messages: [{ role: 'user', content: 'ping' }], tools: [], toolChoice: null, params: { maxTokens: 16, temperature: null, topP: null, topK: null, stop: [] }, thinking: { type: 'disabled' }, stream: false };
1049
+ const start = Date.now();
1050
+ const r = await fetch(url, { method: 'POST', headers, body: JSON.stringify(emitUpstreamBody(outFormat, ir, model)), signal: upstreamTimeout(30000) });
1051
+ const latency = Date.now() - start;
1052
+ if (!r.ok) return { status: 200, json: { ok: false, status: r.status, latency, outFormat, error: (await r.text()).slice(0, 2000) } };
1053
+ const data = await r.json().catch(() => ({}));
1054
+ const col = createCollector();
1055
+ col.add(createUpstreamNormalizer(outFormat)(data));
1056
+ const sample = col.text.join('') || col.think.join('') || '(ok, empty response)';
1057
+ return { status: 200, json: { ok: true, latency, outFormat, sample: sample.slice(0, 500) } };
1058
+ }
1059
+
1060
+ async function fetchModels(body, cfg) {
1061
+ const baseURL = String(body.baseURL || '').replace(/\/+$/, '');
1062
+ if (!baseURL) return { status: 400, json: { ok: false, error: 'Missing baseURL' } };
1063
+ const apiKey = resolveApiKey(cfg, body.key, body.apiKey, baseURL);
1064
+ const headers = {};
1065
+ if (apiKey) {
1066
+ headers['Authorization'] = `Bearer ${apiKey}`;
1067
+ headers['x-api-key'] = apiKey;
1068
+ }
1069
+ const r = await fetch(`${baseURL}/models`, { headers, signal: upstreamTimeout(15000) });
1070
+ if (!r.ok) return { status: 200, json: { ok: false, status: r.status, error: (await r.text()).slice(0, 2000) } };
1071
+ const data = await r.json();
1072
+ let list = [];
1073
+ if (Array.isArray(data.data)) list = data.data.map(m => (typeof m === 'string' ? m : m.id));
1074
+ else if (Array.isArray(data)) list = data.map(m => (typeof m === 'string' ? m : m.id));
1075
+ else if (Array.isArray(data.models)) list = data.models.map(m => (typeof m === 'string' ? m : (m.id || m.name)));
1076
+ list = [...new Set(list.filter(Boolean).map(id => String(id).replace(/^models\//, '')))].sort();
1077
+ return { status: 200, json: { ok: true, models: list } };
1078
+ }
1079
+
1080
+ // Vertex/Gemini: /v1beta/models/{m}:{action} (Gemini API) and
1081
+ // /v1/projects/{p}/locations/{l}/publishers/{pub}/models/{m}:{action} (Vertex AI SDK).
1082
+ const VERTEX_ROUTE = /^\/(?:v1|v1beta|v1beta1)\/(?:projects\/[^/]+\/locations\/[^/]+\/publishers\/[^/]+\/)?models\/([^/:]+):(generateContent|streamGenerateContent)$/;
1083
+
1084
+ // ----------------------------------------------------
1085
+ // Router
1086
+ // ----------------------------------------------------
1087
+ async function route(req, res) {
1088
+ const parsedUrl = new URL(req.url, 'http://127.0.0.1');
1089
+ const pathname = parsedUrl.pathname;
1090
+ const method = req.method;
1091
+
1092
+ const guardError = checkRequestOrigin(req);
1093
+ if (guardError) {
1094
+ req.resume();
1095
+ return sendJson(res, 403, { error: guardError });
1096
+ }
1097
+ if (method === 'OPTIONS') {
1098
+ // Only answer preflight for the UI's own origin (already past checkRequestOrigin).
1099
+ res.writeHead(204, {
1100
+ 'Access-Control-Allow-Origin': req.headers.origin || `http://127.0.0.1:${PORT}`,
1101
+ 'Access-Control-Allow-Methods': 'GET, POST, OPTIONS',
1102
+ 'Access-Control-Allow-Headers': 'Content-Type, Authorization'
1103
+ });
1104
+ return res.end();
1105
+ }
1106
+
1107
+ // Serve Web UI (no-cache: always serve the latest version after file edits)
1108
+ if (method === 'GET' && (pathname === '/' || pathname === '/ui')) {
1109
+ if (fs.existsSync(uiHtmlPath)) {
1110
+ res.writeHead(200, {
1111
+ 'Content-Type': 'text/html; charset=utf-8',
1112
+ 'Cache-Control': 'no-cache, no-store, must-revalidate',
1113
+ 'X-Frame-Options': 'DENY',
1114
+ 'X-Content-Type-Options': 'nosniff'
1115
+ });
1116
+ return res.end(fs.readFileSync(uiHtmlPath, 'utf8'));
1117
+ }
1118
+ }
1119
+
1120
+ // Health check
1121
+ if (method === 'GET' && pathname === '/health') {
1122
+ const { profileKey, profile } = getFirstActiveProfile(TARGETS, req);
1123
+ // ?challenge=<nonce> lets the CLI tell this gateway from a process that replays a /health body.
1124
+ const challenge = parsedUrl.searchParams.get('challenge');
1125
+ return sendJson(res, 200, {
1126
+ status: 'ok',
1127
+ proxy: 'llm-switcher',
1128
+ ...(challenge ? {
1129
+ pid: process.pid,
1130
+ proof: identityProof(challenge, { role: 'gateway', port: PORT, pid: process.pid }, ADMIN_TOKEN.toString())
1131
+ } : {}),
1132
+ port: PORT,
1133
+ configLoaded: Boolean(loadConfig()),
1134
+ activeProfile: profileKey || '(none)',
1135
+ activeProfiles: loadConfig() ? getActiveMap(loadConfig()) : {},
1136
+ mode: profile?.mode || 'hybrid',
1137
+ inFormat: profile?.inFormat || 'auto',
1138
+ outFormat: profile ? resolveOutFormat(profile, '') : 'none',
1139
+ upstream: profile?.baseURL || '(none)'
1140
+ });
1141
+ }
1142
+
1143
+ // OpenAI-style model list (Codex / OpenAI SDK discovery).
1144
+ // Served IDs come from profile.publicModels when set (official-facing names the
1145
+ // CLI already knows, e.g. gpt-5.6-sol) so the client never observes the internal
1146
+ // upstream IDs or slot aliases. Slot aliases (main, review, ...) are never
1147
+ // advertised: Codex sends them in-request and mapModel resolves them server-side.
1148
+ // Without publicModels, fall back to the deduplicated mapped upstream IDs.
1149
+ if (method === 'GET' && (pathname === '/v1/models' || pathname === '/models')) {
1150
+ const { ids, windows } = servedModels(req);
1151
+ const created = Math.floor(Date.now() / 1000);
1152
+ return sendJson(res, 200, {
1153
+ object: 'list',
1154
+ data: ids.map(id => ({ id, object: 'model', created, owned_by: 'system' })),
1155
+ models: ids.map(id => ({ ...codexModelEntry(id, windows.get(id)), id, object: 'model', created, owned_by: 'system' }))
1156
+ });
1157
+ }
1158
+
1159
+ // Individual model metadata: GET /v1/models/{id}
1160
+ if (method === 'GET' && (pathname.startsWith('/v1/models/') || pathname.startsWith('/models/'))) {
1161
+ const modelId = decodeURIComponent(pathname.replace(/^\/(v1\/)?models\//, ''));
1162
+ if (modelId) {
1163
+ const { windows } = servedModels(req);
1164
+ return sendJson(res, 200, { ...codexModelEntry(modelId, windows.get(modelId)), id: modelId, object: 'model', created: Math.floor(Date.now() / 1000), owned_by: 'system' });
1165
+ }
1166
+ }
1167
+
1168
+ if (pathname.startsWith('/api/')) {
1169
+ if (!isAdminRequest(req)) {
1170
+ req.resume();
1171
+ return sendJson(res, 401, { error: 'Unauthorized: send the x-llm-switcher-token header. Open the dashboard with `switch ui`.' });
1172
+ }
1173
+ return routeApi(req, res, method, pathname);
1174
+ }
1175
+
1176
+ // Token count estimation endpoint
1177
+ if (method === 'POST' && (pathname === '/v1/messages/count_tokens' || pathname === '/messages/count_tokens')) {
1178
+ const buf = await readBody(req, MAX_BODY_SIZE);
1179
+ return handleCountTokens(req, res, buf);
1180
+ }
1181
+
1182
+ // Client endpoints, one per input protocol (auto-detected by path).
1183
+ let clientFormat = null;
1184
+ const vmatch = method === 'POST' ? pathname.match(VERTEX_ROUTE) : null;
1185
+
1186
+ // Codex CLI sends GET /v1/responses (and /v1/responses/{id}) to fetch model
1187
+ // metadata and retrieve previous responses. The gateway is stateless, so
1188
+ // return a synthetic stub that satisfies the SDK's metadata lookup without
1189
+ // erroring out.
1190
+ if (method === 'GET' && (pathname === '/v1/responses' || pathname === '/responses' || pathname.startsWith('/v1/responses/') || pathname.startsWith('/responses/'))) {
1191
+ req.resume();
1192
+ if (pathname === '/v1/responses' || pathname === '/responses') {
1193
+ // Model metadata / list — return an empty list
1194
+ return sendJson(res, 200, { object: 'list', data: [] });
1195
+ }
1196
+ // GET /v1/responses/{id} — response retrieval; stateless gateway has no
1197
+ // persisted responses so return 404 in OpenAI's error shape.
1198
+ return sendJson(res, 404, {
1199
+ error: { message: 'Response not found. This gateway is stateless and does not persist responses.', type: 'not_found_error', code: '404' }
1200
+ });
1201
+ }
1202
+
1203
+ if (method === 'POST') {
1204
+ if (pathname === '/v1/messages' || pathname === '/messages') clientFormat = 'anthropic';
1205
+ else if (pathname === '/v1/chat/completions' || pathname === '/chat/completions') clientFormat = 'openai-chat';
1206
+ else if (pathname === '/v1/responses' || pathname === '/responses') clientFormat = 'responses';
1207
+ else if (vmatch) clientFormat = 'vertex';
1208
+ }
1209
+ if (clientFormat) {
1210
+ let buf;
1211
+ try {
1212
+ buf = await readBody(req, MAX_BODY_SIZE);
1213
+ } catch (err) {
1214
+ return sendClientError(res, clientFormat, err.status || 400, err.message);
1215
+ }
1216
+ const opts = vmatch ? { vertexModel: decodeURIComponent(vmatch[1]), vertexStream: vmatch[2] === 'streamGenerateContent' } : {};
1217
+ return handleConvert(clientFormat, req, res, buf, opts);
1218
+ }
1219
+
1220
+ req.resume();
1221
+ return sendJson(res, 404, { error: { message: `Not found: ${method} ${pathname}` } });
1222
+ }
1223
+
1224
+ // The names /v1/models serves, and the window of each. Official names when the profile publishes
1225
+ // them, otherwise the mapped upstream IDs; each window follows model1M of its slot.
1226
+ function servedModels(req) {
1227
+ const { profile } = getFirstActiveProfile(['responses', 'openai-chat', 'anthropic', 'vertex'], req);
1228
+ if (Array.isArray(profile?.publicModels) && profile.publicModels.length) {
1229
+ return { ids: [...new Set(profile.publicModels.filter(Boolean))], windows: publicModelWindows(profile) };
1230
+ }
1231
+ const windows = smallestWindows(Object.entries(profile?.defaultModels || {}).map(([slot, id]) => [id, model1MForSlot(profile, slot)]));
1232
+ return { ids: [...windows.keys()], windows };
1233
+ }
1234
+
1235
+ async function routeApi(req, res, method, pathname) {
1236
+ // GET /api/status
1237
+ if (method === 'GET' && pathname === '/api/status') {
1238
+ const cfg = requireConfig(res);
1239
+ if (!cfg) return;
1240
+ const activeProfiles = getActiveMap(cfg);
1241
+ return sendJson(res, 200, {
1242
+ port: PORT,
1243
+ activeProfile: cfg.activeProfile || null,
1244
+ activeProfiles,
1245
+ revision: configRevision(cfg),
1246
+ claude1MTiers: computeLaunchState(cfg, PORT).claude1MTiers,
1247
+ ...readLaunchFlags(),
1248
+ claudeBaseURL: activeProfiles.anthropic ? `http://127.0.0.1:${PORT} (injected via launcher)` : '(none / official)',
1249
+ config: redactConfig(cfg)
1250
+ });
1251
+ }
1252
+
1253
+ // GET /api/logs (Live Request/Response Inspector)
1254
+ if (method === 'GET' && pathname === '/api/logs') {
1255
+ return sendJson(res, 200, { logs: requestLogs.slice().reverse() });
1256
+ }
1257
+
1258
+ if (method !== 'POST') {
1259
+ req.resume();
1260
+ return sendJson(res, 404, { error: `Not found: ${method} ${pathname}` });
1261
+ }
1262
+
1263
+ let body;
1264
+ try {
1265
+ body = await readJsonBody(req);
1266
+ } catch (err) {
1267
+ return sendJson(res, err.status || 400, { error: err.message });
1268
+ }
1269
+
1270
+ // POST /api/logs/clear
1271
+ if (pathname === '/api/logs/clear') {
1272
+ requestLogs.length = 0;
1273
+ return sendJson(res, 200, { success: true });
1274
+ }
1275
+
1276
+ if (CONFIG_CHANGES.has(pathname)) return serialized(() => routeConfigApi(res, method, pathname, body));
1277
+ return routeConfigApi(res, method, pathname, body);
1278
+ }
1279
+
1280
+ async function routeConfigApi(res, method, pathname, body) {
1281
+ const loaded = requireConfig(res);
1282
+ if (!loaded) return;
1283
+ // A copy: a refused change must never reach the cached config that other requests read.
1284
+ const cfg = structuredClone(loaded);
1285
+ const baseRevision = configRevision(loaded);
1286
+ if (typeof body.revision === 'string' && body.revision !== baseRevision) {
1287
+ return sendJson(res, 409, { error: 'config.json changed since this page loaded it. The page reloads it now; check the change and try again.', revision: baseRevision });
1288
+ }
1289
+
1290
+ // POST /api/switch { target?, profile? | null, deactivate? }
1291
+ if (pathname === '/api/switch') {
1292
+ let err = null;
1293
+ if (body.deactivate) {
1294
+ deactivateProfile(cfg, body.deactivate);
1295
+ } else if (body.target) {
1296
+ err = setTargetProfile(cfg, body.target, body.profile || null);
1297
+ } else if (body.profile) {
1298
+ err = activateProfile(cfg, body.profile);
1299
+ } else {
1300
+ deactivateAll(cfg);
1301
+ }
1302
+ if (err) return sendJson(res, 400, { error: err });
1303
+ const applied = await commit(cfg, baseRevision);
1304
+ return sendJson(res, applied.status || (applied.success === false ? 502 : 200), { success: true, activeProfile: cfg.activeProfile, activeProfiles: cfg.activeProfiles, ...applied });
1305
+ }
1306
+
1307
+ // POST /api/toggle { target?, enabled }
1308
+ if (pathname === '/api/toggle') {
1309
+ const map = getActiveMap(cfg);
1310
+ let err = null;
1311
+ if (body.target) {
1312
+ if (!TARGETS.includes(body.target)) return sendJson(res, 400, { error: `Unknown target "${body.target}"` });
1313
+ let key = null;
1314
+ if (body.enabled) {
1315
+ const candidates = [map[body.target], cfg.activeProfile, ...Object.keys(cfg.profiles)];
1316
+ key = candidates.find(k => hasProfile(cfg, k) && profileAcceptsTarget(cfg.profiles[k], body.target)) || null;
1317
+ if (!key) return sendJson(res, 400, { error: `No profile accepts target "${body.target}"` });
1318
+ }
1319
+ err = setTargetProfile(cfg, body.target, key);
1320
+ } else if (body.enabled) {
1321
+ const key = hasProfile(cfg, cfg.activeProfile) ? cfg.activeProfile : Object.keys(cfg.profiles)[0];
1322
+ if (!key) return sendJson(res, 400, { error: 'No profiles configured' });
1323
+ err = activateProfile(cfg, key);
1324
+ } else {
1325
+ deactivateAll(cfg);
1326
+ }
1327
+ if (err) return sendJson(res, 400, { error: err });
1328
+ const applied = await commit(cfg, baseRevision);
1329
+ return sendJson(res, applied.status || (applied.success === false ? 502 : 200), { success: true, enabled: Boolean(body.enabled), activeProfiles: cfg.activeProfiles, ...applied });
1330
+ }
1331
+
1332
+ // POST /api/save-profile { key, profile }
1333
+ if (pathname === '/api/save-profile') {
1334
+ const { key, profile } = body;
1335
+ if (!isValidProfileKey(key)) {
1336
+ return sendJson(res, 400, { error: 'Invalid profile key: use 1-64 chars of letters, digits, ".", "_" or "-"' });
1337
+ }
1338
+ const invalid = validateProfileInput(profile);
1339
+ if (invalid) return sendJson(res, 400, { error: invalid });
1340
+
1341
+ const existing = hasProfile(cfg, key) ? cfg.profiles[key] : {};
1342
+ // Merge so unmanaged UI fields are not lost (e.g. `endpoints`).
1343
+ const merged = { ...existing, ...profile };
1344
+ // A payload without apiKey, or with the mask, keeps the stored key, but only for the destination
1345
+ // it was stored with: otherwise one request could send the real key to any host.
1346
+ const keepsKey = !Object.hasOwn(profile, 'apiKey') || profile.apiKey === MASKED_KEY;
1347
+ if (keepsKey && existing.apiKey && !sameDestination(merged, existing)) {
1348
+ return sendJson(res, 400, { error: 'The stored API key is sent only to the baseURL and endpoints it was saved with. Enter the key again to use a new URL.' });
1349
+ }
1350
+ merged.apiKey = keepsKey ? (existing.apiKey || '') : String(profile.apiKey || '');
1351
+ for (const k of ['outFormat', 'optimizerURL', 'thinkingMode']) {
1352
+ if (Object.hasOwn(profile, k) && !profile[k]) delete merged[k];
1353
+ }
1354
+ cfg.profiles[key] = merged;
1355
+
1356
+ // A target assigned to this profile whose new inFormat no longer supports it -> unassign that target.
1357
+ const map = getActiveMap(cfg);
1358
+ cfg.activeProfiles = map;
1359
+ let unassigned = false;
1360
+ for (const t of TARGETS) {
1361
+ if (map[t] === key && !profileAcceptsTarget(merged, t)) {
1362
+ map[t] = null;
1363
+ unassigned = true;
1364
+ }
1365
+ }
1366
+
1367
+ // Profile is active (or was just unassigned from a target) -> refresh 1M flags / env files.
1368
+ const applied = isProfileActive(cfg, key) || unassigned ? await commit(cfg, baseRevision) : (saveConfig(cfg), { revision: configRevision(cfg) });
1369
+ return sendJson(res, applied.status || (applied.success === false ? 502 : 200), { success: true, ...applied });
1370
+ }
1371
+
1372
+ // POST /api/delete-profile { key }
1373
+ if (pathname === '/api/delete-profile') {
1374
+ const err = deleteProfile(cfg, body.key);
1375
+ if (err) return sendJson(res, 404, { error: err });
1376
+ const applied = await commit(cfg, baseRevision);
1377
+ return sendJson(res, applied.status || (applied.success === false ? 502 : 200), { success: true, ...applied });
1378
+ }
1379
+
1380
+ // POST /api/blindfold/sync — the CLI asks the owner to bring the interceptor in line with config.json.
1381
+ if (pathname === '/api/blindfold/sync') {
1382
+ const r = await reconcile(cfg);
1383
+ return sendJson(res, r.ok ? 200 : 500, r);
1384
+ }
1385
+
1386
+ // POST /api/test-upstream
1387
+ if (pathname === '/api/test-upstream') {
1388
+ try {
1389
+ const r = await testUpstream(body, cfg);
1390
+ return sendJson(res, r.status, r.json);
1391
+ } catch (err) {
1392
+ return sendJson(res, 200, { ok: false, error: err.name === 'TimeoutError' ? 'Timed out waiting for upstream' : (err.cause?.message || err.message) });
1393
+ }
1394
+ }
1395
+
1396
+ // POST /api/fetch-models
1397
+ if (pathname === '/api/fetch-models') {
1398
+ try {
1399
+ const r = await fetchModels(body, cfg);
1400
+ return sendJson(res, r.status, r.json);
1401
+ } catch (err) {
1402
+ return sendJson(res, 200, { ok: false, error: err.name === 'TimeoutError' ? 'Timed out waiting for upstream' : (err.cause?.message || err.message) });
1403
+ }
1404
+ }
1405
+
1406
+ return sendJson(res, 404, { error: `Not found: ${method} ${pathname}` });
1407
+ }
1408
+
1409
+ const server = http.createServer((req, res) => {
1410
+ route(req, res).catch(err => {
1411
+ console.error('[llm-switcher] Unhandled request error:', err);
1412
+ if (!res.headersSent) {
1413
+ sendJson(res, err.status || 500, { error: { message: `Gateway error: ${err.message}` } });
1414
+ } else {
1415
+ try { res.end(); } catch {}
1416
+ }
1417
+ });
1418
+ });
1419
+
1420
+ function encodeWsFrame(data, opcode = 1) {
1421
+ const payload = Buffer.isBuffer(data) ? data : Buffer.from(data, 'utf8');
1422
+ const len = payload.length;
1423
+ let header;
1424
+ if (len <= 125) {
1425
+ header = Buffer.from([0x80 | (opcode & 0x0f), len]);
1426
+ } else if (len <= 65535) {
1427
+ header = Buffer.alloc(4);
1428
+ header[0] = 0x80 | (opcode & 0x0f);
1429
+ header[1] = 126;
1430
+ header.writeUInt16BE(len, 2);
1431
+ } else {
1432
+ header = Buffer.alloc(10);
1433
+ header[0] = 0x80 | (opcode & 0x0f);
1434
+ header[1] = 127;
1435
+ header.writeBigUInt64BE(BigInt(len), 2);
1436
+ }
1437
+ return Buffer.concat([header, payload]);
1438
+ }
1439
+
1440
+ // Map an upstream HTTP status to a Responses-API error code so Codex can tell a
1441
+ // retryable rate-limit from a fatal request error.
1442
+ function responsesErrorCode(status) {
1443
+ if (status === 429) return 'rate_limit_exceeded';
1444
+ if (status === 401) return 'authentication_error';
1445
+ if (status === 403) return 'permission_denied';
1446
+ if (status === 404) return 'not_found_error';
1447
+ if (status === 400) return 'invalid_request_error';
1448
+ return 'server_error';
1449
+ }
1450
+
1451
+ // A WS frame carries the event payload alone. The half records the same event in the SSE text of
1452
+ // the HTTP /v1/responses path, so one converter keeps one shape in intact whatever the transport.
1453
+ // A failure to copy is dropped: the tap must never come between the renderer and the socket.
1454
+ function tapWsEvent(tap, event, data) {
1455
+ if (!tap) return;
1456
+ try {
1457
+ tap.push(`event: ${event}\ndata: ${JSON.stringify(data)}\n\n`);
1458
+ } catch {}
1459
+ }
1460
+
1461
+ // Terminal failure for the WS (responses-ws) transport: Codex ends a turn only on
1462
+ // response.completed / response.failed, so a bare {type:'error'} frame leaves the turn
1463
+ // hanging. Emit the full created -> in_progress -> failed sequence instead.
1464
+ function sendWsFailed(socket, model, message, status = 500, tap = null) {
1465
+ if (!socket.writable) return;
1466
+ const renderer = createResponsesStream((e, d) => {
1467
+ tapWsEvent(tap, e, d);
1468
+ if (socket.writable) socket.write(encodeWsFrame(JSON.stringify(d)));
1469
+ }, model || 'main');
1470
+ renderer.start();
1471
+ renderer.error(message, responsesErrorCode(status));
1472
+ }
1473
+
1474
+ // 9Router forwards OpenAI-format tools to Gemini/Vertex for ag/* models, which accept only
1475
+ // a strict Schema subset: bare "object" strings, $ref/$defs, anyOf-null unions and
1476
+ // additionalProperties all come back as HTTP 400 INVALID_ARGUMENT. Rewrite tool parameters
1477
+ // into that subset before sending upstream. Non-ag targets keep the OpenAI superset.
1478
+ function geminiSafeTools(tools) {
1479
+ return tools.map(t => {
1480
+ if (!t || t.type !== 'function' || !t.function) return t;
1481
+ return { ...t, function: { ...t.function, parameters: toGeminiSchema(t.function.parameters || { type: 'object', properties: {} }) } };
1482
+ });
1483
+ }
1484
+
1485
+ // One turn of the Codex WS transport. The socket loop runs turns one at a time and owns `ac`.
1486
+ async function handleWsResponseCreate(socket, payload, req, ac) {
1487
+ const clientFormat = 'responses';
1488
+ const { profileKey, profile, error: profileError } = getActiveProfile(clientFormat, req);
1489
+ if (!loadConfig()) {
1490
+ sendWsFailed(socket, payload?.model || 'main', `LLM Switcher config not loaded (${configPath}): ${getConfigLoadError()?.message || 'missing file'}`, 500);
1491
+ return;
1492
+ }
1493
+ if (!profile) {
1494
+ sendWsFailed(socket, payload?.model || 'main', profileError || 'Proxy is currently OFF for responses.', 503);
1495
+ return;
1496
+ }
1497
+
1498
+ let ir;
1499
+ try {
1500
+ ir = parseToIR('responses', payload);
1501
+ } catch (e) {
1502
+ sendWsFailed(socket, payload?.model || 'main', `Cannot parse responses request: ${e.message}`, 400);
1503
+ return;
1504
+ }
1505
+ ir.stream = true;
1506
+
1507
+ const reqStartTime = Date.now();
1508
+ const requestPreview = previewOf(ir);
1509
+ const requestedModel = ir.model || payload.model || '';
1510
+ const mappedModel = mapModel(requestedModel, profile, clientFormat);
1511
+ const outFormat = resolveOutFormat(profile, mappedModel);
1512
+
1513
+ console.log(`[llm-switcher:ws] ${clientFormat} -> ${outFormat} "${requestedModel}" -> "${mappedModel}" [${profile.name || profileKey}]`);
1514
+ const logBase = { clientFormat: 'responses-ws', outFormat, profile: profileKey, model: mappedModel, stream: true, requestPreview };
1515
+ const log = (extra) => logInspection({ ...logBase, duration: Date.now() - reqStartTime, tokens: { prompt: 0, completion: 0 }, ...extra });
1516
+
1517
+ // Contract lab: the Codex WS transport is sampled like any other request, and the events of
1518
+ // this turn are copied for the upload that follows it.
1519
+ const traceId = contractLab.traceFor(mappedModel);
1520
+ const halfTap = traceId ? createHalfTap() : null;
1521
+ const halfRequest = traceId ? capJson(payload) : '';
1522
+
1523
+ let answered = false; // set only after a complete answer; the half upload depends on it
1524
+ try {
1525
+ const { url, headers, upBody } = buildUpstreamRequest(profile, outFormat, ir, mappedModel, req);
1526
+ if (traceId) headers['x-intact-trace'] = traceId;
1527
+ debugLog(`[${profileKey}:ws] ${clientFormat} -> ${outFormat} ${url} ::`, JSON.stringify(upBody).slice(0, 300));
1528
+
1529
+ let upstreamRes;
1530
+ try {
1531
+ upstreamRes = await fetch(url, { method: 'POST', headers, body: JSON.stringify(upBody), signal: ac.signal });
1532
+ } catch (fetchErr) {
1533
+ if (ac.signal.aborted) return log({ status: 499, error: 'client disconnected' });
1534
+ console.error(`[${profileKey}:ws] Network error:`, fetchErr.message);
1535
+ sendWsFailed(socket, mappedModel, `Failed to connect to upstream: ${fetchErr.cause?.message || fetchErr.message}`, 502, halfTap);
1536
+ return log({ status: 502, error: fetchErr.message });
1537
+ }
1538
+
1539
+ if (!upstreamRes.ok) {
1540
+ const errText = await upstreamRes.text().catch(() => '');
1541
+ console.error(`[${profileKey}:ws] Error HTTP ${upstreamRes.status}:`, errText.slice(0, 500));
1542
+ sendWsFailed(socket, mappedModel, extractUpstreamMessage(errText) || `Upstream HTTP ${upstreamRes.status}`, upstreamRes.status, halfTap);
1543
+ return log({ status: upstreamRes.status, error: errText.slice(0, 300) });
1544
+ }
1545
+
1546
+ const normalize = createUpstreamNormalizer(outFormat);
1547
+ const col = createCollector();
1548
+
1549
+ const renderer = createResponsesStream((e, d) => {
1550
+ tapWsEvent(halfTap, e, d);
1551
+ if (socket.writable) {
1552
+ socket.write(encodeWsFrame(JSON.stringify(d)));
1553
+ }
1554
+ }, requestedModel || mappedModel, { toolMeta: ir.toolMeta });
1555
+
1556
+ renderer.start();
1557
+ const splitter = createThinkTagSplitter(t => renderer.think(t), t => renderer.text(t));
1558
+ const streamError = await pumpStream(upstreamRes, { normalize, col, renderer, splitter, signal: ac.signal, sink: socket, tag: `${profileKey}:ws` });
1559
+
1560
+ if (ac.signal.aborted) {
1561
+ return log({ status: 499, error: 'client disconnected mid-stream', responsePreview: col.text.join('').slice(0, 300) });
1562
+ }
1563
+ splitter.flush();
1564
+ const completion = col.completion();
1565
+ if (streamError) {
1566
+ renderer.error(streamError);
1567
+ } else {
1568
+ renderer.finish(col.finish, { completion, prompt: col.prompt, cached: col.cached, reasoning: col.reasoning, hasTools: col.tools.size > 0 });
1569
+ answered = true;
1570
+ }
1571
+ log({
1572
+ status: streamError ? 502 : 200, stream: true,
1573
+ ...(streamError ? { error: String(streamError).slice(0, 300) } : {}),
1574
+ tokens: { prompt: col.prompt, completion },
1575
+ thinkingChars: col.think.join('').length,
1576
+ responsePreview: col.text.join('').slice(0, 300)
1577
+ });
1578
+ } catch (err) {
1579
+ if (ac.signal.aborted) return log({ status: 499, error: 'aborted' });
1580
+ console.error(`[${profileKey}:ws] Error:`, err);
1581
+ sendWsFailed(socket, payload?.model || 'main', err.message, 500, halfTap);
1582
+ log({ status: 500, error: err.message });
1583
+ } finally {
1584
+ // The turn is over; the half is queued and posted on a later turn. A cancelled turn has no
1585
+ // complete answer: uploading it would diff as a loss the converter never made.
1586
+ finishHalf(contractLab, traceId, ac.signal.aborted || !answered, {
1587
+ toolRequest: halfRequest,
1588
+ toolResponse: halfTap?.text() || '',
1589
+ toolVersion: toolVersionFromUA(req.headers['user-agent']),
1590
+ inFormat: clientFormat,
1591
+ outFormat
1592
+ });
1593
+ }
1594
+ }
1595
+
1596
+ server.on('upgrade', (req, socket) => {
1597
+ // Node emits 'upgrade' instead of 'request', so route() never runs here and the
1598
+ // Host/Origin guard has to be applied again. Browsers do not apply same-origin to
1599
+ // WebSocket, so without this any visited page could open ws://127.0.0.1/v1/responses
1600
+ // and spend the profile's API key. An absent Origin stays allowed on purpose: Codex
1601
+ // sends none, and the blindfold interceptor deletes it.
1602
+ const guardError = checkRequestOrigin(req);
1603
+ if (guardError) {
1604
+ socket.write(
1605
+ 'HTTP/1.1 403 Forbidden\r\n' +
1606
+ 'Connection: close\r\n' +
1607
+ 'Content-Type: application/json\r\n\r\n' +
1608
+ `{"error":{"message":${JSON.stringify(guardError)}}}\r\n`
1609
+ );
1610
+ socket.destroy();
1611
+ return;
1612
+ }
1613
+
1614
+ const p = new URL(req.url || '/', 'http://127.0.0.1').pathname;
1615
+ if (p !== '/v1/responses' && p !== '/responses') {
1616
+ socket.write(
1617
+ 'HTTP/1.1 404 Not Found\r\n' +
1618
+ 'Connection: close\r\n' +
1619
+ 'Content-Type: application/json\r\n\r\n' +
1620
+ `{"error":{"message":"Not found: ${req.method} ${p}"}}\r\n`
1621
+ );
1622
+ socket.destroy();
1623
+ return;
1624
+ }
1625
+
1626
+ const key = req.headers['sec-websocket-key'];
1627
+ if (!key) {
1628
+ socket.destroy();
1629
+ return;
1630
+ }
1631
+ const accept = crypto.createHash('sha1').update(key + '258EAFA5-E914-47DA-95CA-C5AB0DC85B11').digest('base64');
1632
+
1633
+ // The official name or the slot, never the upstream ID. The value goes into a raw header line.
1634
+ const { profile } = getActiveProfile('responses', req);
1635
+ const publicMain = profile ? codexPublicModel(profile, 'main') : '';
1636
+ const activeModel = isSafeModelName(publicMain) ? publicMain : 'main';
1637
+
1638
+ socket.write(
1639
+ 'HTTP/1.1 101 Switching Protocols\r\n' +
1640
+ 'Upgrade: websocket\r\n' +
1641
+ 'Connection: Upgrade\r\n' +
1642
+ `Sec-WebSocket-Accept: ${accept}\r\n` +
1643
+ `OpenAI-Model: ${activeModel}\r\n` +
1644
+ 'x-reasoning-included: true\r\n' +
1645
+ 'x-codex-turn-state: ready\r\n\r\n'
1646
+ );
1647
+
1648
+ const read = createFrameReader({ maxMessage: MAX_BODY_SIZE });
1649
+ // Turns run one at a time: two at once interleave their events on one socket. Every turn,
1650
+ // queued or running, holds a controller here, so a cancel or a close stops all of them.
1651
+ const turns = new Set();
1652
+ let turnChain = Promise.resolve();
1653
+ const abortTurns = () => { for (const ac of turns) ac.abort(); };
1654
+
1655
+ const handleMessage = (msg) => {
1656
+ if (msg.type === 'response.create') {
1657
+ const ac = new AbortController();
1658
+ turns.add(ac);
1659
+ turnChain = turnChain
1660
+ .then(() => (ac.signal.aborted ? null : handleWsResponseCreate(socket, msg, req, ac)))
1661
+ .catch(err => {
1662
+ // A throw before the handler's own try: Codex ends a turn only on response.failed.
1663
+ console.error('[llm-switcher:ws] Unhandled turn error:', err);
1664
+ if (!ac.signal.aborted) sendWsFailed(socket, msg.model || 'main', err.message, 500);
1665
+ })
1666
+ .finally(() => turns.delete(ac));
1667
+ } else if (msg.type === 'response.cancel') {
1668
+ abortTurns();
1669
+ } else if (msg.type === 'session.update') {
1670
+ if (socket.writable) socket.write(encodeWsFrame(JSON.stringify({ type: 'session.updated', session: msg.session || {} })));
1671
+ } else if (msg.type === 'conversation.item.create') {
1672
+ if (socket.writable) socket.write(encodeWsFrame(JSON.stringify({ type: 'conversation.item.created', item: msg.item || {} })));
1673
+ }
1674
+ };
1675
+
1676
+ socket.on('data', (chunk) => {
1677
+ for (const f of read(chunk)) {
1678
+ if (f.type === 'error') {
1679
+ // 1009 = message too big. The reader has stopped, so the connection cannot continue.
1680
+ const code = Buffer.alloc(2);
1681
+ code.writeUInt16BE(/exceeds/.test(f.reason) ? 1009 : 1002);
1682
+ if (socket.writable) socket.end(encodeWsFrame(code, 8));
1683
+ abortTurns();
1684
+ return;
1685
+ }
1686
+ if (f.type === 'close') {
1687
+ abortTurns();
1688
+ if (socket.writable) socket.end(encodeWsFrame(Buffer.alloc(0), 8));
1689
+ return;
1690
+ }
1691
+ if (f.type === 'ping') {
1692
+ if (socket.writable) socket.write(encodeWsFrame(f.payload, 10));
1693
+ continue;
1694
+ }
1695
+ if (f.type !== 'text') continue;
1696
+ let msg;
1697
+ try {
1698
+ msg = JSON.parse(f.payload.toString('utf8'));
1699
+ } catch (e) {
1700
+ console.error('[llm-switcher:ws] Bad WS message JSON:', e.message);
1701
+ continue;
1702
+ }
1703
+ handleMessage(msg);
1704
+ }
1705
+ });
1706
+
1707
+ socket.on('close', abortTurns);
1708
+ // The upgraded socket allows half-open: a client FIN alone would not close it, and the turn
1709
+ // would keep spending upstream tokens until the next write failed.
1710
+ socket.on('end', () => {
1711
+ abortTurns();
1712
+ socket.end();
1713
+ });
1714
+
1715
+ socket.on('error', (err) => {
1716
+ debugLog('[llm-switcher:ws] Socket error:', err.message);
1717
+ abortTurns();
1718
+ });
1719
+ });
1720
+
1721
+ server.on('error', (err) => {
1722
+ if (err.code === 'EADDRINUSE') {
1723
+ console.error(`\n[llm-switcher:ERROR] Port ${PORT} is already in use by another process!`);
1724
+ console.error(`- Run 'switch status' to check, or 'switch off' to stop a running LLM Switcher.`);
1725
+ console.error(`- If another tool (e.g. headroom/rtk/proxy) is using port ${PORT}, change "port" in config.json or pass --port.`);
1726
+ process.exit(1);
1727
+ } else {
1728
+ console.error('[llm-switcher:ERROR]', err);
1729
+ }
1730
+ });
1731
+
1732
+ if (!loadConfig()) {
1733
+ console.warn(`[llm-switcher:WARN] Could not load ${configPath}: ${getConfigLoadError()?.message}. Copy config.example.json to config.json.`);
1734
+ }
1735
+
1736
+ server.listen(PORT, '127.0.0.1', () => {
1737
+ // A service start or a restart on a new port finds env-codex.* already pointing at the interceptor.
1738
+ const cfg = loadConfig();
1739
+ if (cfg && !getConfigLoadError()) serialized(() => reconcile(cfg));
1740
+ console.log(`[llm-switcher] Server running on http://127.0.0.1:${PORT}`);
1741
+ console.log(`[llm-switcher] Web UI available at: http://127.0.0.1:${PORT}/ui`);
1742
+ console.log(`[llm-switcher] Endpoints: /v1/messages (anthropic) | /v1/chat/completions (openai) | /v1/responses (codex) | /v1beta/models/* (vertex)`);
1743
+ });