llm-switcher 1.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.gitattributes +16 -0
- package/LICENSE +21 -0
- package/README.md +587 -0
- package/README.vi.md +585 -0
- package/blindfold/blindfold.mjs +633 -0
- package/blindfold/make-certs.sh +88 -0
- package/blindfold/wsframe.mjs +176 -0
- package/codex-catalog-template.json +1 -0
- package/config.example.json +84 -0
- package/contract-exclusions.json +41 -0
- package/contract.mjs +561 -0
- package/docs/LLM-RESPONSE-MATRIX.md +165 -0
- package/docs/TOKEN-OPTIMIZER-INTEROP.md +110 -0
- package/docs/codex-blindfold.md +214 -0
- package/docs/cross-platform.md +136 -0
- package/docs/diagrams/blindfold-request-routing.html +14972 -0
- package/docs/diagrams/blindfold-request-routing.sequence.json +175 -0
- package/docs/diagrams/blindfold-switch-lifecycle.html +14958 -0
- package/docs/diagrams/blindfold-switch-lifecycle.lifecycle.json +159 -0
- package/docs/diagrams/codex-model-name-resolution.html +15005 -0
- package/docs/diagrams/codex-model-name-resolution.workflow.json +71 -0
- package/docs/response-matrix.json +1131 -0
- package/formats.mjs +2308 -0
- package/mcp.mjs +340 -0
- package/package.json +36 -0
- package/proxy.mjs +1743 -0
- package/service.mjs +132 -0
- package/shim.mjs +292 -0
- package/skills/llm-switcher/SKILL.md +88 -0
- package/state.mjs +978 -0
- package/switch +5 -0
- package/switch.cmd +2 -0
- package/switch.mjs +930 -0
- package/tests/blindfold.test.mjs +307 -0
- package/tests/blindfold.wire.test.mjs +170 -0
- package/tests/contract/run.test.mjs +214 -0
- package/tests/contract-check.test.mjs +458 -0
- package/tests/contract-lab.test.mjs +755 -0
- package/tests/datadir.test.mjs +37 -0
- package/tests/formats.test.mjs +794 -0
- package/tests/gateway.e2e.test.mjs +999 -0
- package/tests/helpers.mjs +24 -0
- package/tests/lifecycle.test.mjs +416 -0
- package/tests/live-optimizer-interop.mjs +205 -0
- package/tests/mcp.test.mjs +91 -0
- package/tests/service.test.mjs +69 -0
- package/tests/shim.test.mjs +228 -0
- package/tests/state.test.mjs +675 -0
- package/tests/switch.test.mjs +156 -0
- package/tests/wsframe.test.mjs +154 -0
- package/ui.html +2234 -0
package/proxy.mjs
ADDED
|
@@ -0,0 +1,1743 @@
|
|
|
1
|
+
import http from 'node:http';
|
|
2
|
+
import fs from 'node:fs';
|
|
3
|
+
import path from 'node:path';
|
|
4
|
+
import zlib from 'node:zlib';
|
|
5
|
+
import crypto from 'node:crypto';
|
|
6
|
+
import { promisify } from 'node:util';
|
|
7
|
+
import { fileURLToPath } from 'node:url';
|
|
8
|
+
import { createFrameReader } from './blindfold/wsframe.mjs';
|
|
9
|
+
import {
|
|
10
|
+
OUT_FORMATS, IN_FORMATS, parseToIR, emitUpstreamBody, createUpstreamNormalizer, createCollector,
|
|
11
|
+
createThinkTagSplitter, splitThinkTags, healAnthropicPayload, estimateTokens, THINKING_MODES,
|
|
12
|
+
toGeminiSchema, isAntigravityModel,
|
|
13
|
+
createAnthropicStream, createChatStream, createResponsesStream, createVertexStream,
|
|
14
|
+
buildAnthropicMessage, buildChatMessage, buildResponsesMessage, buildVertexMessage
|
|
15
|
+
} from './formats.mjs';
|
|
16
|
+
import {
|
|
17
|
+
TARGETS, configPath, loadConfig, getConfigLoadError, saveConfig, resolvePort, hasProfile, isValidProfileKey,
|
|
18
|
+
getActiveMap, setTargetProfile, activateProfile, deactivateProfile, deactivateAll, deleteProfile,
|
|
19
|
+
isProfileActive, profileAcceptsTarget, applyLaunchState, readLaunchFlags, redactConfig, MASKED_KEY,
|
|
20
|
+
modelForSlot, primaryModel, codexPublicModel, isSafeModelName, parsePort, CODEX_MODEL_SLOTS,
|
|
21
|
+
ensureAdminToken, identityProof, reconcileBlindfold, checkBlindfoldTarget,
|
|
22
|
+
codexModelEntry, smallestWindows, publicModelWindows, model1MForSlot, computeLaunchState,
|
|
23
|
+
contractLabSettings
|
|
24
|
+
} from './state.mjs';
|
|
25
|
+
import { createContractLab, createHalfTap, tapClientWrites, capText, capJson, toolVersionFromUA, finishHalf, PROBE_HEADER, TRACE_ID_RE } from './contract.mjs';
|
|
26
|
+
|
|
27
|
+
const __dirname = path.dirname(fileURLToPath(import.meta.url));
|
|
28
|
+
const uiHtmlPath = path.join(__dirname, 'ui.html');
|
|
29
|
+
|
|
30
|
+
const MAX_BODY_SIZE = 50 * 1024 * 1024; // 50MB
|
|
31
|
+
const MAX_API_BODY_SIZE = 1024 * 1024; // 1MB for /api/*
|
|
32
|
+
|
|
33
|
+
// Avoid network conflicts: keep localhost / 127.0.0.1 out of external proxies (RTK, Headroom, VPN)
|
|
34
|
+
const currentNoProxy = process.env.NO_PROXY || process.env.no_proxy || '';
|
|
35
|
+
const localHosts = ['127.0.0.1', 'localhost'];
|
|
36
|
+
const existingNoProxy = currentNoProxy.split(',').map(s => s.trim().toLowerCase());
|
|
37
|
+
const missingNoProxy = localHosts.filter(h => !existingNoProxy.includes(h));
|
|
38
|
+
if (missingNoProxy.length > 0) {
|
|
39
|
+
process.env.NO_PROXY = currentNoProxy ? `${currentNoProxy},${missingNoProxy.join(',')}` : missingNoProxy.join(',');
|
|
40
|
+
process.env.no_proxy = process.env.NO_PROXY;
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
// Fixed port for the process lifetime: changing "port" in config.json at runtime has no effect
|
|
44
|
+
// env.cmd / flags would point at a port the server is not listening on.
|
|
45
|
+
const PORT = resolvePort();
|
|
46
|
+
|
|
47
|
+
// In-memory Request / Response Inspector Ring Buffer (up to 40 most recent requests)
|
|
48
|
+
const requestLogs = [];
|
|
49
|
+
const MAX_LOGS = 40;
|
|
50
|
+
function logInspection(entry) {
|
|
51
|
+
requestLogs.push({
|
|
52
|
+
id: `req_${Date.now()}_${Math.random().toString(36).slice(2, 6)}`,
|
|
53
|
+
timestamp: new Date().toLocaleTimeString(),
|
|
54
|
+
...entry
|
|
55
|
+
});
|
|
56
|
+
if (requestLogs.length > MAX_LOGS) {
|
|
57
|
+
requestLogs.shift();
|
|
58
|
+
}
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
function debugLog(...args) {
|
|
62
|
+
if (loadConfig()?.debug) {
|
|
63
|
+
console.log('[DEBUG]', ...args);
|
|
64
|
+
}
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
function sendJson(res, status, obj, headers = {}) {
|
|
68
|
+
if (res.headersSent) {
|
|
69
|
+
try { res.end(); } catch {}
|
|
70
|
+
return;
|
|
71
|
+
}
|
|
72
|
+
res.writeHead(status, { 'Content-Type': 'application/json', ...headers });
|
|
73
|
+
res.end(JSON.stringify(obj));
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
// ----------------------------------------------------
|
|
77
|
+
// Request guards & body reading
|
|
78
|
+
// ----------------------------------------------------
|
|
79
|
+
const LOOPBACK_HOSTS = new Set(['127.0.0.1', 'localhost', '[::1]', '::1']);
|
|
80
|
+
|
|
81
|
+
function hostnameOf(hostHeader) {
|
|
82
|
+
const h = String(hostHeader || '').trim().toLowerCase();
|
|
83
|
+
if (h.startsWith('[')) return h.slice(0, h.indexOf(']') + 1);
|
|
84
|
+
return h.replace(/:\d+$/, '');
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
// Block DNS rebinding (unknown Host pointing at 127.0.0.1) and CSRF from other sites (unknown Origin).
|
|
88
|
+
// Without the Host check, a malicious page could burn tokens via /v1/*.
|
|
89
|
+
function checkRequestOrigin(req) {
|
|
90
|
+
if (req.headers.host && !LOOPBACK_HOSTS.has(hostnameOf(req.headers.host))) {
|
|
91
|
+
return 'Forbidden: untrusted Host header';
|
|
92
|
+
}
|
|
93
|
+
const origin = req.headers.origin;
|
|
94
|
+
if (origin !== undefined) {
|
|
95
|
+
try {
|
|
96
|
+
const u = new URL(origin);
|
|
97
|
+
const port = u.port || (u.protocol === 'https:' ? '443' : '80');
|
|
98
|
+
if (!LOOPBACK_HOSTS.has(u.hostname.toLowerCase()) || port !== String(PORT)) {
|
|
99
|
+
return 'Forbidden: untrusted origin';
|
|
100
|
+
}
|
|
101
|
+
} catch {
|
|
102
|
+
return 'Forbidden: invalid origin';
|
|
103
|
+
}
|
|
104
|
+
}
|
|
105
|
+
return null;
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
// The Host/Origin guard stops browser pages only. A local process that is not the owner must
|
|
109
|
+
// also present the token from admin.token (mode 0600) to use /api/*.
|
|
110
|
+
const ADMIN_TOKEN = Buffer.from(ensureAdminToken());
|
|
111
|
+
|
|
112
|
+
// `switch contract-probe` names the trace id of the exchange it drives, so it can print the id
|
|
113
|
+
// before intact holds it. Only a process that can read admin.token is believed, and the marker
|
|
114
|
+
// header is on BLOCKED_PASSTHROUGH: it never leaves this gateway.
|
|
115
|
+
function probeTraceId(req) {
|
|
116
|
+
const given = String(req.headers[PROBE_HEADER] || '');
|
|
117
|
+
if (!given || !TRACE_ID_RE.test(given)) return null;
|
|
118
|
+
return isAdminRequest(req) ? given : null;
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
function isAdminRequest(req) {
|
|
122
|
+
const given = Buffer.from(String(req.headers['x-llm-switcher-token'] || ''));
|
|
123
|
+
return given.length === ADMIN_TOKEN.length && crypto.timingSafeEqual(given, ADMIN_TOKEN);
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
function readBody(req, limit) {
|
|
127
|
+
return new Promise((resolve, reject) => {
|
|
128
|
+
let size = 0;
|
|
129
|
+
let tooLarge = false;
|
|
130
|
+
const chunks = [];
|
|
131
|
+
req.on('data', chunk => {
|
|
132
|
+
if (tooLarge) return;
|
|
133
|
+
size += chunk.length;
|
|
134
|
+
if (size > limit) {
|
|
135
|
+
tooLarge = true;
|
|
136
|
+
const err = new Error(`Payload Too Large (max ${Math.round(limit / 1024 / 1024)}MB)`);
|
|
137
|
+
err.status = 413;
|
|
138
|
+
reject(err);
|
|
139
|
+
req.resume();
|
|
140
|
+
return;
|
|
141
|
+
}
|
|
142
|
+
chunks.push(chunk);
|
|
143
|
+
});
|
|
144
|
+
req.on('end', () => {
|
|
145
|
+
if (tooLarge) return;
|
|
146
|
+
decodeBody(Buffer.concat(chunks), req.headers['content-encoding'], limit).then(resolve, reject);
|
|
147
|
+
});
|
|
148
|
+
req.on('error', err => {
|
|
149
|
+
err.status = 400;
|
|
150
|
+
reject(err);
|
|
151
|
+
});
|
|
152
|
+
});
|
|
153
|
+
}
|
|
154
|
+
|
|
155
|
+
const INFLATERS = {
|
|
156
|
+
gzip: promisify(zlib.gunzip),
|
|
157
|
+
deflate: promisify(zlib.inflate),
|
|
158
|
+
br: promisify(zlib.brotliDecompress),
|
|
159
|
+
...(typeof zlib.zstdDecompress === 'function' ? { zstd: promisify(zlib.zstdDecompress) } : {})
|
|
160
|
+
};
|
|
161
|
+
|
|
162
|
+
// Decompression runs off the event loop and stops at the raw-body limit: a few KB of gzip can
|
|
163
|
+
// expand to gigabytes.
|
|
164
|
+
async function decodeBody(raw, contentEncoding, limit) {
|
|
165
|
+
let encoding = String(contentEncoding || '').toLowerCase().trim();
|
|
166
|
+
if (!['gzip', 'deflate', 'br', 'zstd'].includes(encoding)) {
|
|
167
|
+
// Magic bytes: some clients compress without saying so.
|
|
168
|
+
if (raw.length >= 4 && raw[0] === 0x28 && raw[1] === 0xb5 && raw[2] === 0x2f && raw[3] === 0xfd && INFLATERS.zstd) encoding = 'zstd';
|
|
169
|
+
else if (raw.length >= 2 && raw[0] === 0x1f && raw[1] === 0x8b) encoding = 'gzip';
|
|
170
|
+
else return raw;
|
|
171
|
+
}
|
|
172
|
+
const inflate = INFLATERS[encoding];
|
|
173
|
+
if (!inflate) throw Object.assign(new Error('zstd decompression not supported in this Node.js version'), { status: 415 });
|
|
174
|
+
try {
|
|
175
|
+
return await inflate(raw, { maxOutputLength: limit });
|
|
176
|
+
} catch (err) {
|
|
177
|
+
if (err.code === 'ERR_BUFFER_TOO_LARGE') {
|
|
178
|
+
throw Object.assign(new Error(`Payload Too Large after decompression (max ${Math.round(limit / 1024 / 1024)}MB)`), { status: 413 });
|
|
179
|
+
}
|
|
180
|
+
err.status = 400;
|
|
181
|
+
throw err;
|
|
182
|
+
}
|
|
183
|
+
}
|
|
184
|
+
|
|
185
|
+
async function readJsonBody(req, limit = MAX_API_BODY_SIZE) {
|
|
186
|
+
const buf = await readBody(req, limit);
|
|
187
|
+
try {
|
|
188
|
+
const v = JSON.parse(buf.toString('utf8') || '{}');
|
|
189
|
+
if (!v || typeof v !== 'object' || Array.isArray(v)) throw new Error('JSON body must be an object');
|
|
190
|
+
return v;
|
|
191
|
+
} catch (e) {
|
|
192
|
+
const err = new Error(`Invalid JSON body: ${e.message}`);
|
|
193
|
+
err.status = 400;
|
|
194
|
+
throw err;
|
|
195
|
+
}
|
|
196
|
+
}
|
|
197
|
+
|
|
198
|
+
// ----------------------------------------------------
|
|
199
|
+
// Profile & model resolution
|
|
200
|
+
// ----------------------------------------------------
|
|
201
|
+
function getActiveProfile(clientFormat, req) {
|
|
202
|
+
const cfg = loadConfig();
|
|
203
|
+
if (!cfg) return { cfg: null, profileKey: null, profile: null };
|
|
204
|
+
|
|
205
|
+
let reqProfile = null;
|
|
206
|
+
if (req) {
|
|
207
|
+
reqProfile = req.headers['x-llm-profile'] || req.headers['x-profile'] || null;
|
|
208
|
+
if (!reqProfile && req.url) {
|
|
209
|
+
try { reqProfile = new URL(req.url, 'http://localhost').searchParams.get('profile'); } catch {}
|
|
210
|
+
}
|
|
211
|
+
}
|
|
212
|
+
if (reqProfile) {
|
|
213
|
+
if (hasProfile(cfg, reqProfile)) return { cfg, profileKey: reqProfile, profile: cfg.profiles[reqProfile] };
|
|
214
|
+
return { cfg, profileKey: reqProfile, profile: null, error: `Profile "${reqProfile}" requested via x-llm-profile/?profile= does not exist` };
|
|
215
|
+
}
|
|
216
|
+
|
|
217
|
+
// Look up the active profile by CLI target (clientFormat: anthropic | responses | openai-chat | vertex).
|
|
218
|
+
// Deleted profile / disabled target counts as OFF, never silently fall through to another profile (different API key!).
|
|
219
|
+
const key = clientFormat ? getActiveMap(cfg)[clientFormat] : (cfg.activeProfile || null);
|
|
220
|
+
if (!key || !hasProfile(cfg, key)) return { cfg, profileKey: key || null, profile: null };
|
|
221
|
+
return { cfg, profileKey: key, profile: cfg.profiles[key] };
|
|
222
|
+
}
|
|
223
|
+
|
|
224
|
+
// For endpoints not tied to a specific client format (/health, /v1/models).
|
|
225
|
+
function getFirstActiveProfile(preferred, req) {
|
|
226
|
+
if (req) {
|
|
227
|
+
const r = getActiveProfile(null, req);
|
|
228
|
+
if (r.profile) return r;
|
|
229
|
+
}
|
|
230
|
+
for (const t of preferred) {
|
|
231
|
+
const r = getActiveProfile(t);
|
|
232
|
+
if (r.profile) return r;
|
|
233
|
+
}
|
|
234
|
+
return { cfg: loadConfig(), profileKey: null, profile: null };
|
|
235
|
+
}
|
|
236
|
+
|
|
237
|
+
// clientFormat is the protocol the request arrived in. An `auto` profile serves every protocol,
|
|
238
|
+
// so the profile's inFormat cannot tell a Codex request from a Claude one.
|
|
239
|
+
function mapModel(requestedModel, profile, clientFormat) {
|
|
240
|
+
if (!requestedModel) return primaryModel(profile);
|
|
241
|
+
const clean = requestedModel.replace(/\[1m\]/gi, '').trim();
|
|
242
|
+
// If the client specified a model with a provider prefix (e.g. ag/..., gh/..., cf/...), keep it as-is
|
|
243
|
+
if (clean.includes('/') && !clean.startsWith('anthropic/')) {
|
|
244
|
+
return clean;
|
|
245
|
+
}
|
|
246
|
+
const m = clean.toLowerCase();
|
|
247
|
+
if (clientFormat === 'responses') {
|
|
248
|
+
// Aliases for the real Codex roles in the docs (model / review_model /
|
|
249
|
+
// agents.default_subagent_model). Unknown names pass through unchanged.
|
|
250
|
+
const aliases = {
|
|
251
|
+
main: ['main', 'default', 'codex-main', 'codex-default'],
|
|
252
|
+
review: ['review', 'codex-review'],
|
|
253
|
+
subagent: ['subagent', 'codex-subagent']
|
|
254
|
+
};
|
|
255
|
+
for (const [slot, names] of Object.entries(aliases)) {
|
|
256
|
+
if (names.includes(m)) return modelForSlot(profile, slot) || clean;
|
|
257
|
+
}
|
|
258
|
+
// The official names the CLI was given (publicModels) carry the role. Resolve
|
|
259
|
+
// them before the fail-closed rule below, otherwise review and subagent traffic
|
|
260
|
+
// collapses onto the main slot.
|
|
261
|
+
for (const slot of CODEX_MODEL_SLOTS) {
|
|
262
|
+
const publicName = codexPublicModel(profile, slot);
|
|
263
|
+
if (publicName && publicName.toLowerCase() === m) return modelForSlot(profile, slot) || clean;
|
|
264
|
+
}
|
|
265
|
+
// Codex model IDs change frequently. Bare OpenAI IDs (config leftovers like gpt-5.6-sol,
|
|
266
|
+
// retired gpt-5.3-codex) have no credentials behind 9Router -> fail closed to the main slot
|
|
267
|
+
// instead of passing through to an upstream 404. Provider-prefixed names (ag/..., cf/...)
|
|
268
|
+
// are preserved by the early return above.
|
|
269
|
+
if (/^(gpt|o\d|codex)([-/]|$)/i.test(clean)) return modelForSlot(profile, 'main') || clean;
|
|
270
|
+
return clean || requestedModel;
|
|
271
|
+
}
|
|
272
|
+
if (clientFormat === 'openai-chat' || clientFormat === 'vertex') {
|
|
273
|
+
if (m === 'default' || m === 'main' || !clean) return modelForSlot(profile, 'default') || clean;
|
|
274
|
+
}
|
|
275
|
+
if (m.includes('fable')) return modelForSlot(profile, 'fable') || clean;
|
|
276
|
+
if (m.includes('opus')) return modelForSlot(profile, 'opus') || clean;
|
|
277
|
+
if (m.includes('haiku')) return modelForSlot(profile, 'haiku') || clean;
|
|
278
|
+
if (m.includes('sonnet')) return modelForSlot(profile, 'sonnet') || clean;
|
|
279
|
+
return clean || requestedModel;
|
|
280
|
+
}
|
|
281
|
+
|
|
282
|
+
// Wait until a slow client has taken the buffered bytes, so the gateway does not read the whole
|
|
283
|
+
// upstream stream into memory. A closed client, or an aborted turn, also ends the wait.
|
|
284
|
+
function drained(stream, signal) {
|
|
285
|
+
if (!stream.writableNeedDrain || stream.destroyed || signal?.aborted) return null;
|
|
286
|
+
return new Promise(resolve => {
|
|
287
|
+
const done = () => {
|
|
288
|
+
stream.off('drain', done);
|
|
289
|
+
stream.off('close', done);
|
|
290
|
+
signal?.removeEventListener('abort', done);
|
|
291
|
+
resolve();
|
|
292
|
+
};
|
|
293
|
+
stream.on('drain', done);
|
|
294
|
+
stream.on('close', done);
|
|
295
|
+
signal?.addEventListener('abort', done);
|
|
296
|
+
});
|
|
297
|
+
}
|
|
298
|
+
|
|
299
|
+
function sendSSE(res, event, data) {
|
|
300
|
+
// event=null -> raw `data:` line (OpenAI Chat / Vertex clients).
|
|
301
|
+
if (event) res.write(`event: ${event}\ndata: ${JSON.stringify(data)}\n\n`);
|
|
302
|
+
else res.write(`data: ${JSON.stringify(data)}\n\n`);
|
|
303
|
+
}
|
|
304
|
+
|
|
305
|
+
// ----------------------------------------------------
|
|
306
|
+
// Client-shaped errors
|
|
307
|
+
// ----------------------------------------------------
|
|
308
|
+
function anthropicErrorType(status) {
|
|
309
|
+
switch (status) {
|
|
310
|
+
case 400: return 'invalid_request_error';
|
|
311
|
+
case 401: return 'authentication_error';
|
|
312
|
+
case 403: return 'permission_error';
|
|
313
|
+
case 404: return 'not_found_error';
|
|
314
|
+
case 413: return 'request_too_large';
|
|
315
|
+
case 429: return 'rate_limit_error';
|
|
316
|
+
case 529: return 'overloaded_error';
|
|
317
|
+
default: return status >= 500 ? 'api_error' : 'invalid_request_error';
|
|
318
|
+
}
|
|
319
|
+
}
|
|
320
|
+
|
|
321
|
+
function vertexStatus(status) {
|
|
322
|
+
switch (status) {
|
|
323
|
+
case 400: return 'INVALID_ARGUMENT';
|
|
324
|
+
case 401: return 'UNAUTHENTICATED';
|
|
325
|
+
case 403: return 'PERMISSION_DENIED';
|
|
326
|
+
case 404: return 'NOT_FOUND';
|
|
327
|
+
case 429: return 'RESOURCE_EXHAUSTED';
|
|
328
|
+
case 503: return 'UNAVAILABLE';
|
|
329
|
+
default: return status >= 500 ? 'INTERNAL' : 'FAILED_PRECONDITION';
|
|
330
|
+
}
|
|
331
|
+
}
|
|
332
|
+
|
|
333
|
+
// Each SDK parses errors in its own shape; Claude Code relies on error.type + retry-after to decide retries.
|
|
334
|
+
function sendClientError(res, clientFormat, status, message, headers = {}) {
|
|
335
|
+
let body;
|
|
336
|
+
if (clientFormat === 'anthropic') {
|
|
337
|
+
body = { type: 'error', error: { type: anthropicErrorType(status), message } };
|
|
338
|
+
} else if (clientFormat === 'vertex') {
|
|
339
|
+
body = { error: { code: status, message, status: vertexStatus(status) } };
|
|
340
|
+
} else {
|
|
341
|
+
body = { error: { message, type: status >= 500 ? 'server_error' : 'invalid_request_error', code: String(status) } };
|
|
342
|
+
}
|
|
343
|
+
sendJson(res, status, body, headers);
|
|
344
|
+
}
|
|
345
|
+
|
|
346
|
+
function extractUpstreamMessage(text) {
|
|
347
|
+
try {
|
|
348
|
+
const j = JSON.parse(text);
|
|
349
|
+
const e = Array.isArray(j) ? j[0]?.error : j.error;
|
|
350
|
+
if (typeof e === 'string') return e;
|
|
351
|
+
if (e?.message) return e.message;
|
|
352
|
+
if (j.message) return j.message;
|
|
353
|
+
} catch {}
|
|
354
|
+
return text;
|
|
355
|
+
}
|
|
356
|
+
|
|
357
|
+
const RETRY_HEADERS = ['retry-after', 'retry-after-ms', 'x-should-retry'];
|
|
358
|
+
function pickRetryHeaders(upstreamRes) {
|
|
359
|
+
const h = {};
|
|
360
|
+
for (const k of RETRY_HEADERS) {
|
|
361
|
+
const v = upstreamRes.headers.get(k);
|
|
362
|
+
if (v) h[k] = v;
|
|
363
|
+
}
|
|
364
|
+
return h;
|
|
365
|
+
}
|
|
366
|
+
|
|
367
|
+
// ----------------------------------------------------
|
|
368
|
+
// Upstream endpoint & headers
|
|
369
|
+
// ----------------------------------------------------
|
|
370
|
+
function resolveOutFormat(profile, mappedModel) {
|
|
371
|
+
if (profile.outFormat && OUT_FORMATS.includes(profile.outFormat)) return profile.outFormat;
|
|
372
|
+
const mode = profile.mode || 'hybrid';
|
|
373
|
+
if (mode === 'direct') return 'anthropic';
|
|
374
|
+
if (mode === 'convert') return 'openai-chat';
|
|
375
|
+
return String(mappedModel || '').toLowerCase().startsWith('claude-') ? 'anthropic' : 'openai-chat';
|
|
376
|
+
}
|
|
377
|
+
|
|
378
|
+
// Client headers that must NOT be forwarded upstream: client credentials (e.g. the Gemini SDK's x-goog-api-key
|
|
379
|
+
// would leak to a third-party upstream), switcher control headers, and hop-by-hop / network identity headers.
|
|
380
|
+
// x-intact-trace is on this list because a client must never choose the id that joins the two
|
|
381
|
+
// halves of a captured exchange. Only the contract lab of this gateway writes it.
|
|
382
|
+
const BLOCKED_PASSTHROUGH = new Set([
|
|
383
|
+
'x-api-key', 'x-goog-api-key', 'x-goog-user-project', 'x-profile', 'x-llm-profile',
|
|
384
|
+
'x-real-ip', 'x-forwarded-for', 'x-forwarded-host', 'x-forwarded-proto', 'x-forwarded-port',
|
|
385
|
+
'x-intact-trace', PROBE_HEADER, 'x-llm-switcher-token'
|
|
386
|
+
]);
|
|
387
|
+
|
|
388
|
+
function upstreamEndpoint(profile, outFormat, model, stream, req) {
|
|
389
|
+
const base = String(profile.baseURL || '').replace(/\/+$/, '');
|
|
390
|
+
const ov = profile.endpoints || {};
|
|
391
|
+
const key = profile.apiKey || '';
|
|
392
|
+
const headers = { 'Content-Type': 'application/json' };
|
|
393
|
+
|
|
394
|
+
// Safely pass through client headers from intermediate tools (x-request-id, traceparent, x-...)
|
|
395
|
+
if (req?.headers) {
|
|
396
|
+
for (const [k, v] of Object.entries(req.headers)) {
|
|
397
|
+
const lk = k.toLowerCase();
|
|
398
|
+
if ((lk.startsWith('x-') || lk === 'traceparent' || lk === 'tracestate') && !BLOCKED_PASSTHROUGH.has(lk)) {
|
|
399
|
+
headers[lk] = v;
|
|
400
|
+
}
|
|
401
|
+
}
|
|
402
|
+
}
|
|
403
|
+
|
|
404
|
+
if (outFormat === 'anthropic') {
|
|
405
|
+
headers['x-api-key'] = key;
|
|
406
|
+
headers['anthropic-version'] = req?.headers?.['anthropic-version'] || '2023-06-01';
|
|
407
|
+
if (req?.headers?.['anthropic-beta']) headers['anthropic-beta'] = req.headers['anthropic-beta'];
|
|
408
|
+
} else {
|
|
409
|
+
headers['authorization'] = `Bearer ${key}`;
|
|
410
|
+
}
|
|
411
|
+
|
|
412
|
+
if (outFormat === 'anthropic') {
|
|
413
|
+
return { url: ov.anthropic || `${base}/messages`, headers };
|
|
414
|
+
}
|
|
415
|
+
if (outFormat === 'vertex') {
|
|
416
|
+
const action = stream ? 'streamGenerateContent' : 'generateContent';
|
|
417
|
+
const url = ov.vertex
|
|
418
|
+
? ov.vertex.replace('{model}', encodeURIComponent(model)).replace('{action}', action)
|
|
419
|
+
: `${base}/models/${encodeURIComponent(model)}:${action}`;
|
|
420
|
+
return { url: stream && !/[?&]alt=sse/.test(url) ? `${url}${url.includes('?') ? '&' : '?'}alt=sse` : url, headers };
|
|
421
|
+
}
|
|
422
|
+
return { url: ov['openai-chat'] || `${base}/chat/completions`, headers };
|
|
423
|
+
}
|
|
424
|
+
|
|
425
|
+
// Read the upstream stream: accept both SSE `data:` and raw JSON lines (Vertex framing).
|
|
426
|
+
async function* readUpstreamPayloads(upstreamRes) {
|
|
427
|
+
const reader = upstreamRes.body.getReader();
|
|
428
|
+
const decoder = new TextDecoder('utf8');
|
|
429
|
+
let buffer = '';
|
|
430
|
+
let finished = false;
|
|
431
|
+
try {
|
|
432
|
+
while (true) {
|
|
433
|
+
const { done, value } = await reader.read();
|
|
434
|
+
if (done) {
|
|
435
|
+
finished = true;
|
|
436
|
+
break;
|
|
437
|
+
}
|
|
438
|
+
buffer += decoder.decode(value, { stream: true });
|
|
439
|
+
const lines = buffer.split('\n');
|
|
440
|
+
buffer = lines.pop();
|
|
441
|
+
for (const line of lines) {
|
|
442
|
+
const parsed = parsePayloadLine(line);
|
|
443
|
+
if (parsed) yield parsed;
|
|
444
|
+
}
|
|
445
|
+
}
|
|
446
|
+
buffer += decoder.decode();
|
|
447
|
+
const tail = parsePayloadLine(buffer);
|
|
448
|
+
if (tail) yield tail;
|
|
449
|
+
} finally {
|
|
450
|
+
// A consumer that stops early (an in-band error) must close the upstream connection too.
|
|
451
|
+
if (!finished) await reader.cancel().catch(() => {});
|
|
452
|
+
try { reader.releaseLock(); } catch {}
|
|
453
|
+
}
|
|
454
|
+
}
|
|
455
|
+
|
|
456
|
+
function parsePayloadLine(line) {
|
|
457
|
+
let s = line.trim();
|
|
458
|
+
if (!s) return null;
|
|
459
|
+
if (s.startsWith('data:')) s = s.slice(5).trim();
|
|
460
|
+
else if (s.startsWith('event:') || s.startsWith(':') || s.startsWith('id:') || s.startsWith('retry:')) return null;
|
|
461
|
+
if (!s || s === '[DONE]' || !s.startsWith('{')) return null;
|
|
462
|
+
try { return JSON.parse(s); } catch { return null; }
|
|
463
|
+
}
|
|
464
|
+
|
|
465
|
+
function clientRenderer(clientFormat, res, model, opts = {}) {
|
|
466
|
+
if (clientFormat === 'anthropic') return createAnthropicStream((e, d) => sendSSE(res, e, d), model);
|
|
467
|
+
if (clientFormat === 'responses') return createResponsesStream((e, d) => sendSSE(res, e, d), model, opts);
|
|
468
|
+
if (clientFormat === 'vertex') return createVertexStream((e, d) => sendSSE(res, null, d), model);
|
|
469
|
+
return createChatStream((e, d) => sendSSE(res, null, d), model);
|
|
470
|
+
}
|
|
471
|
+
|
|
472
|
+
function clientMessage(clientFormat, args) {
|
|
473
|
+
if (clientFormat === 'anthropic') return buildAnthropicMessage(args);
|
|
474
|
+
if (clientFormat === 'responses') return buildResponsesMessage(args);
|
|
475
|
+
if (clientFormat === 'vertex') return buildVertexMessage(args);
|
|
476
|
+
return buildChatMessage(args);
|
|
477
|
+
}
|
|
478
|
+
|
|
479
|
+
function previewOf(ir) {
|
|
480
|
+
const lastUser = (ir.messages || []).slice().reverse().find(m => m.role === 'user');
|
|
481
|
+
if (!lastUser) return '';
|
|
482
|
+
if (typeof lastUser.content === 'string') return lastUser.content.slice(0, 300);
|
|
483
|
+
if (Array.isArray(lastUser.content)) {
|
|
484
|
+
return lastUser.content.map(p => (p.type === 'text' ? p.text : `[${p.type}]`)).join(' ').slice(0, 300);
|
|
485
|
+
}
|
|
486
|
+
return '';
|
|
487
|
+
}
|
|
488
|
+
|
|
489
|
+
// ----------------------------------------------------
|
|
490
|
+
// Mode 1: DIRECT PASS-THROUGH (Native Anthropic to Native Anthropic)
|
|
491
|
+
// ----------------------------------------------------
|
|
492
|
+
const HOP_BY_HOP = new Set(['content-length', 'content-encoding', 'transfer-encoding', 'connection', 'keep-alive']);
|
|
493
|
+
|
|
494
|
+
async function forwardAnthropicDirect(res, payload, bodyBuffer, url, headers, mappedModel, signal, profile) {
|
|
495
|
+
debugLog('Direct forward to native Anthropic endpoint:', url);
|
|
496
|
+
|
|
497
|
+
// Only re-serialize when a fix is actually needed; otherwise forward the client's original bytes.
|
|
498
|
+
let json = payload;
|
|
499
|
+
let modified = false;
|
|
500
|
+
if (mappedModel && json.model !== mappedModel) {
|
|
501
|
+
json = { ...json, model: mappedModel };
|
|
502
|
+
modified = true;
|
|
503
|
+
}
|
|
504
|
+
const healed = healAnthropicPayload(json);
|
|
505
|
+
if (healed.changed) {
|
|
506
|
+
json = healed.payload;
|
|
507
|
+
modified = true;
|
|
508
|
+
debugLog('Healer (direct):', healed.notes.join('; '));
|
|
509
|
+
}
|
|
510
|
+
if (profile?.thinkingMode === 'off' && json.thinking) {
|
|
511
|
+
json = { ...json };
|
|
512
|
+
delete json.thinking;
|
|
513
|
+
modified = true;
|
|
514
|
+
}
|
|
515
|
+
const body = modified ? Buffer.from(JSON.stringify(json), 'utf8') : bodyBuffer;
|
|
516
|
+
|
|
517
|
+
const upstreamRes = await fetch(url, { method: 'POST', headers, body, signal });
|
|
518
|
+
const resHeaders = {};
|
|
519
|
+
for (const [k, v] of upstreamRes.headers.entries()) {
|
|
520
|
+
if (!HOP_BY_HOP.has(k.toLowerCase())) resHeaders[k] = v;
|
|
521
|
+
}
|
|
522
|
+
res.writeHead(upstreamRes.status, resHeaders);
|
|
523
|
+
|
|
524
|
+
let errorPreview = '';
|
|
525
|
+
const usage = createUsageTap();
|
|
526
|
+
const reader = upstreamRes.body.getReader();
|
|
527
|
+
let status = upstreamRes.status;
|
|
528
|
+
let streamError = null;
|
|
529
|
+
try {
|
|
530
|
+
while (true) {
|
|
531
|
+
const { done, value } = await reader.read();
|
|
532
|
+
if (done) break;
|
|
533
|
+
if (!upstreamRes.ok && errorPreview.length < 300) errorPreview += Buffer.from(value).toString('utf8');
|
|
534
|
+
usage.push(value);
|
|
535
|
+
res.write(value);
|
|
536
|
+
await drained(res);
|
|
537
|
+
}
|
|
538
|
+
} catch (streamErr) {
|
|
539
|
+
// The status line is already sent, so only the log can say that the stream broke.
|
|
540
|
+
if (signal.aborted) {
|
|
541
|
+
status = 499;
|
|
542
|
+
streamError = 'client disconnected mid-stream';
|
|
543
|
+
} else {
|
|
544
|
+
console.error('[DirectForward] Stream error:', streamErr.message);
|
|
545
|
+
status = 502;
|
|
546
|
+
streamError = `stream interrupted: ${streamErr.cause?.message || streamErr.message}`;
|
|
547
|
+
}
|
|
548
|
+
} finally {
|
|
549
|
+
res.end();
|
|
550
|
+
}
|
|
551
|
+
const error = streamError || (upstreamRes.ok ? null : errorPreview.slice(0, 300));
|
|
552
|
+
return { status, error, healed: healed.notes, tokens: usage.tokens() };
|
|
553
|
+
}
|
|
554
|
+
|
|
555
|
+
// Reads the token counts out of the passthrough bytes without changing them: message_start and
|
|
556
|
+
// message_delta in a stream, `usage` in a JSON body.
|
|
557
|
+
function createUsageTap(limit = 4 * 1024 * 1024) {
|
|
558
|
+
const decoder = new TextDecoder('utf8');
|
|
559
|
+
let text = '';
|
|
560
|
+
let seen = 0;
|
|
561
|
+
const tokens = { prompt: 0, completion: 0 };
|
|
562
|
+
const take = (u) => {
|
|
563
|
+
if (!u || typeof u !== 'object') return;
|
|
564
|
+
if (typeof u.input_tokens === 'number') tokens.prompt = u.input_tokens + (u.cache_read_input_tokens || 0) + (u.cache_creation_input_tokens || 0);
|
|
565
|
+
if (typeof u.output_tokens === 'number') tokens.completion = u.output_tokens;
|
|
566
|
+
};
|
|
567
|
+
const line = (l) => {
|
|
568
|
+
const t = l.startsWith('data:') ? l.slice(5).trim() : '';
|
|
569
|
+
if (!t.includes('"usage"')) return;
|
|
570
|
+
try {
|
|
571
|
+
const d = JSON.parse(t);
|
|
572
|
+
take(d.message?.usage || d.usage);
|
|
573
|
+
} catch {}
|
|
574
|
+
};
|
|
575
|
+
return {
|
|
576
|
+
push(chunk) {
|
|
577
|
+
if (seen > limit) return;
|
|
578
|
+
seen += chunk.length;
|
|
579
|
+
text += decoder.decode(chunk, { stream: true });
|
|
580
|
+
const lines = text.split('\n');
|
|
581
|
+
text = lines.pop();
|
|
582
|
+
for (const l of lines) line(l);
|
|
583
|
+
},
|
|
584
|
+
tokens() {
|
|
585
|
+
text += decoder.decode();
|
|
586
|
+
if (text.trim().startsWith('{')) {
|
|
587
|
+
try { take(JSON.parse(text).usage); } catch {}
|
|
588
|
+
} else if (text) line(text);
|
|
589
|
+
return tokens;
|
|
590
|
+
}
|
|
591
|
+
};
|
|
592
|
+
}
|
|
593
|
+
|
|
594
|
+
function cleanSchemaDeep(obj) {
|
|
595
|
+
if (!obj || typeof obj !== 'object') return obj;
|
|
596
|
+
if (Array.isArray(obj)) return obj.map(cleanSchemaDeep);
|
|
597
|
+
const res = {};
|
|
598
|
+
for (const [k, v] of Object.entries(obj)) {
|
|
599
|
+
if (k === 'encrypted' || k === '$schema' || k === 'cache_control') continue;
|
|
600
|
+
if (k === 'properties' && v && typeof v === 'object') {
|
|
601
|
+
const cleanProps = {};
|
|
602
|
+
for (const [pk, pv] of Object.entries(v)) {
|
|
603
|
+
if (typeof pv === 'string') {
|
|
604
|
+
cleanProps[pk] = { type: pv === 'object' ? 'object' : pv, ...(pv === 'object' ? { properties: {} } : {}) };
|
|
605
|
+
} else {
|
|
606
|
+
cleanProps[pk] = cleanSchemaDeep(pv);
|
|
607
|
+
}
|
|
608
|
+
}
|
|
609
|
+
res[k] = cleanProps;
|
|
610
|
+
} else {
|
|
611
|
+
res[k] = cleanSchemaDeep(v);
|
|
612
|
+
}
|
|
613
|
+
}
|
|
614
|
+
if (!res.type && res.properties) res.type = 'object';
|
|
615
|
+
if (res.type === 'object' && !res.properties) res.properties = {};
|
|
616
|
+
return res;
|
|
617
|
+
}
|
|
618
|
+
|
|
619
|
+
// The upstream request of the HTTP and the WS transport: the IR as an upstream body, with tool schemas
|
|
620
|
+
// made acceptable to the target:
|
|
621
|
+
// 1. Strip disallowed keywords ('encrypted', '$schema', 'cache_control')
|
|
622
|
+
// 2. Fix invalid schema values where a property has a string value "object" instead of a valid schema object
|
|
623
|
+
// 3. For ag/* targets (Gemini behind 9Router): rewrite to the strict Schema subset
|
|
624
|
+
function buildUpstreamRequest(profile, outFormat, ir, mappedModel, req) {
|
|
625
|
+
const upBody = emitUpstreamBody(outFormat, ir, mappedModel, { thinkingMode: profile.thinkingMode });
|
|
626
|
+
if (upBody?.tools && Array.isArray(upBody.tools)) {
|
|
627
|
+
upBody.tools = cleanSchemaDeep(upBody.tools);
|
|
628
|
+
if (outFormat === 'openai-chat' && isAntigravityModel(mappedModel)) {
|
|
629
|
+
upBody.tools = geminiSafeTools(upBody.tools);
|
|
630
|
+
}
|
|
631
|
+
}
|
|
632
|
+
const { url, headers } = upstreamEndpoint(profile, outFormat, mappedModel, ir.stream, req);
|
|
633
|
+
return { url, headers, upBody };
|
|
634
|
+
}
|
|
635
|
+
|
|
636
|
+
// Feeds upstream stream events to a client renderer until the stream ends, fails or is aborted.
|
|
637
|
+
// Returns the stream error, or null. `sink` is the client stream whose backpressure it waits for.
|
|
638
|
+
async function pumpStream(upstreamRes, { normalize, col, renderer, splitter, signal, sink, tag }) {
|
|
639
|
+
let streamError = null;
|
|
640
|
+
let events = 0;
|
|
641
|
+
try {
|
|
642
|
+
for await (const parsed of readUpstreamPayloads(upstreamRes)) {
|
|
643
|
+
events++;
|
|
644
|
+
const ev = normalize(parsed);
|
|
645
|
+
col.add(ev);
|
|
646
|
+
if (ev.error) {
|
|
647
|
+
streamError = ev.error;
|
|
648
|
+
break;
|
|
649
|
+
}
|
|
650
|
+
for (const t of ev.think) renderer.think(t.text, t.sig);
|
|
651
|
+
if (ev.sig) renderer.think('', ev.sig);
|
|
652
|
+
for (const t of ev.text) splitter.push(t);
|
|
653
|
+
if (ev.tools.length) {
|
|
654
|
+
splitter.flush();
|
|
655
|
+
for (const tc of ev.tools) renderer.tool(tc);
|
|
656
|
+
}
|
|
657
|
+
await drained(sink, signal);
|
|
658
|
+
}
|
|
659
|
+
if (!streamError && events === 0) streamError = 'Upstream returned an empty stream';
|
|
660
|
+
} catch (streamErr) {
|
|
661
|
+
if (!signal.aborted) {
|
|
662
|
+
console.error(`[${tag}] Stream error:`, streamErr.message);
|
|
663
|
+
streamError = streamErr.message || 'stream interrupted';
|
|
664
|
+
}
|
|
665
|
+
}
|
|
666
|
+
return streamError;
|
|
667
|
+
}
|
|
668
|
+
|
|
669
|
+
// Off unless config.json carries a contractLab block with enabled: true. Every call returns at
|
|
670
|
+
// once, so a slow or absent intact cannot reach the answer the client is waiting for.
|
|
671
|
+
const contractLab = createContractLab({ settings: () => contractLabSettings(loadConfig()) });
|
|
672
|
+
|
|
673
|
+
// ----------------------------------------------------
|
|
674
|
+
// LLM Switcher generic pipeline: client --parse--> IR --emit--> upstream
|
|
675
|
+
// Client (input) formats : anthropic | openai-chat | responses (Codex) | vertex
|
|
676
|
+
// Upstream (output) : profile.outFormat or legacy mode mapping
|
|
677
|
+
// direct -> anthropic | convert -> openai-chat
|
|
678
|
+
// hybrid -> claude-* via native anthropic, others via openai-chat
|
|
679
|
+
// ----------------------------------------------------
|
|
680
|
+
async function handleConvert(clientFormat, req, res, bodyBuffer, opts = {}) {
|
|
681
|
+
const { profileKey, profile, error: profileError } = getActiveProfile(clientFormat, req);
|
|
682
|
+
if (!loadConfig()) {
|
|
683
|
+
sendClientError(res, clientFormat, 500, `LLM Switcher config not loaded (${configPath}): ${getConfigLoadError()?.message || 'missing file'}`);
|
|
684
|
+
return;
|
|
685
|
+
}
|
|
686
|
+
if (!profile) {
|
|
687
|
+
sendClientError(res, clientFormat, profileError ? 400 : 503, profileError ||
|
|
688
|
+
`Proxy is currently OFF for ${clientFormat}. Set an active profile for this target in Web UI or via switch command.`);
|
|
689
|
+
return;
|
|
690
|
+
}
|
|
691
|
+
|
|
692
|
+
let payload;
|
|
693
|
+
try {
|
|
694
|
+
payload = JSON.parse(bodyBuffer.toString('utf8'));
|
|
695
|
+
} catch {
|
|
696
|
+
sendClientError(res, clientFormat, 400, 'Invalid JSON body');
|
|
697
|
+
return;
|
|
698
|
+
}
|
|
699
|
+
|
|
700
|
+
const wantIn = profile.inFormat || 'auto';
|
|
701
|
+
if (wantIn !== 'auto' && wantIn !== clientFormat) {
|
|
702
|
+
sendClientError(res, clientFormat, 400, `Profile "${profileKey}" expects "${wantIn}" input, got "${clientFormat}"`);
|
|
703
|
+
return;
|
|
704
|
+
}
|
|
705
|
+
|
|
706
|
+
let ir;
|
|
707
|
+
try {
|
|
708
|
+
ir = parseToIR(clientFormat, payload);
|
|
709
|
+
} catch (e) {
|
|
710
|
+
sendClientError(res, clientFormat, 400, `Cannot parse ${clientFormat} request: ${e.message}`);
|
|
711
|
+
return;
|
|
712
|
+
}
|
|
713
|
+
if (clientFormat === 'vertex') {
|
|
714
|
+
ir.stream = Boolean(opts.vertexStream);
|
|
715
|
+
if (!ir.model && opts.vertexModel) ir.model = opts.vertexModel;
|
|
716
|
+
}
|
|
717
|
+
|
|
718
|
+
const reqStartTime = Date.now();
|
|
719
|
+
const requestPreview = previewOf(ir);
|
|
720
|
+
const requestedModel = ir.model || payload.model || '';
|
|
721
|
+
const mappedModel = mapModel(requestedModel, profile, clientFormat);
|
|
722
|
+
const outFormat = resolveOutFormat(profile, mappedModel);
|
|
723
|
+
console.log(`[llm-switcher] ${clientFormat} -> ${outFormat} "${requestedModel}" -> "${mappedModel}" [${profile.name || profileKey}]`);
|
|
724
|
+
|
|
725
|
+
const logBase = { clientFormat, outFormat, profile: profileKey, model: mappedModel, stream: ir.stream, requestPreview };
|
|
726
|
+
const log = (extra) => logInspection({ ...logBase, duration: Date.now() - reqStartTime, tokens: { prompt: 0, completion: 0 }, ...extra });
|
|
727
|
+
|
|
728
|
+
// Contract lab: a sampled request carries a trace id to intact, and the bytes this gateway
|
|
729
|
+
// writes back are copied for the upload that follows the answer.
|
|
730
|
+
const traceId = probeTraceId(req) || contractLab.traceFor(mappedModel);
|
|
731
|
+
const halfTap = traceId ? createHalfTap() : null;
|
|
732
|
+
if (halfTap) tapClientWrites(res, halfTap);
|
|
733
|
+
|
|
734
|
+
// AbortController to cancel the upstream fetch as soon as the client disconnects (saves tokens)
|
|
735
|
+
const ac = new AbortController();
|
|
736
|
+
const onClientClose = () => {
|
|
737
|
+
if (!res.writableEnded) {
|
|
738
|
+
debugLog(`[${profileKey}] Client connection closed before response ended, aborting upstream request`);
|
|
739
|
+
ac.abort();
|
|
740
|
+
}
|
|
741
|
+
};
|
|
742
|
+
res.on('close', onClientClose);
|
|
743
|
+
let answered = false; // set only after a complete 2xx answer; the half upload depends on it
|
|
744
|
+
|
|
745
|
+
try {
|
|
746
|
+
// Fast path: anthropic in/out goes straight through, preserving original bytes (including thinking signatures).
|
|
747
|
+
// Note: this branch skips the Healer Engine because it bypasses the IR.
|
|
748
|
+
if (clientFormat === 'anthropic' && outFormat === 'anthropic') {
|
|
749
|
+
const { url, headers } = upstreamEndpoint(profile, 'anthropic', mappedModel, ir.stream, req);
|
|
750
|
+
if (traceId) headers['x-intact-trace'] = traceId;
|
|
751
|
+
try {
|
|
752
|
+
const r = await forwardAnthropicDirect(res, payload, bodyBuffer, url, headers, mappedModel, ac.signal, profile);
|
|
753
|
+
answered = !r.error && r.status >= 200 && r.status < 300;
|
|
754
|
+
log({ status: r.status, tokens: r.tokens, responsePreview: r.healed.length ? `(direct forward, healed: ${r.healed.join('; ')})` : '(direct forward)', error: r.error || undefined });
|
|
755
|
+
} catch (err) {
|
|
756
|
+
if (ac.signal.aborted) return log({ status: 499, error: 'client disconnected' });
|
|
757
|
+
console.error(`[${profileKey}] Direct forward error:`, err.message);
|
|
758
|
+
sendClientError(res, clientFormat, 502, `Direct forward error: ${err.message}`);
|
|
759
|
+
log({ status: 502, error: err.message });
|
|
760
|
+
}
|
|
761
|
+
return;
|
|
762
|
+
}
|
|
763
|
+
|
|
764
|
+
const { url, headers, upBody } = buildUpstreamRequest(profile, outFormat, ir, mappedModel, req);
|
|
765
|
+
if (traceId) headers['x-intact-trace'] = traceId;
|
|
766
|
+
debugLog(`[${profileKey}] ${clientFormat} -> ${outFormat} ${url} ::`, JSON.stringify(upBody).slice(0, 500));
|
|
767
|
+
|
|
768
|
+
let upstreamRes;
|
|
769
|
+
try {
|
|
770
|
+
upstreamRes = await fetch(url, { method: 'POST', headers, body: JSON.stringify(upBody), signal: ac.signal });
|
|
771
|
+
} catch (fetchErr) {
|
|
772
|
+
if (ac.signal.aborted) return log({ status: 499, error: 'client disconnected' });
|
|
773
|
+
console.error(`[${profileKey}] Network error:`, fetchErr.message);
|
|
774
|
+
sendClientError(res, clientFormat, 502, `Failed to connect to upstream: ${fetchErr.cause?.message || fetchErr.message}`);
|
|
775
|
+
return log({ status: 502, error: fetchErr.message });
|
|
776
|
+
}
|
|
777
|
+
|
|
778
|
+
if (!upstreamRes.ok) {
|
|
779
|
+
const errText = await upstreamRes.text().catch(() => '');
|
|
780
|
+
console.error(`[${profileKey}] Error HTTP ${upstreamRes.status}:`, errText.slice(0, 500));
|
|
781
|
+
sendClientError(res, clientFormat, upstreamRes.status, extractUpstreamMessage(errText) || `Upstream HTTP ${upstreamRes.status}`, pickRetryHeaders(upstreamRes));
|
|
782
|
+
return log({ status: upstreamRes.status, error: errText.slice(0, 300) });
|
|
783
|
+
}
|
|
784
|
+
|
|
785
|
+
const normalize = createUpstreamNormalizer(outFormat);
|
|
786
|
+
const col = createCollector();
|
|
787
|
+
|
|
788
|
+
// ---- non-stream ----
|
|
789
|
+
if (!ir.stream) {
|
|
790
|
+
let json;
|
|
791
|
+
try {
|
|
792
|
+
json = await upstreamRes.json();
|
|
793
|
+
} catch (e) {
|
|
794
|
+
if (ac.signal.aborted) return log({ status: 499, error: 'client disconnected' });
|
|
795
|
+
sendClientError(res, clientFormat, 502, `Upstream returned non-JSON: ${e.message}`);
|
|
796
|
+
return log({ status: 502, error: e.message });
|
|
797
|
+
}
|
|
798
|
+
col.add(normalize(json));
|
|
799
|
+
if (col.error) {
|
|
800
|
+
sendClientError(res, clientFormat, 502, `Upstream error: ${col.error}`);
|
|
801
|
+
return log({ status: 502, error: String(col.error).slice(0, 300) });
|
|
802
|
+
}
|
|
803
|
+
const split = splitThinkTags(col.text.join(''));
|
|
804
|
+
const think = [...col.think, split.think].filter(Boolean);
|
|
805
|
+
const tools = [...col.tools.values()].sort((a, b) => a.index - b.index);
|
|
806
|
+
const completion = col.completion();
|
|
807
|
+
const out = clientMessage(clientFormat, {
|
|
808
|
+
model: requestedModel || mappedModel, think, text: [split.text], tools, toolMeta: ir.toolMeta,
|
|
809
|
+
finish: col.finish, prompt: col.prompt, completion, cached: col.cached,
|
|
810
|
+
reasoning: col.reasoning, sig: col.sig
|
|
811
|
+
});
|
|
812
|
+
sendJson(res, 200, out);
|
|
813
|
+
answered = true;
|
|
814
|
+
return log({
|
|
815
|
+
status: 200, tokens: { prompt: col.prompt, completion },
|
|
816
|
+
thinkingChars: think.join('').length, responsePreview: split.text.slice(0, 300)
|
|
817
|
+
});
|
|
818
|
+
}
|
|
819
|
+
|
|
820
|
+
// ---- stream ----
|
|
821
|
+
res.writeHead(200, {
|
|
822
|
+
'Content-Type': 'text/event-stream; charset=utf-8',
|
|
823
|
+
'Cache-Control': 'no-cache',
|
|
824
|
+
'Connection': 'keep-alive',
|
|
825
|
+
'X-Accel-Buffering': 'no'
|
|
826
|
+
});
|
|
827
|
+
const renderer = clientRenderer(clientFormat, res, requestedModel || mappedModel, { toolMeta: ir.toolMeta });
|
|
828
|
+
renderer.start();
|
|
829
|
+
const splitter = createThinkTagSplitter(t => renderer.think(t), t => renderer.text(t));
|
|
830
|
+
const streamError = await pumpStream(upstreamRes, { normalize, col, renderer, splitter, signal: ac.signal, sink: res, tag: profileKey });
|
|
831
|
+
|
|
832
|
+
if (ac.signal.aborted) {
|
|
833
|
+
return log({ status: 499, error: 'client disconnected mid-stream', responsePreview: col.text.join('').slice(0, 300) });
|
|
834
|
+
}
|
|
835
|
+
splitter.flush();
|
|
836
|
+
const completion = col.completion();
|
|
837
|
+
if (streamError) {
|
|
838
|
+
// Report the error clearly instead of a fake "end_turn" ending -> the client knows the response was cut off and can retry.
|
|
839
|
+
renderer.error(streamError);
|
|
840
|
+
} else {
|
|
841
|
+
renderer.finish(col.finish, { completion, prompt: col.prompt, cached: col.cached, reasoning: col.reasoning, hasTools: col.tools.size > 0 });
|
|
842
|
+
}
|
|
843
|
+
// OpenAI Chat clients expect a terminal [DONE] line (Responses API does not use [DONE]).
|
|
844
|
+
if (clientFormat === 'openai-chat') res.write('data: [DONE]\n\n');
|
|
845
|
+
res.end();
|
|
846
|
+
answered = !streamError;
|
|
847
|
+
log({
|
|
848
|
+
status: streamError ? 502 : 200, stream: true,
|
|
849
|
+
tokens: { prompt: col.prompt, completion },
|
|
850
|
+
thinkingChars: col.think.join('').length,
|
|
851
|
+
responsePreview: col.text.join('').slice(0, 300),
|
|
852
|
+
error: streamError ? String(streamError).slice(0, 300) : undefined
|
|
853
|
+
});
|
|
854
|
+
} finally {
|
|
855
|
+
res.off('close', onClientClose);
|
|
856
|
+
// A half is uploaded only for a complete 2xx answer: anything else would diff as a loss the
|
|
857
|
+
// converter never made.
|
|
858
|
+
if (traceId) {
|
|
859
|
+
finishHalf(contractLab, traceId, ac.signal.aborted || !answered, {
|
|
860
|
+
toolRequest: capText(bodyBuffer),
|
|
861
|
+
toolResponse: halfTap?.text() || '',
|
|
862
|
+
toolVersion: toolVersionFromUA(req.headers['user-agent']),
|
|
863
|
+
inFormat: clientFormat,
|
|
864
|
+
outFormat
|
|
865
|
+
});
|
|
866
|
+
}
|
|
867
|
+
}
|
|
868
|
+
}
|
|
869
|
+
|
|
870
|
+
// Claude Code calls /v1/messages/count_tokens to measure context. Native Anthropic upstream -> ask for the real count;
|
|
871
|
+
// other upstreams have no equivalent endpoint -> estimate (skip image base64, add a fixed cost per image).
|
|
872
|
+
async function handleCountTokens(req, res, buf) {
|
|
873
|
+
let payload;
|
|
874
|
+
try {
|
|
875
|
+
payload = JSON.parse(buf.toString('utf8'));
|
|
876
|
+
} catch {
|
|
877
|
+
return sendClientError(res, 'anthropic', 400, 'Invalid JSON body');
|
|
878
|
+
}
|
|
879
|
+
const { profile } = getActiveProfile('anthropic', req);
|
|
880
|
+
if (profile) {
|
|
881
|
+
const mappedModel = mapModel(payload.model || '', profile, 'anthropic');
|
|
882
|
+
if (resolveOutFormat(profile, mappedModel) === 'anthropic') {
|
|
883
|
+
const { url, headers } = upstreamEndpoint(profile, 'anthropic', mappedModel, false, req);
|
|
884
|
+
const countUrl = profile.endpoints?.countTokens || url.replace(/\/messages$/, '/messages/count_tokens');
|
|
885
|
+
try {
|
|
886
|
+
const body = healAnthropicPayload({ ...payload, model: mappedModel }).payload;
|
|
887
|
+
const r = await fetch(countUrl, { method: 'POST', headers, body: JSON.stringify(body), signal: AbortSignal.timeout(15000) });
|
|
888
|
+
if (r.ok) {
|
|
889
|
+
const j = await r.json();
|
|
890
|
+
if (typeof j.input_tokens === 'number') return sendJson(res, 200, { input_tokens: j.input_tokens });
|
|
891
|
+
} else {
|
|
892
|
+
debugLog(`count_tokens upstream HTTP ${r.status}, falling back to estimate`);
|
|
893
|
+
}
|
|
894
|
+
} catch (err) {
|
|
895
|
+
debugLog('count_tokens upstream failed, falling back to estimate:', err.message);
|
|
896
|
+
}
|
|
897
|
+
}
|
|
898
|
+
}
|
|
899
|
+
return sendJson(res, 200, { input_tokens: estimateTokens(payload) });
|
|
900
|
+
}
|
|
901
|
+
|
|
902
|
+
// ----------------------------------------------------
|
|
903
|
+
// Admin API helpers
|
|
904
|
+
// ----------------------------------------------------
|
|
905
|
+
// loadConfig keeps serving the last good copy when config.json stops parsing. The admin API must
|
|
906
|
+
// not act on that copy: a save would replace the user's hand edit with stale data.
|
|
907
|
+
function requireConfig(res) {
|
|
908
|
+
const cfg = loadConfig();
|
|
909
|
+
const loadError = getConfigLoadError();
|
|
910
|
+
if (cfg && loadError && fs.existsSync(configPath)) {
|
|
911
|
+
sendJson(res, 409, { error: `config.json (${configPath}) does not parse: ${loadError.message}. Fix the file; the gateway does not overwrite it until it parses.` });
|
|
912
|
+
return null;
|
|
913
|
+
}
|
|
914
|
+
if (!cfg) {
|
|
915
|
+
sendJson(res, 500, { error: `Config not loaded (${configPath}): ${loadError?.message || 'missing file'}. Copy config.example.json to config.json.` });
|
|
916
|
+
}
|
|
917
|
+
return cfg;
|
|
918
|
+
}
|
|
919
|
+
|
|
920
|
+
// Config changes and interceptor reconciles run one at a time, from read to save. A change that
|
|
921
|
+
// waits for its interceptor check must not be saved by a concurrent one, and two must not both
|
|
922
|
+
// start an interceptor.
|
|
923
|
+
let adminChain = Promise.resolve();
|
|
924
|
+
function serialized(fn) {
|
|
925
|
+
const run = adminChain.then(fn);
|
|
926
|
+
adminChain = run.catch(() => {});
|
|
927
|
+
return run;
|
|
928
|
+
}
|
|
929
|
+
const CONFIG_CHANGES = new Set(['/api/switch', '/api/toggle', '/api/save-profile', '/api/delete-profile', '/api/blindfold/sync']);
|
|
930
|
+
|
|
931
|
+
// Callers run it inside serialized().
|
|
932
|
+
function reconcile(cfg) {
|
|
933
|
+
return reconcileBlindfold(cfg, PORT).then(r => {
|
|
934
|
+
if (!r.ok) console.error(`[llm-switcher:blindfold] ${r.error}`);
|
|
935
|
+
else if (r.action === 'started') console.log('[llm-switcher:blindfold] interceptor started');
|
|
936
|
+
return r;
|
|
937
|
+
});
|
|
938
|
+
}
|
|
939
|
+
|
|
940
|
+
// Returns what the caller must show: removed settings.json values, and a blindfold failure.
|
|
941
|
+
// `base` is the revision the change started from.
|
|
942
|
+
async function commit(cfg, base) {
|
|
943
|
+
// Refuse before saving: a saved interceptor port that a squatter holds would route Codex through it.
|
|
944
|
+
const problem = await checkBlindfoldTarget(cfg, PORT);
|
|
945
|
+
if (problem) return { success: false, error: `Not saved: ${problem}` };
|
|
946
|
+
// The admin lock covers this process only. The CLI can save config.json during the check above,
|
|
947
|
+
// and a save now would overwrite it.
|
|
948
|
+
const current = loadConfig();
|
|
949
|
+
if (!current || configRevision(current) !== base) {
|
|
950
|
+
return { success: false, status: 409, error: 'Not saved: config.json changed while this change was checked. Reload and try again.' };
|
|
951
|
+
}
|
|
952
|
+
saveConfig(cfg);
|
|
953
|
+
const revision = configRevision(cfg);
|
|
954
|
+
const st = applyLaunchState(cfg, PORT);
|
|
955
|
+
const removed = st.settings?.removed || [];
|
|
956
|
+
if (removed.length) console.log(`[llm-switcher] settings.json: removed switcher-written values: ${removed.join(', ')}`);
|
|
957
|
+
const bf = await reconcile(cfg);
|
|
958
|
+
return {
|
|
959
|
+
revision,
|
|
960
|
+
settingsRemoved: removed,
|
|
961
|
+
...(st.envWriteError ? { envWriteError: st.envWriteError } : {}),
|
|
962
|
+
...(bf.ok ? {} : { success: false, error: `Saved, but the blindfold interceptor is not in line: ${bf.error}` })
|
|
963
|
+
};
|
|
964
|
+
}
|
|
965
|
+
|
|
966
|
+
const VALID_MODES = ['hybrid', 'convert', 'direct'];
|
|
967
|
+
|
|
968
|
+
const CONTROL_CHARS = /[\u0000-\u001f\u007f-\u009f]/;
|
|
969
|
+
|
|
970
|
+
function validateProfileInput(p) {
|
|
971
|
+
if (!p || typeof p !== 'object' || Array.isArray(p)) return 'Profile must be an object';
|
|
972
|
+
// These strings reach the terminal through `switch status`; a control character could rewrite it.
|
|
973
|
+
for (const k of ['name', 'baseURL', 'optimizerURL']) {
|
|
974
|
+
if (typeof p[k] === 'string' && CONTROL_CHARS.test(p[k])) return `${k} must not contain control characters`;
|
|
975
|
+
}
|
|
976
|
+
if (p.inFormat && p.inFormat !== 'auto' && !IN_FORMATS.includes(p.inFormat)) return `Invalid inFormat "${p.inFormat}"`;
|
|
977
|
+
if (p.outFormat && !OUT_FORMATS.includes(p.outFormat)) return `Invalid outFormat "${p.outFormat}"`;
|
|
978
|
+
if (p.mode && !VALID_MODES.includes(p.mode)) return `Invalid mode "${p.mode}"`;
|
|
979
|
+
if (p.thinkingMode && !THINKING_MODES.includes(p.thinkingMode)) return `Invalid thinkingMode "${p.thinkingMode}"`;
|
|
980
|
+
if (p.baseURL !== undefined) {
|
|
981
|
+
try {
|
|
982
|
+
const u = new URL(p.baseURL);
|
|
983
|
+
if (!['http:', 'https:'].includes(u.protocol)) return 'baseURL must be http(s)';
|
|
984
|
+
} catch {
|
|
985
|
+
return `Invalid baseURL "${p.baseURL}"`;
|
|
986
|
+
}
|
|
987
|
+
}
|
|
988
|
+
// Names the CLI receives end up on a command line. state.mjs drops an unsafe one
|
|
989
|
+
// before the write; refusing it here says why instead of losing it in silence.
|
|
990
|
+
if (p.publicModels !== undefined) {
|
|
991
|
+
if (!Array.isArray(p.publicModels)) return 'publicModels must be an array';
|
|
992
|
+
for (const name of p.publicModels) {
|
|
993
|
+
if (name !== '' && !isSafeModelName(name)) return `Invalid publicModels entry "${name}"`;
|
|
994
|
+
}
|
|
995
|
+
}
|
|
996
|
+
if (p.codexRoles !== undefined) {
|
|
997
|
+
if (!p.codexRoles || typeof p.codexRoles !== 'object' || Array.isArray(p.codexRoles)) {
|
|
998
|
+
return 'codexRoles must be an object';
|
|
999
|
+
}
|
|
1000
|
+
for (const [slot, name] of Object.entries(p.codexRoles)) {
|
|
1001
|
+
if (name !== '' && !isSafeModelName(name)) return `Invalid codexRoles.${slot} value "${name}"`;
|
|
1002
|
+
}
|
|
1003
|
+
}
|
|
1004
|
+
if (p.blindfoldPort !== undefined && p.blindfoldPort !== '' && !parsePort(p.blindfoldPort)) {
|
|
1005
|
+
return `Invalid blindfoldPort "${p.blindfoldPort}"`;
|
|
1006
|
+
}
|
|
1007
|
+
if (p.blindfoldHost !== undefined && p.blindfoldHost !== '' && !/^[A-Za-z0-9.-]{1,253}$/.test(p.blindfoldHost)) {
|
|
1008
|
+
return `Invalid blindfoldHost "${p.blindfoldHost}"`;
|
|
1009
|
+
}
|
|
1010
|
+
if (p.blindfoldPrefix !== undefined && p.blindfoldPrefix !== '' && !/^\/[A-Za-z0-9._~/-]{0,200}$/.test(p.blindfoldPrefix)) {
|
|
1011
|
+
return `Invalid blindfoldPrefix "${p.blindfoldPrefix}"`;
|
|
1012
|
+
}
|
|
1013
|
+
return null;
|
|
1014
|
+
}
|
|
1015
|
+
|
|
1016
|
+
// API keys are masked with MASKED_KEY in the UI; if the client sends back the masked value, reuse the stored real key.
|
|
1017
|
+
// Only for the stored baseURL: otherwise the masked value would send the real key to any host the caller names.
|
|
1018
|
+
const sameBaseURL = (a, b) => String(a || '').replace(/\/+$/, '') === String(b || '').replace(/\/+$/, '');
|
|
1019
|
+
// endpoints override baseURL per format, so they are part of where the key goes.
|
|
1020
|
+
const sameDestination = (a, b) => sameBaseURL(a.baseURL, b.baseURL) && JSON.stringify(a.endpoints || {}) === JSON.stringify(b.endpoints || {});
|
|
1021
|
+
|
|
1022
|
+
// Changes when config.json changes. The dashboard sends it back, so a change made from a stale
|
|
1023
|
+
// page is refused instead of overwriting what another tab or the CLI saved.
|
|
1024
|
+
function configRevision(cfg) {
|
|
1025
|
+
return crypto.createHash('sha256').update(JSON.stringify(cfg)).digest('hex').slice(0, 16);
|
|
1026
|
+
}
|
|
1027
|
+
|
|
1028
|
+
function resolveApiKey(cfg, profileKey, apiKey, baseURL) {
|
|
1029
|
+
if (apiKey !== MASKED_KEY) return apiKey || '';
|
|
1030
|
+
if (!hasProfile(cfg, profileKey)) return '';
|
|
1031
|
+
const stored = cfg.profiles[profileKey];
|
|
1032
|
+
if (baseURL !== undefined && !sameBaseURL(baseURL, stored.baseURL)) return '';
|
|
1033
|
+
return stored.apiKey || '';
|
|
1034
|
+
}
|
|
1035
|
+
|
|
1036
|
+
function upstreamTimeout(ms) {
|
|
1037
|
+
return AbortSignal.timeout(ms);
|
|
1038
|
+
}
|
|
1039
|
+
|
|
1040
|
+
async function testUpstream(body, cfg) {
|
|
1041
|
+
const baseURL = String(body.baseURL || '').replace(/\/+$/, '');
|
|
1042
|
+
if (!baseURL) return { status: 400, json: { ok: false, error: 'Missing baseURL' } };
|
|
1043
|
+
const apiKey = resolveApiKey(cfg, body.key, body.apiKey, baseURL);
|
|
1044
|
+
const model = body.model || 'default';
|
|
1045
|
+
const profile = { baseURL, apiKey, mode: body.mode, outFormat: body.outFormat || undefined };
|
|
1046
|
+
const outFormat = resolveOutFormat(profile, model);
|
|
1047
|
+
const { url, headers } = upstreamEndpoint(profile, outFormat, model, false, null);
|
|
1048
|
+
const ir = { model, system: '', messages: [{ role: 'user', content: 'ping' }], tools: [], toolChoice: null, params: { maxTokens: 16, temperature: null, topP: null, topK: null, stop: [] }, thinking: { type: 'disabled' }, stream: false };
|
|
1049
|
+
const start = Date.now();
|
|
1050
|
+
const r = await fetch(url, { method: 'POST', headers, body: JSON.stringify(emitUpstreamBody(outFormat, ir, model)), signal: upstreamTimeout(30000) });
|
|
1051
|
+
const latency = Date.now() - start;
|
|
1052
|
+
if (!r.ok) return { status: 200, json: { ok: false, status: r.status, latency, outFormat, error: (await r.text()).slice(0, 2000) } };
|
|
1053
|
+
const data = await r.json().catch(() => ({}));
|
|
1054
|
+
const col = createCollector();
|
|
1055
|
+
col.add(createUpstreamNormalizer(outFormat)(data));
|
|
1056
|
+
const sample = col.text.join('') || col.think.join('') || '(ok, empty response)';
|
|
1057
|
+
return { status: 200, json: { ok: true, latency, outFormat, sample: sample.slice(0, 500) } };
|
|
1058
|
+
}
|
|
1059
|
+
|
|
1060
|
+
async function fetchModels(body, cfg) {
|
|
1061
|
+
const baseURL = String(body.baseURL || '').replace(/\/+$/, '');
|
|
1062
|
+
if (!baseURL) return { status: 400, json: { ok: false, error: 'Missing baseURL' } };
|
|
1063
|
+
const apiKey = resolveApiKey(cfg, body.key, body.apiKey, baseURL);
|
|
1064
|
+
const headers = {};
|
|
1065
|
+
if (apiKey) {
|
|
1066
|
+
headers['Authorization'] = `Bearer ${apiKey}`;
|
|
1067
|
+
headers['x-api-key'] = apiKey;
|
|
1068
|
+
}
|
|
1069
|
+
const r = await fetch(`${baseURL}/models`, { headers, signal: upstreamTimeout(15000) });
|
|
1070
|
+
if (!r.ok) return { status: 200, json: { ok: false, status: r.status, error: (await r.text()).slice(0, 2000) } };
|
|
1071
|
+
const data = await r.json();
|
|
1072
|
+
let list = [];
|
|
1073
|
+
if (Array.isArray(data.data)) list = data.data.map(m => (typeof m === 'string' ? m : m.id));
|
|
1074
|
+
else if (Array.isArray(data)) list = data.map(m => (typeof m === 'string' ? m : m.id));
|
|
1075
|
+
else if (Array.isArray(data.models)) list = data.models.map(m => (typeof m === 'string' ? m : (m.id || m.name)));
|
|
1076
|
+
list = [...new Set(list.filter(Boolean).map(id => String(id).replace(/^models\//, '')))].sort();
|
|
1077
|
+
return { status: 200, json: { ok: true, models: list } };
|
|
1078
|
+
}
|
|
1079
|
+
|
|
1080
|
+
// Vertex/Gemini: /v1beta/models/{m}:{action} (Gemini API) and
|
|
1081
|
+
// /v1/projects/{p}/locations/{l}/publishers/{pub}/models/{m}:{action} (Vertex AI SDK).
|
|
1082
|
+
const VERTEX_ROUTE = /^\/(?:v1|v1beta|v1beta1)\/(?:projects\/[^/]+\/locations\/[^/]+\/publishers\/[^/]+\/)?models\/([^/:]+):(generateContent|streamGenerateContent)$/;
|
|
1083
|
+
|
|
1084
|
+
// ----------------------------------------------------
|
|
1085
|
+
// Router
|
|
1086
|
+
// ----------------------------------------------------
|
|
1087
|
+
async function route(req, res) {
|
|
1088
|
+
const parsedUrl = new URL(req.url, 'http://127.0.0.1');
|
|
1089
|
+
const pathname = parsedUrl.pathname;
|
|
1090
|
+
const method = req.method;
|
|
1091
|
+
|
|
1092
|
+
const guardError = checkRequestOrigin(req);
|
|
1093
|
+
if (guardError) {
|
|
1094
|
+
req.resume();
|
|
1095
|
+
return sendJson(res, 403, { error: guardError });
|
|
1096
|
+
}
|
|
1097
|
+
if (method === 'OPTIONS') {
|
|
1098
|
+
// Only answer preflight for the UI's own origin (already past checkRequestOrigin).
|
|
1099
|
+
res.writeHead(204, {
|
|
1100
|
+
'Access-Control-Allow-Origin': req.headers.origin || `http://127.0.0.1:${PORT}`,
|
|
1101
|
+
'Access-Control-Allow-Methods': 'GET, POST, OPTIONS',
|
|
1102
|
+
'Access-Control-Allow-Headers': 'Content-Type, Authorization'
|
|
1103
|
+
});
|
|
1104
|
+
return res.end();
|
|
1105
|
+
}
|
|
1106
|
+
|
|
1107
|
+
// Serve Web UI (no-cache: always serve the latest version after file edits)
|
|
1108
|
+
if (method === 'GET' && (pathname === '/' || pathname === '/ui')) {
|
|
1109
|
+
if (fs.existsSync(uiHtmlPath)) {
|
|
1110
|
+
res.writeHead(200, {
|
|
1111
|
+
'Content-Type': 'text/html; charset=utf-8',
|
|
1112
|
+
'Cache-Control': 'no-cache, no-store, must-revalidate',
|
|
1113
|
+
'X-Frame-Options': 'DENY',
|
|
1114
|
+
'X-Content-Type-Options': 'nosniff'
|
|
1115
|
+
});
|
|
1116
|
+
return res.end(fs.readFileSync(uiHtmlPath, 'utf8'));
|
|
1117
|
+
}
|
|
1118
|
+
}
|
|
1119
|
+
|
|
1120
|
+
// Health check
|
|
1121
|
+
if (method === 'GET' && pathname === '/health') {
|
|
1122
|
+
const { profileKey, profile } = getFirstActiveProfile(TARGETS, req);
|
|
1123
|
+
// ?challenge=<nonce> lets the CLI tell this gateway from a process that replays a /health body.
|
|
1124
|
+
const challenge = parsedUrl.searchParams.get('challenge');
|
|
1125
|
+
return sendJson(res, 200, {
|
|
1126
|
+
status: 'ok',
|
|
1127
|
+
proxy: 'llm-switcher',
|
|
1128
|
+
...(challenge ? {
|
|
1129
|
+
pid: process.pid,
|
|
1130
|
+
proof: identityProof(challenge, { role: 'gateway', port: PORT, pid: process.pid }, ADMIN_TOKEN.toString())
|
|
1131
|
+
} : {}),
|
|
1132
|
+
port: PORT,
|
|
1133
|
+
configLoaded: Boolean(loadConfig()),
|
|
1134
|
+
activeProfile: profileKey || '(none)',
|
|
1135
|
+
activeProfiles: loadConfig() ? getActiveMap(loadConfig()) : {},
|
|
1136
|
+
mode: profile?.mode || 'hybrid',
|
|
1137
|
+
inFormat: profile?.inFormat || 'auto',
|
|
1138
|
+
outFormat: profile ? resolveOutFormat(profile, '') : 'none',
|
|
1139
|
+
upstream: profile?.baseURL || '(none)'
|
|
1140
|
+
});
|
|
1141
|
+
}
|
|
1142
|
+
|
|
1143
|
+
// OpenAI-style model list (Codex / OpenAI SDK discovery).
|
|
1144
|
+
// Served IDs come from profile.publicModels when set (official-facing names the
|
|
1145
|
+
// CLI already knows, e.g. gpt-5.6-sol) so the client never observes the internal
|
|
1146
|
+
// upstream IDs or slot aliases. Slot aliases (main, review, ...) are never
|
|
1147
|
+
// advertised: Codex sends them in-request and mapModel resolves them server-side.
|
|
1148
|
+
// Without publicModels, fall back to the deduplicated mapped upstream IDs.
|
|
1149
|
+
if (method === 'GET' && (pathname === '/v1/models' || pathname === '/models')) {
|
|
1150
|
+
const { ids, windows } = servedModels(req);
|
|
1151
|
+
const created = Math.floor(Date.now() / 1000);
|
|
1152
|
+
return sendJson(res, 200, {
|
|
1153
|
+
object: 'list',
|
|
1154
|
+
data: ids.map(id => ({ id, object: 'model', created, owned_by: 'system' })),
|
|
1155
|
+
models: ids.map(id => ({ ...codexModelEntry(id, windows.get(id)), id, object: 'model', created, owned_by: 'system' }))
|
|
1156
|
+
});
|
|
1157
|
+
}
|
|
1158
|
+
|
|
1159
|
+
// Individual model metadata: GET /v1/models/{id}
|
|
1160
|
+
if (method === 'GET' && (pathname.startsWith('/v1/models/') || pathname.startsWith('/models/'))) {
|
|
1161
|
+
const modelId = decodeURIComponent(pathname.replace(/^\/(v1\/)?models\//, ''));
|
|
1162
|
+
if (modelId) {
|
|
1163
|
+
const { windows } = servedModels(req);
|
|
1164
|
+
return sendJson(res, 200, { ...codexModelEntry(modelId, windows.get(modelId)), id: modelId, object: 'model', created: Math.floor(Date.now() / 1000), owned_by: 'system' });
|
|
1165
|
+
}
|
|
1166
|
+
}
|
|
1167
|
+
|
|
1168
|
+
if (pathname.startsWith('/api/')) {
|
|
1169
|
+
if (!isAdminRequest(req)) {
|
|
1170
|
+
req.resume();
|
|
1171
|
+
return sendJson(res, 401, { error: 'Unauthorized: send the x-llm-switcher-token header. Open the dashboard with `switch ui`.' });
|
|
1172
|
+
}
|
|
1173
|
+
return routeApi(req, res, method, pathname);
|
|
1174
|
+
}
|
|
1175
|
+
|
|
1176
|
+
// Token count estimation endpoint
|
|
1177
|
+
if (method === 'POST' && (pathname === '/v1/messages/count_tokens' || pathname === '/messages/count_tokens')) {
|
|
1178
|
+
const buf = await readBody(req, MAX_BODY_SIZE);
|
|
1179
|
+
return handleCountTokens(req, res, buf);
|
|
1180
|
+
}
|
|
1181
|
+
|
|
1182
|
+
// Client endpoints, one per input protocol (auto-detected by path).
|
|
1183
|
+
let clientFormat = null;
|
|
1184
|
+
const vmatch = method === 'POST' ? pathname.match(VERTEX_ROUTE) : null;
|
|
1185
|
+
|
|
1186
|
+
// Codex CLI sends GET /v1/responses (and /v1/responses/{id}) to fetch model
|
|
1187
|
+
// metadata and retrieve previous responses. The gateway is stateless, so
|
|
1188
|
+
// return a synthetic stub that satisfies the SDK's metadata lookup without
|
|
1189
|
+
// erroring out.
|
|
1190
|
+
if (method === 'GET' && (pathname === '/v1/responses' || pathname === '/responses' || pathname.startsWith('/v1/responses/') || pathname.startsWith('/responses/'))) {
|
|
1191
|
+
req.resume();
|
|
1192
|
+
if (pathname === '/v1/responses' || pathname === '/responses') {
|
|
1193
|
+
// Model metadata / list — return an empty list
|
|
1194
|
+
return sendJson(res, 200, { object: 'list', data: [] });
|
|
1195
|
+
}
|
|
1196
|
+
// GET /v1/responses/{id} — response retrieval; stateless gateway has no
|
|
1197
|
+
// persisted responses so return 404 in OpenAI's error shape.
|
|
1198
|
+
return sendJson(res, 404, {
|
|
1199
|
+
error: { message: 'Response not found. This gateway is stateless and does not persist responses.', type: 'not_found_error', code: '404' }
|
|
1200
|
+
});
|
|
1201
|
+
}
|
|
1202
|
+
|
|
1203
|
+
if (method === 'POST') {
|
|
1204
|
+
if (pathname === '/v1/messages' || pathname === '/messages') clientFormat = 'anthropic';
|
|
1205
|
+
else if (pathname === '/v1/chat/completions' || pathname === '/chat/completions') clientFormat = 'openai-chat';
|
|
1206
|
+
else if (pathname === '/v1/responses' || pathname === '/responses') clientFormat = 'responses';
|
|
1207
|
+
else if (vmatch) clientFormat = 'vertex';
|
|
1208
|
+
}
|
|
1209
|
+
if (clientFormat) {
|
|
1210
|
+
let buf;
|
|
1211
|
+
try {
|
|
1212
|
+
buf = await readBody(req, MAX_BODY_SIZE);
|
|
1213
|
+
} catch (err) {
|
|
1214
|
+
return sendClientError(res, clientFormat, err.status || 400, err.message);
|
|
1215
|
+
}
|
|
1216
|
+
const opts = vmatch ? { vertexModel: decodeURIComponent(vmatch[1]), vertexStream: vmatch[2] === 'streamGenerateContent' } : {};
|
|
1217
|
+
return handleConvert(clientFormat, req, res, buf, opts);
|
|
1218
|
+
}
|
|
1219
|
+
|
|
1220
|
+
req.resume();
|
|
1221
|
+
return sendJson(res, 404, { error: { message: `Not found: ${method} ${pathname}` } });
|
|
1222
|
+
}
|
|
1223
|
+
|
|
1224
|
+
// The names /v1/models serves, and the window of each. Official names when the profile publishes
|
|
1225
|
+
// them, otherwise the mapped upstream IDs; each window follows model1M of its slot.
|
|
1226
|
+
function servedModels(req) {
|
|
1227
|
+
const { profile } = getFirstActiveProfile(['responses', 'openai-chat', 'anthropic', 'vertex'], req);
|
|
1228
|
+
if (Array.isArray(profile?.publicModels) && profile.publicModels.length) {
|
|
1229
|
+
return { ids: [...new Set(profile.publicModels.filter(Boolean))], windows: publicModelWindows(profile) };
|
|
1230
|
+
}
|
|
1231
|
+
const windows = smallestWindows(Object.entries(profile?.defaultModels || {}).map(([slot, id]) => [id, model1MForSlot(profile, slot)]));
|
|
1232
|
+
return { ids: [...windows.keys()], windows };
|
|
1233
|
+
}
|
|
1234
|
+
|
|
1235
|
+
async function routeApi(req, res, method, pathname) {
|
|
1236
|
+
// GET /api/status
|
|
1237
|
+
if (method === 'GET' && pathname === '/api/status') {
|
|
1238
|
+
const cfg = requireConfig(res);
|
|
1239
|
+
if (!cfg) return;
|
|
1240
|
+
const activeProfiles = getActiveMap(cfg);
|
|
1241
|
+
return sendJson(res, 200, {
|
|
1242
|
+
port: PORT,
|
|
1243
|
+
activeProfile: cfg.activeProfile || null,
|
|
1244
|
+
activeProfiles,
|
|
1245
|
+
revision: configRevision(cfg),
|
|
1246
|
+
claude1MTiers: computeLaunchState(cfg, PORT).claude1MTiers,
|
|
1247
|
+
...readLaunchFlags(),
|
|
1248
|
+
claudeBaseURL: activeProfiles.anthropic ? `http://127.0.0.1:${PORT} (injected via launcher)` : '(none / official)',
|
|
1249
|
+
config: redactConfig(cfg)
|
|
1250
|
+
});
|
|
1251
|
+
}
|
|
1252
|
+
|
|
1253
|
+
// GET /api/logs (Live Request/Response Inspector)
|
|
1254
|
+
if (method === 'GET' && pathname === '/api/logs') {
|
|
1255
|
+
return sendJson(res, 200, { logs: requestLogs.slice().reverse() });
|
|
1256
|
+
}
|
|
1257
|
+
|
|
1258
|
+
if (method !== 'POST') {
|
|
1259
|
+
req.resume();
|
|
1260
|
+
return sendJson(res, 404, { error: `Not found: ${method} ${pathname}` });
|
|
1261
|
+
}
|
|
1262
|
+
|
|
1263
|
+
let body;
|
|
1264
|
+
try {
|
|
1265
|
+
body = await readJsonBody(req);
|
|
1266
|
+
} catch (err) {
|
|
1267
|
+
return sendJson(res, err.status || 400, { error: err.message });
|
|
1268
|
+
}
|
|
1269
|
+
|
|
1270
|
+
// POST /api/logs/clear
|
|
1271
|
+
if (pathname === '/api/logs/clear') {
|
|
1272
|
+
requestLogs.length = 0;
|
|
1273
|
+
return sendJson(res, 200, { success: true });
|
|
1274
|
+
}
|
|
1275
|
+
|
|
1276
|
+
if (CONFIG_CHANGES.has(pathname)) return serialized(() => routeConfigApi(res, method, pathname, body));
|
|
1277
|
+
return routeConfigApi(res, method, pathname, body);
|
|
1278
|
+
}
|
|
1279
|
+
|
|
1280
|
+
async function routeConfigApi(res, method, pathname, body) {
|
|
1281
|
+
const loaded = requireConfig(res);
|
|
1282
|
+
if (!loaded) return;
|
|
1283
|
+
// A copy: a refused change must never reach the cached config that other requests read.
|
|
1284
|
+
const cfg = structuredClone(loaded);
|
|
1285
|
+
const baseRevision = configRevision(loaded);
|
|
1286
|
+
if (typeof body.revision === 'string' && body.revision !== baseRevision) {
|
|
1287
|
+
return sendJson(res, 409, { error: 'config.json changed since this page loaded it. The page reloads it now; check the change and try again.', revision: baseRevision });
|
|
1288
|
+
}
|
|
1289
|
+
|
|
1290
|
+
// POST /api/switch { target?, profile? | null, deactivate? }
|
|
1291
|
+
if (pathname === '/api/switch') {
|
|
1292
|
+
let err = null;
|
|
1293
|
+
if (body.deactivate) {
|
|
1294
|
+
deactivateProfile(cfg, body.deactivate);
|
|
1295
|
+
} else if (body.target) {
|
|
1296
|
+
err = setTargetProfile(cfg, body.target, body.profile || null);
|
|
1297
|
+
} else if (body.profile) {
|
|
1298
|
+
err = activateProfile(cfg, body.profile);
|
|
1299
|
+
} else {
|
|
1300
|
+
deactivateAll(cfg);
|
|
1301
|
+
}
|
|
1302
|
+
if (err) return sendJson(res, 400, { error: err });
|
|
1303
|
+
const applied = await commit(cfg, baseRevision);
|
|
1304
|
+
return sendJson(res, applied.status || (applied.success === false ? 502 : 200), { success: true, activeProfile: cfg.activeProfile, activeProfiles: cfg.activeProfiles, ...applied });
|
|
1305
|
+
}
|
|
1306
|
+
|
|
1307
|
+
// POST /api/toggle { target?, enabled }
|
|
1308
|
+
if (pathname === '/api/toggle') {
|
|
1309
|
+
const map = getActiveMap(cfg);
|
|
1310
|
+
let err = null;
|
|
1311
|
+
if (body.target) {
|
|
1312
|
+
if (!TARGETS.includes(body.target)) return sendJson(res, 400, { error: `Unknown target "${body.target}"` });
|
|
1313
|
+
let key = null;
|
|
1314
|
+
if (body.enabled) {
|
|
1315
|
+
const candidates = [map[body.target], cfg.activeProfile, ...Object.keys(cfg.profiles)];
|
|
1316
|
+
key = candidates.find(k => hasProfile(cfg, k) && profileAcceptsTarget(cfg.profiles[k], body.target)) || null;
|
|
1317
|
+
if (!key) return sendJson(res, 400, { error: `No profile accepts target "${body.target}"` });
|
|
1318
|
+
}
|
|
1319
|
+
err = setTargetProfile(cfg, body.target, key);
|
|
1320
|
+
} else if (body.enabled) {
|
|
1321
|
+
const key = hasProfile(cfg, cfg.activeProfile) ? cfg.activeProfile : Object.keys(cfg.profiles)[0];
|
|
1322
|
+
if (!key) return sendJson(res, 400, { error: 'No profiles configured' });
|
|
1323
|
+
err = activateProfile(cfg, key);
|
|
1324
|
+
} else {
|
|
1325
|
+
deactivateAll(cfg);
|
|
1326
|
+
}
|
|
1327
|
+
if (err) return sendJson(res, 400, { error: err });
|
|
1328
|
+
const applied = await commit(cfg, baseRevision);
|
|
1329
|
+
return sendJson(res, applied.status || (applied.success === false ? 502 : 200), { success: true, enabled: Boolean(body.enabled), activeProfiles: cfg.activeProfiles, ...applied });
|
|
1330
|
+
}
|
|
1331
|
+
|
|
1332
|
+
// POST /api/save-profile { key, profile }
|
|
1333
|
+
if (pathname === '/api/save-profile') {
|
|
1334
|
+
const { key, profile } = body;
|
|
1335
|
+
if (!isValidProfileKey(key)) {
|
|
1336
|
+
return sendJson(res, 400, { error: 'Invalid profile key: use 1-64 chars of letters, digits, ".", "_" or "-"' });
|
|
1337
|
+
}
|
|
1338
|
+
const invalid = validateProfileInput(profile);
|
|
1339
|
+
if (invalid) return sendJson(res, 400, { error: invalid });
|
|
1340
|
+
|
|
1341
|
+
const existing = hasProfile(cfg, key) ? cfg.profiles[key] : {};
|
|
1342
|
+
// Merge so unmanaged UI fields are not lost (e.g. `endpoints`).
|
|
1343
|
+
const merged = { ...existing, ...profile };
|
|
1344
|
+
// A payload without apiKey, or with the mask, keeps the stored key, but only for the destination
|
|
1345
|
+
// it was stored with: otherwise one request could send the real key to any host.
|
|
1346
|
+
const keepsKey = !Object.hasOwn(profile, 'apiKey') || profile.apiKey === MASKED_KEY;
|
|
1347
|
+
if (keepsKey && existing.apiKey && !sameDestination(merged, existing)) {
|
|
1348
|
+
return sendJson(res, 400, { error: 'The stored API key is sent only to the baseURL and endpoints it was saved with. Enter the key again to use a new URL.' });
|
|
1349
|
+
}
|
|
1350
|
+
merged.apiKey = keepsKey ? (existing.apiKey || '') : String(profile.apiKey || '');
|
|
1351
|
+
for (const k of ['outFormat', 'optimizerURL', 'thinkingMode']) {
|
|
1352
|
+
if (Object.hasOwn(profile, k) && !profile[k]) delete merged[k];
|
|
1353
|
+
}
|
|
1354
|
+
cfg.profiles[key] = merged;
|
|
1355
|
+
|
|
1356
|
+
// A target assigned to this profile whose new inFormat no longer supports it -> unassign that target.
|
|
1357
|
+
const map = getActiveMap(cfg);
|
|
1358
|
+
cfg.activeProfiles = map;
|
|
1359
|
+
let unassigned = false;
|
|
1360
|
+
for (const t of TARGETS) {
|
|
1361
|
+
if (map[t] === key && !profileAcceptsTarget(merged, t)) {
|
|
1362
|
+
map[t] = null;
|
|
1363
|
+
unassigned = true;
|
|
1364
|
+
}
|
|
1365
|
+
}
|
|
1366
|
+
|
|
1367
|
+
// Profile is active (or was just unassigned from a target) -> refresh 1M flags / env files.
|
|
1368
|
+
const applied = isProfileActive(cfg, key) || unassigned ? await commit(cfg, baseRevision) : (saveConfig(cfg), { revision: configRevision(cfg) });
|
|
1369
|
+
return sendJson(res, applied.status || (applied.success === false ? 502 : 200), { success: true, ...applied });
|
|
1370
|
+
}
|
|
1371
|
+
|
|
1372
|
+
// POST /api/delete-profile { key }
|
|
1373
|
+
if (pathname === '/api/delete-profile') {
|
|
1374
|
+
const err = deleteProfile(cfg, body.key);
|
|
1375
|
+
if (err) return sendJson(res, 404, { error: err });
|
|
1376
|
+
const applied = await commit(cfg, baseRevision);
|
|
1377
|
+
return sendJson(res, applied.status || (applied.success === false ? 502 : 200), { success: true, ...applied });
|
|
1378
|
+
}
|
|
1379
|
+
|
|
1380
|
+
// POST /api/blindfold/sync — the CLI asks the owner to bring the interceptor in line with config.json.
|
|
1381
|
+
if (pathname === '/api/blindfold/sync') {
|
|
1382
|
+
const r = await reconcile(cfg);
|
|
1383
|
+
return sendJson(res, r.ok ? 200 : 500, r);
|
|
1384
|
+
}
|
|
1385
|
+
|
|
1386
|
+
// POST /api/test-upstream
|
|
1387
|
+
if (pathname === '/api/test-upstream') {
|
|
1388
|
+
try {
|
|
1389
|
+
const r = await testUpstream(body, cfg);
|
|
1390
|
+
return sendJson(res, r.status, r.json);
|
|
1391
|
+
} catch (err) {
|
|
1392
|
+
return sendJson(res, 200, { ok: false, error: err.name === 'TimeoutError' ? 'Timed out waiting for upstream' : (err.cause?.message || err.message) });
|
|
1393
|
+
}
|
|
1394
|
+
}
|
|
1395
|
+
|
|
1396
|
+
// POST /api/fetch-models
|
|
1397
|
+
if (pathname === '/api/fetch-models') {
|
|
1398
|
+
try {
|
|
1399
|
+
const r = await fetchModels(body, cfg);
|
|
1400
|
+
return sendJson(res, r.status, r.json);
|
|
1401
|
+
} catch (err) {
|
|
1402
|
+
return sendJson(res, 200, { ok: false, error: err.name === 'TimeoutError' ? 'Timed out waiting for upstream' : (err.cause?.message || err.message) });
|
|
1403
|
+
}
|
|
1404
|
+
}
|
|
1405
|
+
|
|
1406
|
+
return sendJson(res, 404, { error: `Not found: ${method} ${pathname}` });
|
|
1407
|
+
}
|
|
1408
|
+
|
|
1409
|
+
const server = http.createServer((req, res) => {
|
|
1410
|
+
route(req, res).catch(err => {
|
|
1411
|
+
console.error('[llm-switcher] Unhandled request error:', err);
|
|
1412
|
+
if (!res.headersSent) {
|
|
1413
|
+
sendJson(res, err.status || 500, { error: { message: `Gateway error: ${err.message}` } });
|
|
1414
|
+
} else {
|
|
1415
|
+
try { res.end(); } catch {}
|
|
1416
|
+
}
|
|
1417
|
+
});
|
|
1418
|
+
});
|
|
1419
|
+
|
|
1420
|
+
function encodeWsFrame(data, opcode = 1) {
|
|
1421
|
+
const payload = Buffer.isBuffer(data) ? data : Buffer.from(data, 'utf8');
|
|
1422
|
+
const len = payload.length;
|
|
1423
|
+
let header;
|
|
1424
|
+
if (len <= 125) {
|
|
1425
|
+
header = Buffer.from([0x80 | (opcode & 0x0f), len]);
|
|
1426
|
+
} else if (len <= 65535) {
|
|
1427
|
+
header = Buffer.alloc(4);
|
|
1428
|
+
header[0] = 0x80 | (opcode & 0x0f);
|
|
1429
|
+
header[1] = 126;
|
|
1430
|
+
header.writeUInt16BE(len, 2);
|
|
1431
|
+
} else {
|
|
1432
|
+
header = Buffer.alloc(10);
|
|
1433
|
+
header[0] = 0x80 | (opcode & 0x0f);
|
|
1434
|
+
header[1] = 127;
|
|
1435
|
+
header.writeBigUInt64BE(BigInt(len), 2);
|
|
1436
|
+
}
|
|
1437
|
+
return Buffer.concat([header, payload]);
|
|
1438
|
+
}
|
|
1439
|
+
|
|
1440
|
+
// Map an upstream HTTP status to a Responses-API error code so Codex can tell a
|
|
1441
|
+
// retryable rate-limit from a fatal request error.
|
|
1442
|
+
function responsesErrorCode(status) {
|
|
1443
|
+
if (status === 429) return 'rate_limit_exceeded';
|
|
1444
|
+
if (status === 401) return 'authentication_error';
|
|
1445
|
+
if (status === 403) return 'permission_denied';
|
|
1446
|
+
if (status === 404) return 'not_found_error';
|
|
1447
|
+
if (status === 400) return 'invalid_request_error';
|
|
1448
|
+
return 'server_error';
|
|
1449
|
+
}
|
|
1450
|
+
|
|
1451
|
+
// A WS frame carries the event payload alone. The half records the same event in the SSE text of
|
|
1452
|
+
// the HTTP /v1/responses path, so one converter keeps one shape in intact whatever the transport.
|
|
1453
|
+
// A failure to copy is dropped: the tap must never come between the renderer and the socket.
|
|
1454
|
+
function tapWsEvent(tap, event, data) {
|
|
1455
|
+
if (!tap) return;
|
|
1456
|
+
try {
|
|
1457
|
+
tap.push(`event: ${event}\ndata: ${JSON.stringify(data)}\n\n`);
|
|
1458
|
+
} catch {}
|
|
1459
|
+
}
|
|
1460
|
+
|
|
1461
|
+
// Terminal failure for the WS (responses-ws) transport: Codex ends a turn only on
|
|
1462
|
+
// response.completed / response.failed, so a bare {type:'error'} frame leaves the turn
|
|
1463
|
+
// hanging. Emit the full created -> in_progress -> failed sequence instead.
|
|
1464
|
+
function sendWsFailed(socket, model, message, status = 500, tap = null) {
|
|
1465
|
+
if (!socket.writable) return;
|
|
1466
|
+
const renderer = createResponsesStream((e, d) => {
|
|
1467
|
+
tapWsEvent(tap, e, d);
|
|
1468
|
+
if (socket.writable) socket.write(encodeWsFrame(JSON.stringify(d)));
|
|
1469
|
+
}, model || 'main');
|
|
1470
|
+
renderer.start();
|
|
1471
|
+
renderer.error(message, responsesErrorCode(status));
|
|
1472
|
+
}
|
|
1473
|
+
|
|
1474
|
+
// 9Router forwards OpenAI-format tools to Gemini/Vertex for ag/* models, which accept only
|
|
1475
|
+
// a strict Schema subset: bare "object" strings, $ref/$defs, anyOf-null unions and
|
|
1476
|
+
// additionalProperties all come back as HTTP 400 INVALID_ARGUMENT. Rewrite tool parameters
|
|
1477
|
+
// into that subset before sending upstream. Non-ag targets keep the OpenAI superset.
|
|
1478
|
+
function geminiSafeTools(tools) {
|
|
1479
|
+
return tools.map(t => {
|
|
1480
|
+
if (!t || t.type !== 'function' || !t.function) return t;
|
|
1481
|
+
return { ...t, function: { ...t.function, parameters: toGeminiSchema(t.function.parameters || { type: 'object', properties: {} }) } };
|
|
1482
|
+
});
|
|
1483
|
+
}
|
|
1484
|
+
|
|
1485
|
+
// One turn of the Codex WS transport. The socket loop runs turns one at a time and owns `ac`.
|
|
1486
|
+
async function handleWsResponseCreate(socket, payload, req, ac) {
|
|
1487
|
+
const clientFormat = 'responses';
|
|
1488
|
+
const { profileKey, profile, error: profileError } = getActiveProfile(clientFormat, req);
|
|
1489
|
+
if (!loadConfig()) {
|
|
1490
|
+
sendWsFailed(socket, payload?.model || 'main', `LLM Switcher config not loaded (${configPath}): ${getConfigLoadError()?.message || 'missing file'}`, 500);
|
|
1491
|
+
return;
|
|
1492
|
+
}
|
|
1493
|
+
if (!profile) {
|
|
1494
|
+
sendWsFailed(socket, payload?.model || 'main', profileError || 'Proxy is currently OFF for responses.', 503);
|
|
1495
|
+
return;
|
|
1496
|
+
}
|
|
1497
|
+
|
|
1498
|
+
let ir;
|
|
1499
|
+
try {
|
|
1500
|
+
ir = parseToIR('responses', payload);
|
|
1501
|
+
} catch (e) {
|
|
1502
|
+
sendWsFailed(socket, payload?.model || 'main', `Cannot parse responses request: ${e.message}`, 400);
|
|
1503
|
+
return;
|
|
1504
|
+
}
|
|
1505
|
+
ir.stream = true;
|
|
1506
|
+
|
|
1507
|
+
const reqStartTime = Date.now();
|
|
1508
|
+
const requestPreview = previewOf(ir);
|
|
1509
|
+
const requestedModel = ir.model || payload.model || '';
|
|
1510
|
+
const mappedModel = mapModel(requestedModel, profile, clientFormat);
|
|
1511
|
+
const outFormat = resolveOutFormat(profile, mappedModel);
|
|
1512
|
+
|
|
1513
|
+
console.log(`[llm-switcher:ws] ${clientFormat} -> ${outFormat} "${requestedModel}" -> "${mappedModel}" [${profile.name || profileKey}]`);
|
|
1514
|
+
const logBase = { clientFormat: 'responses-ws', outFormat, profile: profileKey, model: mappedModel, stream: true, requestPreview };
|
|
1515
|
+
const log = (extra) => logInspection({ ...logBase, duration: Date.now() - reqStartTime, tokens: { prompt: 0, completion: 0 }, ...extra });
|
|
1516
|
+
|
|
1517
|
+
// Contract lab: the Codex WS transport is sampled like any other request, and the events of
|
|
1518
|
+
// this turn are copied for the upload that follows it.
|
|
1519
|
+
const traceId = contractLab.traceFor(mappedModel);
|
|
1520
|
+
const halfTap = traceId ? createHalfTap() : null;
|
|
1521
|
+
const halfRequest = traceId ? capJson(payload) : '';
|
|
1522
|
+
|
|
1523
|
+
let answered = false; // set only after a complete answer; the half upload depends on it
|
|
1524
|
+
try {
|
|
1525
|
+
const { url, headers, upBody } = buildUpstreamRequest(profile, outFormat, ir, mappedModel, req);
|
|
1526
|
+
if (traceId) headers['x-intact-trace'] = traceId;
|
|
1527
|
+
debugLog(`[${profileKey}:ws] ${clientFormat} -> ${outFormat} ${url} ::`, JSON.stringify(upBody).slice(0, 300));
|
|
1528
|
+
|
|
1529
|
+
let upstreamRes;
|
|
1530
|
+
try {
|
|
1531
|
+
upstreamRes = await fetch(url, { method: 'POST', headers, body: JSON.stringify(upBody), signal: ac.signal });
|
|
1532
|
+
} catch (fetchErr) {
|
|
1533
|
+
if (ac.signal.aborted) return log({ status: 499, error: 'client disconnected' });
|
|
1534
|
+
console.error(`[${profileKey}:ws] Network error:`, fetchErr.message);
|
|
1535
|
+
sendWsFailed(socket, mappedModel, `Failed to connect to upstream: ${fetchErr.cause?.message || fetchErr.message}`, 502, halfTap);
|
|
1536
|
+
return log({ status: 502, error: fetchErr.message });
|
|
1537
|
+
}
|
|
1538
|
+
|
|
1539
|
+
if (!upstreamRes.ok) {
|
|
1540
|
+
const errText = await upstreamRes.text().catch(() => '');
|
|
1541
|
+
console.error(`[${profileKey}:ws] Error HTTP ${upstreamRes.status}:`, errText.slice(0, 500));
|
|
1542
|
+
sendWsFailed(socket, mappedModel, extractUpstreamMessage(errText) || `Upstream HTTP ${upstreamRes.status}`, upstreamRes.status, halfTap);
|
|
1543
|
+
return log({ status: upstreamRes.status, error: errText.slice(0, 300) });
|
|
1544
|
+
}
|
|
1545
|
+
|
|
1546
|
+
const normalize = createUpstreamNormalizer(outFormat);
|
|
1547
|
+
const col = createCollector();
|
|
1548
|
+
|
|
1549
|
+
const renderer = createResponsesStream((e, d) => {
|
|
1550
|
+
tapWsEvent(halfTap, e, d);
|
|
1551
|
+
if (socket.writable) {
|
|
1552
|
+
socket.write(encodeWsFrame(JSON.stringify(d)));
|
|
1553
|
+
}
|
|
1554
|
+
}, requestedModel || mappedModel, { toolMeta: ir.toolMeta });
|
|
1555
|
+
|
|
1556
|
+
renderer.start();
|
|
1557
|
+
const splitter = createThinkTagSplitter(t => renderer.think(t), t => renderer.text(t));
|
|
1558
|
+
const streamError = await pumpStream(upstreamRes, { normalize, col, renderer, splitter, signal: ac.signal, sink: socket, tag: `${profileKey}:ws` });
|
|
1559
|
+
|
|
1560
|
+
if (ac.signal.aborted) {
|
|
1561
|
+
return log({ status: 499, error: 'client disconnected mid-stream', responsePreview: col.text.join('').slice(0, 300) });
|
|
1562
|
+
}
|
|
1563
|
+
splitter.flush();
|
|
1564
|
+
const completion = col.completion();
|
|
1565
|
+
if (streamError) {
|
|
1566
|
+
renderer.error(streamError);
|
|
1567
|
+
} else {
|
|
1568
|
+
renderer.finish(col.finish, { completion, prompt: col.prompt, cached: col.cached, reasoning: col.reasoning, hasTools: col.tools.size > 0 });
|
|
1569
|
+
answered = true;
|
|
1570
|
+
}
|
|
1571
|
+
log({
|
|
1572
|
+
status: streamError ? 502 : 200, stream: true,
|
|
1573
|
+
...(streamError ? { error: String(streamError).slice(0, 300) } : {}),
|
|
1574
|
+
tokens: { prompt: col.prompt, completion },
|
|
1575
|
+
thinkingChars: col.think.join('').length,
|
|
1576
|
+
responsePreview: col.text.join('').slice(0, 300)
|
|
1577
|
+
});
|
|
1578
|
+
} catch (err) {
|
|
1579
|
+
if (ac.signal.aborted) return log({ status: 499, error: 'aborted' });
|
|
1580
|
+
console.error(`[${profileKey}:ws] Error:`, err);
|
|
1581
|
+
sendWsFailed(socket, payload?.model || 'main', err.message, 500, halfTap);
|
|
1582
|
+
log({ status: 500, error: err.message });
|
|
1583
|
+
} finally {
|
|
1584
|
+
// The turn is over; the half is queued and posted on a later turn. A cancelled turn has no
|
|
1585
|
+
// complete answer: uploading it would diff as a loss the converter never made.
|
|
1586
|
+
finishHalf(contractLab, traceId, ac.signal.aborted || !answered, {
|
|
1587
|
+
toolRequest: halfRequest,
|
|
1588
|
+
toolResponse: halfTap?.text() || '',
|
|
1589
|
+
toolVersion: toolVersionFromUA(req.headers['user-agent']),
|
|
1590
|
+
inFormat: clientFormat,
|
|
1591
|
+
outFormat
|
|
1592
|
+
});
|
|
1593
|
+
}
|
|
1594
|
+
}
|
|
1595
|
+
|
|
1596
|
+
server.on('upgrade', (req, socket) => {
|
|
1597
|
+
// Node emits 'upgrade' instead of 'request', so route() never runs here and the
|
|
1598
|
+
// Host/Origin guard has to be applied again. Browsers do not apply same-origin to
|
|
1599
|
+
// WebSocket, so without this any visited page could open ws://127.0.0.1/v1/responses
|
|
1600
|
+
// and spend the profile's API key. An absent Origin stays allowed on purpose: Codex
|
|
1601
|
+
// sends none, and the blindfold interceptor deletes it.
|
|
1602
|
+
const guardError = checkRequestOrigin(req);
|
|
1603
|
+
if (guardError) {
|
|
1604
|
+
socket.write(
|
|
1605
|
+
'HTTP/1.1 403 Forbidden\r\n' +
|
|
1606
|
+
'Connection: close\r\n' +
|
|
1607
|
+
'Content-Type: application/json\r\n\r\n' +
|
|
1608
|
+
`{"error":{"message":${JSON.stringify(guardError)}}}\r\n`
|
|
1609
|
+
);
|
|
1610
|
+
socket.destroy();
|
|
1611
|
+
return;
|
|
1612
|
+
}
|
|
1613
|
+
|
|
1614
|
+
const p = new URL(req.url || '/', 'http://127.0.0.1').pathname;
|
|
1615
|
+
if (p !== '/v1/responses' && p !== '/responses') {
|
|
1616
|
+
socket.write(
|
|
1617
|
+
'HTTP/1.1 404 Not Found\r\n' +
|
|
1618
|
+
'Connection: close\r\n' +
|
|
1619
|
+
'Content-Type: application/json\r\n\r\n' +
|
|
1620
|
+
`{"error":{"message":"Not found: ${req.method} ${p}"}}\r\n`
|
|
1621
|
+
);
|
|
1622
|
+
socket.destroy();
|
|
1623
|
+
return;
|
|
1624
|
+
}
|
|
1625
|
+
|
|
1626
|
+
const key = req.headers['sec-websocket-key'];
|
|
1627
|
+
if (!key) {
|
|
1628
|
+
socket.destroy();
|
|
1629
|
+
return;
|
|
1630
|
+
}
|
|
1631
|
+
const accept = crypto.createHash('sha1').update(key + '258EAFA5-E914-47DA-95CA-C5AB0DC85B11').digest('base64');
|
|
1632
|
+
|
|
1633
|
+
// The official name or the slot, never the upstream ID. The value goes into a raw header line.
|
|
1634
|
+
const { profile } = getActiveProfile('responses', req);
|
|
1635
|
+
const publicMain = profile ? codexPublicModel(profile, 'main') : '';
|
|
1636
|
+
const activeModel = isSafeModelName(publicMain) ? publicMain : 'main';
|
|
1637
|
+
|
|
1638
|
+
socket.write(
|
|
1639
|
+
'HTTP/1.1 101 Switching Protocols\r\n' +
|
|
1640
|
+
'Upgrade: websocket\r\n' +
|
|
1641
|
+
'Connection: Upgrade\r\n' +
|
|
1642
|
+
`Sec-WebSocket-Accept: ${accept}\r\n` +
|
|
1643
|
+
`OpenAI-Model: ${activeModel}\r\n` +
|
|
1644
|
+
'x-reasoning-included: true\r\n' +
|
|
1645
|
+
'x-codex-turn-state: ready\r\n\r\n'
|
|
1646
|
+
);
|
|
1647
|
+
|
|
1648
|
+
const read = createFrameReader({ maxMessage: MAX_BODY_SIZE });
|
|
1649
|
+
// Turns run one at a time: two at once interleave their events on one socket. Every turn,
|
|
1650
|
+
// queued or running, holds a controller here, so a cancel or a close stops all of them.
|
|
1651
|
+
const turns = new Set();
|
|
1652
|
+
let turnChain = Promise.resolve();
|
|
1653
|
+
const abortTurns = () => { for (const ac of turns) ac.abort(); };
|
|
1654
|
+
|
|
1655
|
+
const handleMessage = (msg) => {
|
|
1656
|
+
if (msg.type === 'response.create') {
|
|
1657
|
+
const ac = new AbortController();
|
|
1658
|
+
turns.add(ac);
|
|
1659
|
+
turnChain = turnChain
|
|
1660
|
+
.then(() => (ac.signal.aborted ? null : handleWsResponseCreate(socket, msg, req, ac)))
|
|
1661
|
+
.catch(err => {
|
|
1662
|
+
// A throw before the handler's own try: Codex ends a turn only on response.failed.
|
|
1663
|
+
console.error('[llm-switcher:ws] Unhandled turn error:', err);
|
|
1664
|
+
if (!ac.signal.aborted) sendWsFailed(socket, msg.model || 'main', err.message, 500);
|
|
1665
|
+
})
|
|
1666
|
+
.finally(() => turns.delete(ac));
|
|
1667
|
+
} else if (msg.type === 'response.cancel') {
|
|
1668
|
+
abortTurns();
|
|
1669
|
+
} else if (msg.type === 'session.update') {
|
|
1670
|
+
if (socket.writable) socket.write(encodeWsFrame(JSON.stringify({ type: 'session.updated', session: msg.session || {} })));
|
|
1671
|
+
} else if (msg.type === 'conversation.item.create') {
|
|
1672
|
+
if (socket.writable) socket.write(encodeWsFrame(JSON.stringify({ type: 'conversation.item.created', item: msg.item || {} })));
|
|
1673
|
+
}
|
|
1674
|
+
};
|
|
1675
|
+
|
|
1676
|
+
socket.on('data', (chunk) => {
|
|
1677
|
+
for (const f of read(chunk)) {
|
|
1678
|
+
if (f.type === 'error') {
|
|
1679
|
+
// 1009 = message too big. The reader has stopped, so the connection cannot continue.
|
|
1680
|
+
const code = Buffer.alloc(2);
|
|
1681
|
+
code.writeUInt16BE(/exceeds/.test(f.reason) ? 1009 : 1002);
|
|
1682
|
+
if (socket.writable) socket.end(encodeWsFrame(code, 8));
|
|
1683
|
+
abortTurns();
|
|
1684
|
+
return;
|
|
1685
|
+
}
|
|
1686
|
+
if (f.type === 'close') {
|
|
1687
|
+
abortTurns();
|
|
1688
|
+
if (socket.writable) socket.end(encodeWsFrame(Buffer.alloc(0), 8));
|
|
1689
|
+
return;
|
|
1690
|
+
}
|
|
1691
|
+
if (f.type === 'ping') {
|
|
1692
|
+
if (socket.writable) socket.write(encodeWsFrame(f.payload, 10));
|
|
1693
|
+
continue;
|
|
1694
|
+
}
|
|
1695
|
+
if (f.type !== 'text') continue;
|
|
1696
|
+
let msg;
|
|
1697
|
+
try {
|
|
1698
|
+
msg = JSON.parse(f.payload.toString('utf8'));
|
|
1699
|
+
} catch (e) {
|
|
1700
|
+
console.error('[llm-switcher:ws] Bad WS message JSON:', e.message);
|
|
1701
|
+
continue;
|
|
1702
|
+
}
|
|
1703
|
+
handleMessage(msg);
|
|
1704
|
+
}
|
|
1705
|
+
});
|
|
1706
|
+
|
|
1707
|
+
socket.on('close', abortTurns);
|
|
1708
|
+
// The upgraded socket allows half-open: a client FIN alone would not close it, and the turn
|
|
1709
|
+
// would keep spending upstream tokens until the next write failed.
|
|
1710
|
+
socket.on('end', () => {
|
|
1711
|
+
abortTurns();
|
|
1712
|
+
socket.end();
|
|
1713
|
+
});
|
|
1714
|
+
|
|
1715
|
+
socket.on('error', (err) => {
|
|
1716
|
+
debugLog('[llm-switcher:ws] Socket error:', err.message);
|
|
1717
|
+
abortTurns();
|
|
1718
|
+
});
|
|
1719
|
+
});
|
|
1720
|
+
|
|
1721
|
+
server.on('error', (err) => {
|
|
1722
|
+
if (err.code === 'EADDRINUSE') {
|
|
1723
|
+
console.error(`\n[llm-switcher:ERROR] Port ${PORT} is already in use by another process!`);
|
|
1724
|
+
console.error(`- Run 'switch status' to check, or 'switch off' to stop a running LLM Switcher.`);
|
|
1725
|
+
console.error(`- If another tool (e.g. headroom/rtk/proxy) is using port ${PORT}, change "port" in config.json or pass --port.`);
|
|
1726
|
+
process.exit(1);
|
|
1727
|
+
} else {
|
|
1728
|
+
console.error('[llm-switcher:ERROR]', err);
|
|
1729
|
+
}
|
|
1730
|
+
});
|
|
1731
|
+
|
|
1732
|
+
if (!loadConfig()) {
|
|
1733
|
+
console.warn(`[llm-switcher:WARN] Could not load ${configPath}: ${getConfigLoadError()?.message}. Copy config.example.json to config.json.`);
|
|
1734
|
+
}
|
|
1735
|
+
|
|
1736
|
+
server.listen(PORT, '127.0.0.1', () => {
|
|
1737
|
+
// A service start or a restart on a new port finds env-codex.* already pointing at the interceptor.
|
|
1738
|
+
const cfg = loadConfig();
|
|
1739
|
+
if (cfg && !getConfigLoadError()) serialized(() => reconcile(cfg));
|
|
1740
|
+
console.log(`[llm-switcher] Server running on http://127.0.0.1:${PORT}`);
|
|
1741
|
+
console.log(`[llm-switcher] Web UI available at: http://127.0.0.1:${PORT}/ui`);
|
|
1742
|
+
console.log(`[llm-switcher] Endpoints: /v1/messages (anthropic) | /v1/chat/completions (openai) | /v1/responses (codex) | /v1beta/models/* (vertex)`);
|
|
1743
|
+
});
|