llm-switcher 1.2.0 → 1.2.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (48) hide show
  1. package/CHANGELOG.md +23 -0
  2. package/README.md +1 -0
  3. package/blindfold/make-certs.sh +3 -1
  4. package/catalog.mjs +132 -42
  5. package/classifier.mjs +238 -0
  6. package/docs/TOKEN-OPTIMIZER-INTEROP.md +110 -110
  7. package/docs/response-matrix.json +1130 -1130
  8. package/formats.mjs +30 -1
  9. package/icons/antigravity.png +0 -0
  10. package/icons/claude.png +0 -0
  11. package/icons/codex.png +0 -0
  12. package/icons/deepseek.png +0 -0
  13. package/icons/gemini.png +0 -0
  14. package/icons/github.png +0 -0
  15. package/icons/groq.png +0 -0
  16. package/icons/intact.svg +1 -0
  17. package/icons/ollama.png +0 -0
  18. package/icons/openai.png +0 -0
  19. package/icons/openrouter.png +0 -0
  20. package/icons/qwen.png +0 -0
  21. package/icons/vertex.png +0 -0
  22. package/mcp.mjs +2 -2
  23. package/package.json +1 -1
  24. package/proxy.mjs +34 -20
  25. package/scripts/run-tests.mjs +3 -1
  26. package/skills/llm-switcher/SKILL.md +93 -93
  27. package/state.mjs +58 -4
  28. package/switch +0 -0
  29. package/switch.mjs +28 -4
  30. package/tests/blindfold-e2e.test.mjs +2 -2
  31. package/tests/blindfold-task5.test.mjs +1 -1
  32. package/tests/blindfold.task3.test.mjs +2 -2
  33. package/tests/blindfold.wire.test.mjs +3 -1
  34. package/tests/catalog.test.mjs +117 -19
  35. package/tests/classifier.test.mjs +210 -0
  36. package/tests/codex-daemon.test.mjs +92 -0
  37. package/tests/formats.test.mjs +30 -0
  38. package/tests/helpers.mjs +24 -24
  39. package/tests/lifecycle.test.mjs +49 -37
  40. package/tests/live-optimizer-interop.mjs +205 -205
  41. package/tests/make-certs.test.mjs +22 -0
  42. package/tests/real-user-sim.test.mjs +3 -0
  43. package/tests/shim.test.mjs +8 -3
  44. package/tests/switch.test.mjs +30 -24
  45. package/tests/ui.test.mjs +90 -0
  46. package/tests/version.test.mjs +78 -0
  47. package/ui.html +1895 -1616
  48. package/version.mjs +53 -0
package/formats.mjs CHANGED
@@ -19,6 +19,7 @@
19
19
  // params: { maxTokens, temperature, topP, topK, stop[],
20
20
  // presencePenalty?, frequencyPenalty? },
21
21
  // thinking: { type:'enabled'|'disabled'|'adaptive', budget?, effort? } | null,
22
+ // responseFormat: { type:'json_object' } | { type:'json_schema', name?, schema, strict? } | null,
22
23
  // stream: bool
23
24
  // }
24
25
  // ============================================================
@@ -327,10 +328,23 @@ function baseIR() {
327
328
  return {
328
329
  model: '', system: '', messages: [], tools: [], toolChoice: null,
329
330
  params: { maxTokens: null, temperature: null, topP: null, topK: null, stop: [] },
330
- thinking: null, stream: false
331
+ thinking: null, responseFormat: null, stream: false
331
332
  };
332
333
  }
333
334
 
335
+ // Structured output (Anthropic output_config.format, Responses text.format) -> IR responseFormat.
336
+ // `text` asks for nothing.
337
+ function responseFormatIR(f) {
338
+ if (!f || typeof f !== 'object') return null;
339
+ if (f.type === 'json_object') return { type: 'json_object' };
340
+ if (f.type !== 'json_schema') return null;
341
+ if (!f.schema || typeof f.schema !== 'object') return { type: 'json_object' };
342
+ const out = { type: 'json_schema', schema: f.schema };
343
+ if (typeof f.name === 'string' && f.name) out.name = f.name;
344
+ if (typeof f.strict === 'boolean') out.strict = f.strict;
345
+ return out;
346
+ }
347
+
334
348
  // Anthropic Messages API -> IR (logic ported from the old transformAnthropicToOpenAI).
335
349
  function anthropicToIR(payload) {
336
350
  const ir = baseIR();
@@ -423,6 +437,7 @@ function anthropicToIR(payload) {
423
437
  if (Array.isArray(payload.stop_sequences)) ir.params.stop = stopList(payload.stop_sequences);
424
438
 
425
439
  ir.thinking = thinkingFromAnthropicParam(payload.thinking, payload.output_config?.effort);
440
+ ir.responseFormat = responseFormatIR(payload.output_config?.format || payload.output_format);
426
441
  return ir;
427
442
  }
428
443
 
@@ -644,6 +659,7 @@ function responsesToIR(payload) {
644
659
  if (payload.reasoning && typeof payload.reasoning === 'object') {
645
660
  ir.thinking = thinkingFromReasoningParam(payload.reasoning);
646
661
  }
662
+ ir.responseFormat = responseFormatIR(payload.text?.format);
647
663
  return ir;
648
664
  }
649
665
 
@@ -822,6 +838,13 @@ function irToChatBody(ir, model, opts = {}) {
822
838
  if (typeof ir.params.frequencyPenalty === 'number') body.frequency_penalty = ir.params.frequencyPenalty;
823
839
  if (ir.params.stop.length) body.stop = ir.params.stop;
824
840
  if (ir.params.parallelToolCalls === false && body.tools) body.parallel_tool_calls = false;
841
+ if (ir.responseFormat?.type === 'json_schema') {
842
+ const js = { name: ir.responseFormat.name || 'response', schema: ir.responseFormat.schema };
843
+ if (typeof ir.responseFormat.strict === 'boolean') js.strict = ir.responseFormat.strict;
844
+ body.response_format = { type: 'json_schema', json_schema: js };
845
+ } else if (ir.responseFormat?.type === 'json_object') {
846
+ body.response_format = { type: 'json_object' };
847
+ }
825
848
 
826
849
  if (wantsThinking && mode === 'native') {
827
850
  body.reasoning_effort = ir.thinking?.effort && !['max', 'xhigh'].includes(ir.thinking.effort)
@@ -941,6 +964,8 @@ function irToAnthropicBody(ir, model) {
941
964
  if (typeof ir.params.topP === 'number') body.top_p = ir.params.topP;
942
965
  if (typeof ir.params.topK === 'number') body.top_k = ir.params.topK;
943
966
  if (ir.params.stop.length) body.stop_sequences = ir.params.stop;
967
+ // Anthropic has no JSON mode without a schema.
968
+ if (ir.responseFormat?.type === 'json_schema') body.output_config = { format: { type: 'json_schema', schema: ir.responseFormat.schema } };
944
969
  if (ir.thinking && ir.thinking.type === 'adaptive') {
945
970
  body.thinking = { type: 'adaptive' };
946
971
  } else if (ir.thinking && ir.thinking.type !== 'disabled' && body.max_tokens > 1024) {
@@ -1062,6 +1087,10 @@ function irToVertexBody(ir, model) {
1062
1087
  gc.thinkingConfig.thinkingBudget = clampBudget(ir.thinking.budget ?? (ir.thinking.effort ? effortToBudget(ir.thinking.effort) : 2048));
1063
1088
  }
1064
1089
  }
1090
+ if (ir.responseFormat) {
1091
+ gc.responseMimeType = 'application/json';
1092
+ if (ir.responseFormat.type === 'json_schema') gc.responseSchema = toGeminiSchema(ir.responseFormat.schema);
1093
+ }
1065
1094
  if (Object.keys(gc).length) body.generationConfig = gc;
1066
1095
  return body;
1067
1096
  }
Binary file
Binary file
Binary file
Binary file
Binary file
Binary file
package/icons/groq.png ADDED
Binary file
@@ -0,0 +1 @@
1
+ <svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 32 32"><defs><linearGradient id="lg" x1="0" y1="0" x2="1" y2="1"><stop offset="0" stop-color="#7c6cf6"/><stop offset="1" stop-color="#2dd4bf"/></linearGradient></defs><rect width="32" height="32" rx="9" fill="url(#lg)"/><path d="M16 6.5 24 9.6v6.1c0 5-3.4 8.6-8 10-4.6-1.4-8-5-8-10V9.6Z" fill="none" stroke="#fff" stroke-width="2.2" stroke-linejoin="round"/><path d="m12.4 16.2 2.6 2.6 4.8-5" fill="none" stroke="#fff" stroke-width="2.4" stroke-linecap="round" stroke-linejoin="round"/></svg>
Binary file
Binary file
Binary file
package/icons/qwen.png ADDED
Binary file
Binary file
package/mcp.mjs CHANGED
@@ -143,11 +143,11 @@ async function handleToolCall(name, args) {
143
143
  const port = getMcpPort();
144
144
 
145
145
  if (name === 'switcher_models') {
146
- const { loadCatalogCache, refreshCatalog } = await import('./catalog.mjs');
146
+ const { syncLocalCatalog, refreshCatalog } = await import('./catalog.mjs');
147
147
  if (args?.refresh) {
148
148
  await refreshCatalog(STATE_DIR);
149
149
  }
150
- const cache = loadCatalogCache(STATE_DIR);
150
+ const cache = syncLocalCatalog(STATE_DIR);
151
151
  const t = args?.tool?.toLowerCase();
152
152
  const res = {};
153
153
  if (!t || t === 'claude') res.claude = cache.claude;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "llm-switcher",
3
- "version": "1.2.0",
3
+ "version": "1.2.3",
4
4
  "description": "Zero-dependency multi-protocol edge gateway & provider switcher for Claude Code, Codex, OpenAI and Gemini clients",
5
5
  "keywords": [
6
6
  "llm",
package/proxy.mjs CHANGED
@@ -23,7 +23,8 @@ import {
23
23
  codexModelEntry, smallestWindows, publicModelWindows, model1MForSlot,
24
24
  contractLabSettings, STATE_DIR
25
25
  } from './state.mjs';
26
- import { classifyCodexRole, classifyClaudeTier, loadCatalogCache, refreshCatalog, checkVersionAndRefresh } from './catalog.mjs';
26
+ import { classifyCodexRole, classifyClaudeTier, syncLocalCatalog, refreshCatalog, checkVersionAndRefresh } from './catalog.mjs';
27
+ import { checkForUpdate } from './version.mjs';
27
28
  import { createContractLab, createHalfTap, tapClientWrites, capText, capJson, toolVersionFromUA, finishHalf, PROBE_HEADER, TRACE_ID_RE } from './contract.mjs';
28
29
 
29
30
  const __dirname = path.dirname(fileURLToPath(import.meta.url));
@@ -1018,15 +1019,6 @@ function validateProfileInput(p) {
1018
1019
  if (name !== '' && !isSafeModelName(name)) return `Invalid codexRoles.${slot} value "${name}"`;
1019
1020
  }
1020
1021
  }
1021
- if (p.blindfoldPort !== undefined && p.blindfoldPort !== '' && !parsePort(p.blindfoldPort)) {
1022
- return `Invalid blindfoldPort "${p.blindfoldPort}"`;
1023
- }
1024
- if (p.blindfoldHost !== undefined && p.blindfoldHost !== '' && !/^[A-Za-z0-9.-]{1,253}$/.test(p.blindfoldHost)) {
1025
- return `Invalid blindfoldHost "${p.blindfoldHost}"`;
1026
- }
1027
- if (p.blindfoldPrefix !== undefined && p.blindfoldPrefix !== '' && !/^\/[A-Za-z0-9._~/-]{0,200}$/.test(p.blindfoldPrefix)) {
1028
- return `Invalid blindfoldPrefix "${p.blindfoldPrefix}"`;
1029
- }
1030
1022
  return null;
1031
1023
  }
1032
1024
 
@@ -1118,11 +1110,32 @@ async function route(req, res) {
1118
1110
  res.writeHead(204, {
1119
1111
  'Access-Control-Allow-Origin': req.headers.origin || `http://127.0.0.1:${PORT}`,
1120
1112
  'Access-Control-Allow-Methods': 'GET, POST, OPTIONS',
1121
- 'Access-Control-Allow-Headers': 'Content-Type, Authorization'
1113
+ 'Access-Control-Allow-Headers': 'Content-Type, Authorization, x-llm-switcher-token'
1122
1114
  });
1123
1115
  return res.end();
1124
1116
  }
1125
1117
 
1118
+ // Serve static tool icons (cached, immutable, strictly image/png and image/svg+xml)
1119
+ if (method === 'GET' && pathname.startsWith('/icons/')) {
1120
+ const iconName = path.basename(pathname);
1121
+ if (!/\.(png|svg)$/i.test(iconName)) {
1122
+ res.writeHead(404, { 'Content-Type': 'text/plain' });
1123
+ return res.end('Icon not found');
1124
+ }
1125
+ const iconPath = path.join(__dirname, 'icons', iconName);
1126
+ if (fs.existsSync(iconPath)) {
1127
+ const ext = path.extname(iconName).toLowerCase();
1128
+ const contentType = ext === '.svg' ? 'image/svg+xml' : 'image/png';
1129
+ res.writeHead(200, {
1130
+ 'Content-Type': contentType,
1131
+ 'Cache-Control': 'public, max-age=604800, immutable'
1132
+ });
1133
+ return fs.createReadStream(iconPath).pipe(res);
1134
+ }
1135
+ res.writeHead(404, { 'Content-Type': 'text/plain' });
1136
+ return res.end('Icon not found');
1137
+ }
1138
+
1126
1139
  // Serve Web UI (no-cache: always serve the latest version after file edits)
1127
1140
  if (method === 'GET' && (pathname === '/' || pathname === '/ui')) {
1128
1141
  if (fs.existsSync(uiHtmlPath)) {
@@ -1275,6 +1288,15 @@ async function routeApi(req, res, method, pathname) {
1275
1288
  return sendJson(res, 200, { logs: requestLogs.slice().reverse() });
1276
1289
  }
1277
1290
 
1291
+ if (method === 'GET' && pathname === '/api/version') {
1292
+ return sendJson(res, 200, await checkForUpdate({ stateDir: STATE_DIR }));
1293
+ }
1294
+
1295
+ // GET /api/catalog (Dynamic Model Discovery)
1296
+ if (method === 'GET' && pathname === '/api/catalog') {
1297
+ return sendJson(res, 200, syncLocalCatalog(STATE_DIR));
1298
+ }
1299
+
1278
1300
  if (method !== 'POST') {
1279
1301
  req.resume();
1280
1302
  return sendJson(res, 404, { error: `Not found: ${method} ${pathname}` });
@@ -1287,21 +1309,13 @@ async function routeApi(req, res, method, pathname) {
1287
1309
  return sendJson(res, err.status || 400, { error: err.message });
1288
1310
  }
1289
1311
 
1290
- // GET /api/catalog
1291
- if (method === 'GET' && pathname === '/api/catalog') {
1292
- return sendJson(res, 200, loadCatalogCache(STATE_DIR));
1293
- }
1294
-
1295
1312
  // POST /api/catalog/refresh
1296
1313
  if (method === 'POST' && pathname === '/api/catalog/refresh') {
1297
1314
  const active = getActiveMap(requireConfig(res) || {});
1298
1315
  const cfg = requireConfig(res) || {};
1299
1316
  const claudeProfile = active.claude ? cfg.profiles?.[active.claude] : null;
1300
1317
  const codexProfile = active.codex ? cfg.profiles?.[active.codex] : null;
1301
- const updated = await refreshCatalog(STATE_DIR, {
1302
- claudeKey: claudeProfile?.apiKey,
1303
- codexKey: codexProfile?.apiKey
1304
- });
1318
+ const updated = await refreshCatalog(STATE_DIR, { claudeProfile, codexProfile });
1305
1319
  return sendJson(res, 200, { success: true, catalog: updated });
1306
1320
  }
1307
1321
 
@@ -15,5 +15,7 @@ const files = [];
15
15
  }
16
16
  })(path.join(root, 'tests'));
17
17
 
18
- const r = spawnSync(process.execPath, ['--test', ...process.argv.slice(2), ...files], { cwd: root, stdio: 'inherit' });
18
+ // A version check must never reach the real npm registry.
19
+ const env = { ...process.env, LLM_SWITCHER_REGISTRY_URL: 'http://127.0.0.1:9/' };
20
+ const r = spawnSync(process.execPath, ['--test', ...process.argv.slice(2), ...files], { cwd: root, stdio: 'inherit', env });
19
21
  process.exit(r.status ?? 1);
@@ -1,93 +1,93 @@
1
- ---
2
- name: llm-switcher
3
- description: Mandatory guidance for AI coding agents and token compression tools (Headroom, RTK, Ponytail, Repomix). Enforces routing all LLM traffic through the local LLM Switcher edge gateway (http://127.0.0.1:3456) instead of directly calling provider endpoints.
4
- ---
5
-
6
- # LLM Switcher Edge Gateway Skill
7
-
8
- This skill governs how AI coding agents (Claude Code, Codex, Cursor, Windsurf, Opencode) and prompt optimization/compression tools (Headroom, RTK, Ponytail, Repomix) must interact with LLM providers on this workstation.
9
-
10
- ## 1. The Core Directive
11
-
12
- > **ALL OUTBOUND LLM REQUESTS MUST ROUTE THROUGH LLM SWITCHER (`http://127.0.0.1:3456`).**
13
- >
14
- > Never configure any tool, script, sub-agent, or token compression proxy to call external provider endpoints directly (such as `https://api.anthropic.com`, `https://api.openai.com`, or `https://generativelanguage.googleapis.com`).
15
- >
16
- > If a token compression tool (Headroom, RTK, Ponytail) is used, it MUST be positioned as an **inner processing layer**, and its **upstream destination MUST be configured to point to LLM Switcher** at `http://127.0.0.1:3456`.
17
-
18
- ```
19
- [Client CLI / Agent]
20
- │
21
- ▼
22
- [Token Compressor: RTK / Headroom / Ponytail] <-- Inner Layer
23
- │
24
- ▼ (MANDATORY: Upstream MUST point to http://127.0.0.1:3456)
25
- [LLM Switcher Edge Gateway (:3456)] <-- Outermost Gatekeeper
26
- │
27
- ▼ (Heals schemas, maps 1M context, extracts thinking, converts protocol)
28
- [Internet / intact (Recommended) / 9Router / Upstream LLM]
29
- ```
30
-
31
- ## 2. Why This is Mandatory
32
-
33
- Direct outbound calls from token compressors will break the developer environment for several reasons:
34
-
35
- 1. **Schema Breakage & The Healer Engine:**
36
- Token optimizers frequently prune conversation history to save tokens. In doing so, they often delete an `assistant` turn containing a `tool_use`, while leaving the subsequent `user` turn containing a `tool_result`.
37
- - Anthropic and OpenAI APIs strictly reject this with `HTTP 400 (orphaned tool_use_id)`.
38
- - **LLM Switcher** intercepts these malformed requests and runs its **Healer Engine**, automatically converting orphaned results into safe context text blocks and merging consecutive same-role turns so upstream providers accept the payload cleanly.
39
- 2. **Thinking / Reasoning Protection:**
40
- Many token compressors naively strip `thinking` parameters or reasoning budgets to reduce token count.
41
- - When routed through LLM Switcher, the gateway detects reasoning models (such as `ag/claude-opus-4-6-thinking` or `ag/gemini-3.8-flash`) and **automatically restores thinking parameters**, ensuring the model retains its full reasoning capabilities.
42
- 3. **Official Context Windows & Dynamic Model Discovery:**
43
- Context windows follow the model you select without mutating permanent settings. The gateway auto-discovers official models from Anthropic and OpenAI, refreshing mappings on tool version upgrades (`switch models`).
44
- 4. **Server Gateway Synergy (intact / 9Router):**
45
- LLM Switcher pairs directly with server-side gateways like **intact** (strongly recommended for credential proxying, account pooling, and contract lab self-healing) or 9Router, handling retries and quotas far better than any local tool.
46
-
47
- ## 3. Configuring Token Compression Tools
48
-
49
- When setting up or assisting the user with token compression tools, apply these exact configurations:
50
-
51
- ### A. Headroom (Prompt Compression Proxy)
52
- When configuring Headroom's upstream endpoint:
53
- - In Headroom config (`headroom.yaml` or environment variables):
54
- ```bash
55
- ANTHROPIC_BASE_URL="http://127.0.0.1:3456"
56
- OPENAI_BASE_URL="http://127.0.0.1:3456/v1"
57
- ```
58
- - Point your CLI to Headroom's listening port (e.g. `http://127.0.0.1:8787`). Headroom compresses the prompt and forwards it to LLM Switcher on port `3456`.
59
-
60
- ### B. RTK (Rust Token Killer / Repomix)
61
- - In RTK configuration or execution flags:
62
- ```bash
63
- # Ensure RTK's target upstream proxy is LLM Switcher
64
- export UPSTREAM_URL="http://127.0.0.1:3456"
65
- ```
66
-
67
- ### C. Ponytail / Custom Proxies
68
- - Always set the proxy's upstream host to:
69
- `http://127.0.0.1:3456` (or `http://127.0.0.1:3456/v1` for OpenAI-compatible tools).
70
-
71
- ### D. Sub-Agents & Spawned Terminal Commands
72
- When an agent spawns sub-processes that call LLM CLIs:
73
- - **Never advise sourcing `env.sh` or `env.cmd`.** Those files carry no proxy variables any
74
- more, so sourcing one changes nothing — and a stale variable an older version once wrote
75
- would still point the tool at a port where nothing listens.
76
- - Run `switch claude <profile>` or `switch codex <profile>` instead. The shims in
77
- `~/.llm-switcher/bin` inject the proxy variables into the tool process itself and clear
78
- anything stale first.
79
- - `claude` and `codex` on `PATH` already are those shims: invoke them normally, and never
80
- wrap them in another script.
81
-
82
- ## 4. Operational Rules for AI Agents
83
-
84
- 1. **NEVER edit `~/.claude/settings.json` directly.**
85
- LLM Switcher uses zero-mutation launcher flags (`active.flag`). Writing custom base URLs or auth tokens into `settings.json` triggers warning banners in Claude Code and risks breaking the user's official subscription. Use `switch` CLI commands instead.
86
- 2. **Check Gateway Health Before Complex Operations:**
87
- Run `switch status` or call the `switcher_audit` MCP tool to confirm:
88
- - LLM Switcher is active on port `3456`.
89
- - The active profile matches the intended CLI target (Claude Code, Codex, or OpenAI).
90
- 3. **Verify Routing When Errors Occur:**
91
- If a tool fails with `HTTP 400`, `HTTP 502`, or connection errors:
92
- - Run `switch doctor` to audit port collisions and environment variables.
93
- - Inspect recent request logs via `http://127.0.0.1:3456/ui` (Tab 4: Live Inspector) to see if an intermediary tool mangled the payload.
1
+ ---
2
+ name: llm-switcher
3
+ description: Mandatory guidance for AI coding agents and token compression tools (Headroom, RTK, Ponytail, Repomix). Enforces routing all LLM traffic through the local LLM Switcher edge gateway (http://127.0.0.1:3456) instead of directly calling provider endpoints.
4
+ ---
5
+
6
+ # LLM Switcher Edge Gateway Skill
7
+
8
+ This skill governs how AI coding agents (Claude Code, Codex, Cursor, Windsurf, Opencode) and prompt optimization/compression tools (Headroom, RTK, Ponytail, Repomix) must interact with LLM providers on this workstation.
9
+
10
+ ## 1. The Core Directive
11
+
12
+ > **ALL OUTBOUND LLM REQUESTS MUST ROUTE THROUGH LLM SWITCHER (`http://127.0.0.1:3456`).**
13
+ >
14
+ > Never configure any tool, script, sub-agent, or token compression proxy to call external provider endpoints directly (such as `https://api.anthropic.com`, `https://api.openai.com`, or `https://generativelanguage.googleapis.com`).
15
+ >
16
+ > If a token compression tool (Headroom, RTK, Ponytail) is used, it MUST be positioned as an **inner processing layer**, and its **upstream destination MUST be configured to point to LLM Switcher** at `http://127.0.0.1:3456`.
17
+
18
+ ```
19
+ [Client CLI / Agent]
20
+ │
21
+ ▼
22
+ [Token Compressor: RTK / Headroom / Ponytail] <-- Inner Layer
23
+ │
24
+ ▼ (MANDATORY: Upstream MUST point to http://127.0.0.1:3456)
25
+ [LLM Switcher Edge Gateway (:3456)] <-- Outermost Gatekeeper
26
+ │
27
+ ▼ (Heals schemas, maps 1M context, extracts thinking, converts protocol)
28
+ [Internet / intact (Recommended) / 9Router / Upstream LLM]
29
+ ```
30
+
31
+ ## 2. Why This is Mandatory
32
+
33
+ Direct outbound calls from token compressors will break the developer environment for several reasons:
34
+
35
+ 1. **Schema Breakage & The Healer Engine:**
36
+ Token optimizers frequently prune conversation history to save tokens. In doing so, they often delete an `assistant` turn containing a `tool_use`, while leaving the subsequent `user` turn containing a `tool_result`.
37
+ - Anthropic and OpenAI APIs strictly reject this with `HTTP 400 (orphaned tool_use_id)`.
38
+ - **LLM Switcher** intercepts these malformed requests and runs its **Healer Engine**, automatically converting orphaned results into safe context text blocks and merging consecutive same-role turns so upstream providers accept the payload cleanly.
39
+ 2. **Thinking / Reasoning Protection:**
40
+ Many token compressors naively strip `thinking` parameters or reasoning budgets to reduce token count.
41
+ - When routed through LLM Switcher, the gateway detects reasoning models (such as `ag/claude-opus-4-6-thinking` or `ag/gemini-3.8-flash`) and **automatically restores thinking parameters**, ensuring the model retains its full reasoning capabilities.
42
+ 3. **Official Context Windows & Dynamic Model Discovery:**
43
+ Context windows follow the model you select without mutating permanent settings. The gateway auto-discovers official models from Anthropic and OpenAI, refreshing mappings on tool version upgrades (`switch models`).
44
+ 4. **Server Gateway Synergy (intact / 9Router):**
45
+ LLM Switcher pairs directly with server-side gateways like **intact** (strongly recommended for credential proxying, account pooling, and contract lab self-healing) or 9Router, handling retries and quotas far better than any local tool.
46
+
47
+ ## 3. Configuring Token Compression Tools
48
+
49
+ When setting up or assisting the user with token compression tools, apply these exact configurations:
50
+
51
+ ### A. Headroom (Prompt Compression Proxy)
52
+ When configuring Headroom's upstream endpoint:
53
+ - In Headroom config (`headroom.yaml` or environment variables):
54
+ ```bash
55
+ ANTHROPIC_BASE_URL="http://127.0.0.1:3456"
56
+ OPENAI_BASE_URL="http://127.0.0.1:3456/v1"
57
+ ```
58
+ - Point your CLI to Headroom's listening port (e.g. `http://127.0.0.1:8787`). Headroom compresses the prompt and forwards it to LLM Switcher on port `3456`.
59
+
60
+ ### B. RTK (Rust Token Killer / Repomix)
61
+ - In RTK configuration or execution flags:
62
+ ```bash
63
+ # Ensure RTK's target upstream proxy is LLM Switcher
64
+ export UPSTREAM_URL="http://127.0.0.1:3456"
65
+ ```
66
+
67
+ ### C. Ponytail / Custom Proxies
68
+ - Always set the proxy's upstream host to:
69
+ `http://127.0.0.1:3456` (or `http://127.0.0.1:3456/v1` for OpenAI-compatible tools).
70
+
71
+ ### D. Sub-Agents & Spawned Terminal Commands
72
+ When an agent spawns sub-processes that call LLM CLIs:
73
+ - **Never advise sourcing `env.sh` or `env.cmd`.** Those files carry no proxy variables any
74
+ more, so sourcing one changes nothing — and a stale variable an older version once wrote
75
+ would still point the tool at a port where nothing listens.
76
+ - Run `switch claude <profile>` or `switch codex <profile>` instead. The shims in
77
+ `~/.llm-switcher/bin` inject the proxy variables into the tool process itself and clear
78
+ anything stale first.
79
+ - `claude` and `codex` on `PATH` already are those shims: invoke them normally, and never
80
+ wrap them in another script.
81
+
82
+ ## 4. Operational Rules for AI Agents
83
+
84
+ 1. **NEVER edit `~/.claude/settings.json` directly.**
85
+ LLM Switcher uses zero-mutation launcher flags (`active.flag`). Writing custom base URLs or auth tokens into `settings.json` triggers warning banners in Claude Code and risks breaking the user's official subscription. Use `switch` CLI commands instead.
86
+ 2. **Check Gateway Health Before Complex Operations:**
87
+ Run `switch status` or call the `switcher_audit` MCP tool to confirm:
88
+ - LLM Switcher is active on port `3456`.
89
+ - The active profile matches the intended CLI target (Claude Code, Codex, or OpenAI).
90
+ 3. **Verify Routing When Errors Occur:**
91
+ If a tool fails with `HTTP 400`, `HTTP 502`, or connection errors:
92
+ - Run `switch doctor` to audit port collisions and environment variables.
93
+ - Inspect recent request logs via `http://127.0.0.1:3456/ui` (Tab 4: Live Inspector) to see if an intermediary tool mangled the payload.
package/state.mjs CHANGED
@@ -1441,26 +1441,78 @@ function writeToolFiles(pairs, shPath, cmdPath) {
1441
1441
  // launcher files are never opened (A13: emptied, never deleted).
1442
1442
  export function emptyToolEnvFiles(tool) {
1443
1443
  const codex = tool === 'codex';
1444
- withLaunchLock(() => {
1444
+ const codexRouteChanged = withLaunchLock(() => {
1445
+ const before = codex ? readOrEmpty(paths.envCodexSh) : '';
1445
1446
  writeAtomic(codex ? paths.envCodexCmd : paths.envClaudeCmd, '');
1446
1447
  writeAtomic(codex ? paths.envCodexSh : paths.envClaudeSh, '');
1448
+ return before !== '';
1447
1449
  });
1450
+ return { codexRouteChanged, codexDaemon: codexRouteChanged ? restartCodexDaemon([]) : undefined };
1448
1451
  }
1449
1452
 
1450
1453
  // Write the per-tool env files from activeProfiles. settings.json is never touched here: it
1451
1454
  // belongs to the coding tool (R1), and model-catalog.json is no longer written (R9).
1452
1455
  export function applyLaunchState(cfg, port, opts = {}) {
1453
- return withLaunchLock(() => writeLaunchState(cfg, port, opts));
1456
+ const st = withLaunchLock(() => writeLaunchState(cfg, port, opts));
1457
+ if (st.codexRouteChanged) st.codexDaemon = restartCodexDaemon(st.envCodex);
1458
+ return st;
1459
+ }
1460
+
1461
+ const LOOPBACK_URL = /^https?:\/\/(127\.0\.0\.1|localhost|\[::1\])(:\d+)?\/?$/i;
1462
+
1463
+ /**
1464
+ * The interactive Codex TUI talks to a shared `codex app-server` daemon, and the daemon keeps the
1465
+ * environment it started with. A new route reaches it only through a restart. Only the launch files
1466
+ * that the installed codex shim reads can route that daemon, so any other state dir (tests) is a no-op.
1467
+ */
1468
+ export function restartCodexDaemon(pairs, {
1469
+ home = os.homedir(),
1470
+ codexHome = process.env.CODEX_HOME || path.join(home, '.codex'),
1471
+ stateDir = STATE_DIR,
1472
+ baseEnv = process.env,
1473
+ ca = paths.blindfoldCA,
1474
+ spawnFn = spawn
1475
+ } = {}) {
1476
+ let shim = '';
1477
+ try { shim = fs.readFileSync(path.join(home, '.llm-switcher', 'bin', 'codex'), 'utf8'); } catch {}
1478
+ if (!shim.includes(`SWITCHER_DIR="${stateDir}"`)) return { restarted: false, reason: 'the codex shim does not read this state dir' };
1479
+ const bin = path.join(codexHome, 'packages', 'app-server-daemon', 'current', 'bin', 'codex');
1480
+ // The socket is a link to the running daemon; it no longer resolves once the daemon stopped.
1481
+ if (!fs.existsSync(bin) || !fs.existsSync(path.join(codexHome, 'app-server-control', 'app-server-control.sock'))) {
1482
+ return { restarted: false, reason: 'no Codex daemon runs' };
1483
+ }
1484
+
1485
+ // Remove only the values this switcher writes (R8); a proxy or CA the user set stays.
1486
+ const env = { ...baseEnv };
1487
+ for (const k of ['HTTPS_PROXY', 'https_proxy', 'HTTP_PROXY', 'http_proxy']) if (LOOPBACK_URL.test(env[k] || '')) delete env[k];
1488
+ for (const k of ['NO_PROXY', 'no_proxy']) if (env[k] === '127.0.0.1,localhost') delete env[k];
1489
+ if (env.CODEX_CA_CERTIFICATE === ca) delete env.CODEX_CA_CERTIFICATE;
1490
+ for (const [k, v] of pairs) env[k] = v;
1491
+
1492
+ try {
1493
+ const child = spawnFn(bin, ['app-server', 'daemon', 'restart'], { env, stdio: 'ignore', detached: true });
1494
+ child.on?.('error', () => {});
1495
+ child.unref?.();
1496
+ return { restarted: true };
1497
+ } catch (err) {
1498
+ return { restarted: false, reason: err.message };
1499
+ }
1500
+ }
1501
+
1502
+ function readOrEmpty(file) {
1503
+ try { return fs.readFileSync(file, 'utf8'); } catch { return ''; }
1454
1504
  }
1455
1505
 
1456
1506
  function writeLaunchState(cfg, port) {
1457
1507
  const st = computeLaunchState(cfg, port);
1508
+ const codexBefore = readOrEmpty(paths.envCodexSh);
1458
1509
 
1459
1510
  // Env files first, the flag last: a flag that says "active" while a file is missing or stale
1460
1511
  // makes the launcher route with the wrong variables. A failed write leaves the flag as it was.
1461
1512
  try {
1462
1513
  writeToolFiles(st.envClaude, paths.envClaudeSh, paths.envClaudeCmd);
1463
1514
  writeToolFiles(st.envCodex, paths.envCodexSh, paths.envCodexCmd);
1515
+ st.codexRouteChanged = readOrEmpty(paths.envCodexSh) !== codexBefore;
1464
1516
  // Recorded on switch on so a shell opened while the gateway ran can still recognize its own
1465
1517
  // stale loopback URL later, after the port changed (R8).
1466
1518
  if (st.active) writeAtomic(paths.gatewayPort, `${port}\n`);
@@ -1476,7 +1528,8 @@ function writeLaunchState(cfg, port) {
1476
1528
  }
1477
1529
 
1478
1530
  export function clearLaunchState(port) {
1479
- withLaunchLock(() => {
1531
+ const codexRouteChanged = withLaunchLock(() => {
1532
+ const codexBefore = readOrEmpty(paths.envCodexSh);
1480
1533
  writeOrRemove(paths.activeFlag, null);
1481
1534
  // Stubs rather than deletions, and the tool files emptied rather than removed (R8, A13):
1482
1535
  // `switch off claude` must leave an empty claude file behind, and bare `switch off` must
@@ -1486,9 +1539,10 @@ export function clearLaunchState(port) {
1486
1539
  for (const f of [paths.envClaudeCmd, paths.envClaudeSh, paths.envCodexCmd, paths.envCodexSh]) {
1487
1540
  writeAtomic(f, '');
1488
1541
  }
1542
+ return codexBefore !== '';
1489
1543
  });
1490
1544
  // settings.json belongs to the coding tool: the switcher neither reads nor writes it (R1).
1491
- return { changed: false, removed: [] };
1545
+ return { changed: false, removed: [], codexRouteChanged, codexDaemon: codexRouteChanged ? restartCodexDaemon([]) : undefined };
1492
1546
  }
1493
1547
 
1494
1548
  export function readLaunchFlags() {
package/switch CHANGED
File without changes
package/switch.mjs CHANGED
@@ -26,6 +26,12 @@ const proxyScript = path.join(ROOT_DIR, 'proxy.mjs');
26
26
  const proxyLogPath = paths.proxyLog;
27
27
  const userProfile = os.homedir();
28
28
 
29
+ // The version needs no config.json: it must also answer before the first setup.
30
+ if (['version', '--version', '-v'].includes(String(process.argv[2] || '').toLowerCase())) {
31
+ await showVersion();
32
+ process.exit(0);
33
+ }
34
+
29
35
  const config = loadConfig();
30
36
  if (!config) {
31
37
  const err = getConfigLoadError();
@@ -415,6 +421,7 @@ async function turnOn(profileName, cliTarget) {
415
421
  const savedBytes = fs.readFileSync(configPath);
416
422
  const st = applyLaunchState(planned, port);
417
423
  reportSettings(st.settings);
424
+ reportCodexDaemon(st);
418
425
  const bf = await requestBlindfoldSync(port);
419
426
  if (!bf.ok) {
420
427
  console.error(`[Error] ${bf.error}`);
@@ -520,7 +527,7 @@ async function turnOff(targetArg) {
520
527
  // The order is fixed by R7b: the CAS write, then this tool's env files, then the active-tools
521
528
  // update. If the CAS write cannot land, the env file and the interceptor stay untouched.
522
529
  const saved = casOff([target]);
523
- emptyToolEnvFiles(target);
530
+ reportCodexDaemon(emptyToolEnvFiles(target));
524
531
  const bf = await syncOrStopBlindfold(port, config);
525
532
  if (!bf.ok) {
526
533
  console.error(`[Error] ${target} is switched back, but the interceptor is not in line: ${bf.error}`);
@@ -534,6 +541,7 @@ async function turnOff(targetArg) {
534
541
  saveConfig(config);
535
542
  const st = applyLaunchState(config, port);
536
543
  reportSettings(st.settings);
544
+ reportCodexDaemon(st);
537
545
  const bf = await syncOrStopBlindfold(port, config);
538
546
  if (!bf.ok) {
539
547
  console.error(`[Error] ${target} is switched back, but the blindfold interceptor is not in line: ${bf.error}`);
@@ -581,6 +589,12 @@ function auditActiveClis() {
581
589
  function reportSettings(result) {
582
590
  if (result?.removed?.length) console.log(`[settings.json] Removed switcher-written values: ${result.removed.join(', ')}`);
583
591
  if (result?.error) console.warn(`[settings.json] Not cleaned: ${result.error}`);
592
+ reportCodexDaemon(result);
593
+ }
594
+
595
+ // The interactive Codex TUI keeps using its shared daemon, so a new route needs that daemon restarted.
596
+ function reportCodexDaemon(st) {
597
+ if (st?.codexDaemon?.restarted) console.log('[Codex] Restarted the codex app-server daemon: open Codex sessions reconnect with the new route.');
584
598
  }
585
599
 
586
600
  async function showStatus() {
@@ -848,7 +862,7 @@ async function runDoctor() {
848
862
  const p = config.profiles[key];
849
863
  if (!p) warn(`[WARN] Target ${t} points to missing profile "${show(key)}".`);
850
864
  else if (!p.baseURL || /YOUR-|REPLACE-ME/i.test(`${p.baseURL} ${p.apiKey}`)) warn(`[WARN] Profile "${show(key)}" (${t}) still has placeholder baseURL/apiKey.`);
851
- if (p && t === 'responses') {
865
+ if (p && t === 'codex') {
852
866
  const codexWarning = codexPublicModelsWarning(show(key), p);
853
867
  if (codexWarning) warn(`[WARN] ${codexWarning}`);
854
868
  }
@@ -1036,13 +1050,22 @@ async function runContractCheck() {
1036
1050
  }
1037
1051
  }
1038
1052
 
1053
+ async function showVersion() {
1054
+ const { checkForUpdate } = await import('./version.mjs');
1055
+ const v = await checkForUpdate({ stateDir: STATE_DIR });
1056
+ console.log(`llm-switcher ${v.current}`);
1057
+ if (v.updateAvailable) console.log(`A new version is available: ${v.latest}. Update with: ${v.updateCommand}`);
1058
+ else if (!v.latest) console.log('Could not check the npm registry for a newer version.');
1059
+ else console.log('This is the latest version.');
1060
+ }
1061
+
1039
1062
  async function showModels(target) {
1040
- const { loadCatalogCache, refreshCatalog } = await import('./catalog.mjs');
1063
+ const { syncLocalCatalog, refreshCatalog } = await import('./catalog.mjs');
1041
1064
  if (optionValue('--refresh') || process.argv.includes('--refresh')) {
1042
1065
  console.log('Refreshing models catalog from official endpoints...');
1043
1066
  await refreshCatalog(STATE_DIR);
1044
1067
  }
1045
- const cache = loadCatalogCache(STATE_DIR);
1068
+ const cache = syncLocalCatalog(STATE_DIR);
1046
1069
  console.log('=== LLM Switcher Discovered Models Catalog ===\n');
1047
1070
  const t = target ? target.toLowerCase() : '';
1048
1071
  if (!t || t === 'claude') {
@@ -1105,6 +1128,7 @@ if (cmd === 'off' || cmd === 'stop') {
1105
1128
  console.log('Usage:');
1106
1129
  console.log(' switch ui # Open Web UI dashboard');
1107
1130
  console.log(' switch status # Show multi-CLI active status');
1131
+ console.log(' switch version # Show the version and check for a newer one');
1108
1132
  console.log(' switch doctor # Audit environment, settings & routing');
1109
1133
  console.log(' switch on [profile] # Start gateway & activate profile for all compatible targets');
1110
1134
  console.log(' switch <profile> # Activate profile for all compatible targets');