llm-switcher 1.2.10 → 1.2.12
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +12 -0
- package/README.md +15 -1
- package/README.vi.md +16 -1
- package/docs/TOKEN-OPTIMIZER-INTEROP.md +110 -110
- package/docs/response-matrix.json +1130 -1130
- package/package.json +1 -1
- package/proxy.mjs +64 -2
- package/service.mjs +12 -6
- package/skills/llm-switcher/SKILL.md +93 -93
- package/switch +0 -0
- package/switch.cmd +2 -2
- package/switch.mjs +100 -15
- package/tests/gateway.e2e.test.mjs +11 -0
- package/tests/helpers.mjs +24 -24
- package/tests/live-optimizer-interop.mjs +205 -205
- package/tests/service.test.mjs +27 -1
- package/tests/update.e2e.test.mjs +197 -0
- package/tests/update.test.mjs +145 -0
- package/ui.html +70 -1
- package/update.mjs +98 -0
package/package.json
CHANGED
package/proxy.mjs
CHANGED
|
@@ -4,6 +4,7 @@ import path from 'node:path';
|
|
|
4
4
|
import zlib from 'node:zlib';
|
|
5
5
|
import crypto from 'node:crypto';
|
|
6
6
|
import { promisify } from 'node:util';
|
|
7
|
+
import { spawn } from 'node:child_process';
|
|
7
8
|
import { fileURLToPath } from 'node:url';
|
|
8
9
|
import { createFrameReader } from './blindfold/wsframe.mjs';
|
|
9
10
|
import {
|
|
@@ -21,10 +22,11 @@ import {
|
|
|
21
22
|
modelForSlot, primaryModel, codexPublicModel, isSafeModelName, parsePort, CODEX_MODEL_SLOTS,
|
|
22
23
|
ensureAdminToken, identityProof, reconcileBlindfold, checkBlindfoldTarget,
|
|
23
24
|
codexModelEntry, smallestWindows, publicModelWindows, model1MForSlot,
|
|
24
|
-
contractLabSettings, STATE_DIR
|
|
25
|
+
contractLabSettings, STATE_DIR, stopRecordedBlindfold
|
|
25
26
|
} from './state.mjs';
|
|
26
27
|
import { classifyCodexRole, classifyClaudeTier, syncLocalCatalog, refreshCatalog, checkVersionAndRefresh } from './catalog.mjs';
|
|
27
28
|
import { checkForUpdate } from './version.mjs';
|
|
29
|
+
import { applyUpdate } from './update.mjs';
|
|
28
30
|
import { createContractLab, createHalfTap, tapClientWrites, capText, capJson, toolVersionFromUA, finishHalf, PROBE_HEADER, TRACE_ID_RE } from './contract.mjs';
|
|
29
31
|
|
|
30
32
|
const __dirname = path.dirname(fileURLToPath(import.meta.url));
|
|
@@ -46,6 +48,7 @@ if (missingNoProxy.length > 0) {
|
|
|
46
48
|
// Fixed port for the process lifetime: changing "port" in config.json at runtime has no effect
|
|
47
49
|
// env.cmd / flags would point at a port the server is not listening on.
|
|
48
50
|
const PORT = resolvePort();
|
|
51
|
+
let updateInProgress = false;
|
|
49
52
|
|
|
50
53
|
// In-memory Request / Response Inspector Ring Buffer (up to 40 most recent requests)
|
|
51
54
|
const requestLogs = [];
|
|
@@ -1123,6 +1126,14 @@ async function testUpstream(body, cfg) {
|
|
|
1123
1126
|
const apiKey = resolveApiKey(cfg, body.key, body.apiKey, baseURL);
|
|
1124
1127
|
const model = body.model || 'default';
|
|
1125
1128
|
const profile = { baseURL, apiKey, mode: body.mode, outFormat: body.outFormat || undefined };
|
|
1129
|
+
// A Bifrost model answers only its own client: a plain test request would draw a fake 429.
|
|
1130
|
+
// The model lookup itself proves the key and the model, so it is the test.
|
|
1131
|
+
const lookupStart = Date.now();
|
|
1132
|
+
const bifrostUA = await bifrostUAFor(profile, model);
|
|
1133
|
+
if (bifrostUA) {
|
|
1134
|
+
return { status: 200, json: { ok: true, latency: Date.now() - lookupStart, outFormat: 'bifrost',
|
|
1135
|
+
sample: `Bifrost: the key works and ${model} is served. Only the real client (${bifrostUA}) can call it, so no test message was sent.` } };
|
|
1136
|
+
}
|
|
1126
1137
|
const outFormat = resolveOutFormat(profile, model);
|
|
1127
1138
|
const { url, headers } = upstreamEndpoint(profile, outFormat, model, false, null);
|
|
1128
1139
|
const ir = { model, system: '', messages: [{ role: 'user', content: 'ping' }], tools: [], toolChoice: null, params: { maxTokens: 16, temperature: null, topP: null, topK: null, stop: [] }, thinking: { type: 'disabled' }, stream: false };
|
|
@@ -1408,6 +1419,28 @@ async function routeApi(req, res, method, pathname) {
|
|
|
1408
1419
|
return sendJson(res, 200, { success: true, catalog: updated });
|
|
1409
1420
|
}
|
|
1410
1421
|
|
|
1422
|
+
// POST /api/update: progress as SSE; after a successful update the gateway hands over to the new code.
|
|
1423
|
+
if (pathname === '/api/update') {
|
|
1424
|
+
if (updateInProgress) return sendJson(res, 409, { error: 'An update is already in progress' });
|
|
1425
|
+
updateInProgress = true;
|
|
1426
|
+
res.writeHead(200, { 'Content-Type': 'text/event-stream', 'Cache-Control': 'no-cache', 'X-Accel-Buffering': 'no' });
|
|
1427
|
+
let result = null;
|
|
1428
|
+
try {
|
|
1429
|
+
result = await applyUpdate({ logger: (text) => sendSSE(res, 'log', { text }) });
|
|
1430
|
+
sendSSE(res, 'result', result);
|
|
1431
|
+
} catch (err) {
|
|
1432
|
+
sendSSE(res, 'error', { message: err.message });
|
|
1433
|
+
}
|
|
1434
|
+
res.end();
|
|
1435
|
+
if (result?.updated) {
|
|
1436
|
+
console.log(`[llm-switcher:update] v${result.from} -> v${result.to}. Handing over to the new code.`);
|
|
1437
|
+
handOver();
|
|
1438
|
+
} else {
|
|
1439
|
+
updateInProgress = false;
|
|
1440
|
+
}
|
|
1441
|
+
return;
|
|
1442
|
+
}
|
|
1443
|
+
|
|
1411
1444
|
// POST /api/logs/clear
|
|
1412
1445
|
if (pathname === '/api/logs/clear') {
|
|
1413
1446
|
requestLogs.length = 0;
|
|
@@ -1961,7 +1994,36 @@ if (getMigrationCollision()) {
|
|
|
1961
1994
|
console.error(`[llm-switcher:ERROR] Configuration migration collision: ${JSON.stringify(getMigrationCollision().clashingKeys)}. Fix config.json; it is not being rewritten.`);
|
|
1962
1995
|
}
|
|
1963
1996
|
|
|
1964
|
-
|
|
1997
|
+
// The new code starts as a child that binds the port as soon as this listener closes. This process
|
|
1998
|
+
// finishes its open streams and then waits as the parent, so a service manager keeps the pid it
|
|
1999
|
+
// started, and `switch off` still stops the gateway by the pid that listens.
|
|
2000
|
+
async function handOver() {
|
|
2001
|
+
server.close();
|
|
2002
|
+
server.closeIdleConnections?.();
|
|
2003
|
+
// The new gateway starts its own interceptor, so new interceptor code runs too.
|
|
2004
|
+
await stopRecordedBlindfold().catch(() => {});
|
|
2005
|
+
const args = process.argv.slice(2).filter(a => a !== '--autoupdate');
|
|
2006
|
+
const child = spawn(process.execPath, [fileURLToPath(import.meta.url), ...args], { stdio: 'inherit', windowsHide: true });
|
|
2007
|
+
child.on('exit', (code) => process.exit(code ?? 1));
|
|
2008
|
+
for (const sig of ['SIGINT', 'SIGTERM']) process.on(sig, () => { child.kill(sig); process.exit(0); });
|
|
2009
|
+
}
|
|
2010
|
+
|
|
2011
|
+
// --autoupdate is on the service command line: each logon starts the newest release.
|
|
2012
|
+
let handedOver = false;
|
|
2013
|
+
if (process.argv.includes('--autoupdate')) {
|
|
2014
|
+
try {
|
|
2015
|
+
const r = await applyUpdate({ logger: (msg) => console.log(`[llm-switcher:autoupdate] ${msg}`) });
|
|
2016
|
+
console.log(`[llm-switcher:autoupdate] ${r.reason}`);
|
|
2017
|
+
if (r.updated) {
|
|
2018
|
+
handedOver = true;
|
|
2019
|
+
await handOver();
|
|
2020
|
+
}
|
|
2021
|
+
} catch (err) {
|
|
2022
|
+
console.warn(`[llm-switcher:autoupdate] ${err.message} Starting the installed version.`);
|
|
2023
|
+
}
|
|
2024
|
+
}
|
|
2025
|
+
|
|
2026
|
+
if (!handedOver) server.listen(PORT, '127.0.0.1', () => {
|
|
1965
2027
|
// A service start or a restart on a new port finds env-codex.* already pointing at the interceptor.
|
|
1966
2028
|
const cfg = loadConfig();
|
|
1967
2029
|
if (getMigrationCollision()) {
|
package/service.mjs
CHANGED
|
@@ -14,13 +14,14 @@ export function serviceEnv(env = process.env) {
|
|
|
14
14
|
const xmlEscape = (s) => String(s).replace(/&/g, '&').replace(/</g, '<').replace(/>/g, '>').replace(/"/g, '"');
|
|
15
15
|
const systemdQuote = (s) => `"${String(s).replace(/(["\\])/g, '\\$1')}"`;
|
|
16
16
|
|
|
17
|
-
export function systemdUnit({ nodeBin, script, port, env = [] }) {
|
|
17
|
+
export function systemdUnit({ nodeBin, script, port, env = [], autoupdate = false }) {
|
|
18
|
+
const args = autoupdate ? `${systemdQuote(script)} --port ${port} --autoupdate` : `${systemdQuote(script)} --port ${port}`;
|
|
18
19
|
return `[Unit]
|
|
19
20
|
Description=LLM Switcher Local Gateway
|
|
20
21
|
After=network.target
|
|
21
22
|
|
|
22
23
|
[Service]
|
|
23
|
-
ExecStart=${systemdQuote(nodeBin)} ${
|
|
24
|
+
ExecStart=${systemdQuote(nodeBin)} ${args}
|
|
24
25
|
${env.map(([k, v]) => `Environment=${systemdQuote(`${k}=${v}`)}\n`).join('')}Restart=always
|
|
25
26
|
|
|
26
27
|
[Install]
|
|
@@ -28,7 +29,7 @@ WantedBy=default.target
|
|
|
28
29
|
`;
|
|
29
30
|
}
|
|
30
31
|
|
|
31
|
-
export function launchdPlist({ nodeBin, script, port, logPath, env = [] }) {
|
|
32
|
+
export function launchdPlist({ nodeBin, script, port, logPath, env = [], autoupdate = false }) {
|
|
32
33
|
const envBlock = env.length
|
|
33
34
|
? ` <key>EnvironmentVariables</key>
|
|
34
35
|
<dict>
|
|
@@ -48,7 +49,7 @@ ${env.map(([k, v]) => ` <key>${xmlEscape(k)}</key>\n <string>${xmlEscape(v
|
|
|
48
49
|
<string>${xmlEscape(script)}</string>
|
|
49
50
|
<string>--port</string>
|
|
50
51
|
<string>${port}</string>
|
|
51
|
-
</array>
|
|
52
|
+
${autoupdate ? ' <string>--autoupdate</string>\n' : ''} </array>
|
|
52
53
|
${envBlock} <key>StandardOutPath</key>
|
|
53
54
|
<string>${xmlEscape(logPath)}</string>
|
|
54
55
|
<key>StandardErrorPath</key>
|
|
@@ -65,7 +66,8 @@ ${envBlock} <key>StandardOutPath</key>
|
|
|
65
66
|
// Task Scheduler reads the command and its arguments from two elements, so no path goes through
|
|
66
67
|
// the quoting rules of a /TR command line. A task made with /TR also stops after 72 hours by
|
|
67
68
|
// default; PT0S removes that limit.
|
|
68
|
-
export function scheduledTaskXml({ nodeBin, script, port, userId }) {
|
|
69
|
+
export function scheduledTaskXml({ nodeBin, script, port, userId, autoupdate = false }) {
|
|
70
|
+
const args = autoupdate ? `"${script}" --port ${port} --autoupdate` : `"${script}" --port ${port}`;
|
|
69
71
|
return `<?xml version="1.0" encoding="UTF-16"?>
|
|
70
72
|
<Task version="1.2" xmlns="http://schemas.microsoft.com/windows/2004/02/mit/task">
|
|
71
73
|
<RegistrationInfo>
|
|
@@ -94,7 +96,7 @@ export function scheduledTaskXml({ nodeBin, script, port, userId }) {
|
|
|
94
96
|
<Actions Context="Author">
|
|
95
97
|
<Exec>
|
|
96
98
|
<Command>${xmlEscape(nodeBin)}</Command>
|
|
97
|
-
<Arguments>${xmlEscape(
|
|
99
|
+
<Arguments>${xmlEscape(args)}</Arguments>
|
|
98
100
|
</Exec>
|
|
99
101
|
</Actions>
|
|
100
102
|
</Task>
|
|
@@ -115,6 +117,10 @@ export function portFromServiceText(text) {
|
|
|
115
117
|
return Number.isInteger(n) && n > 0 && n <= 65535 ? n : null;
|
|
116
118
|
}
|
|
117
119
|
|
|
120
|
+
export function autoupdateFromServiceText(text) {
|
|
121
|
+
return /--autoupdate\b/.test(String(text));
|
|
122
|
+
}
|
|
123
|
+
|
|
118
124
|
/** Writes a service definition. Returns the backup path when it replaced different content. */
|
|
119
125
|
export function writeServiceFile(file, content) {
|
|
120
126
|
fs.mkdirSync(path.dirname(file), { recursive: true });
|
|
@@ -1,93 +1,93 @@
|
|
|
1
|
-
---
|
|
2
|
-
name: llm-switcher
|
|
3
|
-
description: Mandatory guidance for AI coding agents and token compression tools (Headroom, RTK, Ponytail, Repomix). Enforces routing all LLM traffic through the local LLM Switcher edge gateway (http://127.0.0.1:3456) instead of directly calling provider endpoints.
|
|
4
|
-
---
|
|
5
|
-
|
|
6
|
-
# LLM Switcher Edge Gateway Skill
|
|
7
|
-
|
|
8
|
-
This skill governs how AI coding agents (Claude Code, Codex, Cursor, Windsurf, Opencode) and prompt optimization/compression tools (Headroom, RTK, Ponytail, Repomix) must interact with LLM providers on this workstation.
|
|
9
|
-
|
|
10
|
-
## 1. The Core Directive
|
|
11
|
-
|
|
12
|
-
> **ALL OUTBOUND LLM REQUESTS MUST ROUTE THROUGH LLM SWITCHER (`http://127.0.0.1:3456`).**
|
|
13
|
-
>
|
|
14
|
-
> Never configure any tool, script, sub-agent, or token compression proxy to call external provider endpoints directly (such as `https://api.anthropic.com`, `https://api.openai.com`, or `https://generativelanguage.googleapis.com`).
|
|
15
|
-
>
|
|
16
|
-
> If a token compression tool (Headroom, RTK, Ponytail) is used, it MUST be positioned as an **inner processing layer**, and its **upstream destination MUST be configured to point to LLM Switcher** at `http://127.0.0.1:3456`.
|
|
17
|
-
|
|
18
|
-
```
|
|
19
|
-
[Client CLI / Agent]
|
|
20
|
-
│
|
|
21
|
-
▼
|
|
22
|
-
[Token Compressor: RTK / Headroom / Ponytail] <-- Inner Layer
|
|
23
|
-
│
|
|
24
|
-
▼ (MANDATORY: Upstream MUST point to http://127.0.0.1:3456)
|
|
25
|
-
[LLM Switcher Edge Gateway (:3456)] <-- Outermost Gatekeeper
|
|
26
|
-
│
|
|
27
|
-
▼ (Heals schemas, maps 1M context, extracts thinking, converts protocol)
|
|
28
|
-
[Internet / intact (Recommended) / 9Router / Upstream LLM]
|
|
29
|
-
```
|
|
30
|
-
|
|
31
|
-
## 2. Why This is Mandatory
|
|
32
|
-
|
|
33
|
-
Direct outbound calls from token compressors will break the developer environment for several reasons:
|
|
34
|
-
|
|
35
|
-
1. **Schema Breakage & The Healer Engine:**
|
|
36
|
-
Token optimizers frequently prune conversation history to save tokens. In doing so, they often delete an `assistant` turn containing a `tool_use`, while leaving the subsequent `user` turn containing a `tool_result`.
|
|
37
|
-
- Anthropic and OpenAI APIs strictly reject this with `HTTP 400 (orphaned tool_use_id)`.
|
|
38
|
-
- **LLM Switcher** intercepts these malformed requests and runs its **Healer Engine**, automatically converting orphaned results into safe context text blocks and merging consecutive same-role turns so upstream providers accept the payload cleanly.
|
|
39
|
-
2. **Thinking / Reasoning Protection:**
|
|
40
|
-
Many token compressors naively strip `thinking` parameters or reasoning budgets to reduce token count.
|
|
41
|
-
- When routed through LLM Switcher, the gateway detects reasoning models (such as `ag/claude-opus-4-6-thinking` or `ag/gemini-3.8-flash`) and **automatically restores thinking parameters**, ensuring the model retains its full reasoning capabilities.
|
|
42
|
-
3. **Official Context Windows & Dynamic Model Discovery:**
|
|
43
|
-
Context windows follow the model you select without mutating permanent settings. The gateway auto-discovers official models from Anthropic and OpenAI, refreshing mappings on tool version upgrades (`switch models`).
|
|
44
|
-
4. **Server Gateway Synergy (intact / 9Router):**
|
|
45
|
-
LLM Switcher pairs directly with server-side gateways like **intact** (strongly recommended for credential proxying, account pooling, and contract lab self-healing) or 9Router, handling retries and quotas far better than any local tool.
|
|
46
|
-
|
|
47
|
-
## 3. Configuring Token Compression Tools
|
|
48
|
-
|
|
49
|
-
When setting up or assisting the user with token compression tools, apply these exact configurations:
|
|
50
|
-
|
|
51
|
-
### A. Headroom (Prompt Compression Proxy)
|
|
52
|
-
When configuring Headroom's upstream endpoint:
|
|
53
|
-
- In Headroom config (`headroom.yaml` or environment variables):
|
|
54
|
-
```bash
|
|
55
|
-
ANTHROPIC_BASE_URL="http://127.0.0.1:3456"
|
|
56
|
-
OPENAI_BASE_URL="http://127.0.0.1:3456/v1"
|
|
57
|
-
```
|
|
58
|
-
- Point your CLI to Headroom's listening port (e.g. `http://127.0.0.1:8787`). Headroom compresses the prompt and forwards it to LLM Switcher on port `3456`.
|
|
59
|
-
|
|
60
|
-
### B. RTK (Rust Token Killer / Repomix)
|
|
61
|
-
- In RTK configuration or execution flags:
|
|
62
|
-
```bash
|
|
63
|
-
# Ensure RTK's target upstream proxy is LLM Switcher
|
|
64
|
-
export UPSTREAM_URL="http://127.0.0.1:3456"
|
|
65
|
-
```
|
|
66
|
-
|
|
67
|
-
### C. Ponytail / Custom Proxies
|
|
68
|
-
- Always set the proxy's upstream host to:
|
|
69
|
-
`http://127.0.0.1:3456` (or `http://127.0.0.1:3456/v1` for OpenAI-compatible tools).
|
|
70
|
-
|
|
71
|
-
### D. Sub-Agents & Spawned Terminal Commands
|
|
72
|
-
When an agent spawns sub-processes that call LLM CLIs:
|
|
73
|
-
- **Never advise sourcing `env.sh` or `env.cmd`.** Those files carry no proxy variables any
|
|
74
|
-
more, so sourcing one changes nothing — and a stale variable an older version once wrote
|
|
75
|
-
would still point the tool at a port where nothing listens.
|
|
76
|
-
- Run `switch claude <profile>` or `switch codex <profile>` instead. The shims in
|
|
77
|
-
`~/.llm-switcher/bin` inject the proxy variables into the tool process itself and clear
|
|
78
|
-
anything stale first.
|
|
79
|
-
- `claude` and `codex` on `PATH` already are those shims: invoke them normally, and never
|
|
80
|
-
wrap them in another script.
|
|
81
|
-
|
|
82
|
-
## 4. Operational Rules for AI Agents
|
|
83
|
-
|
|
84
|
-
1. **NEVER edit `~/.claude/settings.json` directly.**
|
|
85
|
-
LLM Switcher uses zero-mutation launcher flags (`active.flag`). Writing custom base URLs or auth tokens into `settings.json` triggers warning banners in Claude Code and risks breaking the user's official subscription. Use `switch` CLI commands instead.
|
|
86
|
-
2. **Check Gateway Health Before Complex Operations:**
|
|
87
|
-
Run `switch status` or call the `switcher_audit` MCP tool to confirm:
|
|
88
|
-
- LLM Switcher is active on port `3456`.
|
|
89
|
-
- The active profile matches the intended CLI target (Claude Code, Codex, or OpenAI).
|
|
90
|
-
3. **Verify Routing When Errors Occur:**
|
|
91
|
-
If a tool fails with `HTTP 400`, `HTTP 502`, or connection errors:
|
|
92
|
-
- Run `switch doctor` to audit port collisions and environment variables.
|
|
93
|
-
- Inspect recent request logs via `http://127.0.0.1:3456/ui` (Tab 4: Live Inspector) to see if an intermediary tool mangled the payload.
|
|
1
|
+
---
|
|
2
|
+
name: llm-switcher
|
|
3
|
+
description: Mandatory guidance for AI coding agents and token compression tools (Headroom, RTK, Ponytail, Repomix). Enforces routing all LLM traffic through the local LLM Switcher edge gateway (http://127.0.0.1:3456) instead of directly calling provider endpoints.
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# LLM Switcher Edge Gateway Skill
|
|
7
|
+
|
|
8
|
+
This skill governs how AI coding agents (Claude Code, Codex, Cursor, Windsurf, Opencode) and prompt optimization/compression tools (Headroom, RTK, Ponytail, Repomix) must interact with LLM providers on this workstation.
|
|
9
|
+
|
|
10
|
+
## 1. The Core Directive
|
|
11
|
+
|
|
12
|
+
> **ALL OUTBOUND LLM REQUESTS MUST ROUTE THROUGH LLM SWITCHER (`http://127.0.0.1:3456`).**
|
|
13
|
+
>
|
|
14
|
+
> Never configure any tool, script, sub-agent, or token compression proxy to call external provider endpoints directly (such as `https://api.anthropic.com`, `https://api.openai.com`, or `https://generativelanguage.googleapis.com`).
|
|
15
|
+
>
|
|
16
|
+
> If a token compression tool (Headroom, RTK, Ponytail) is used, it MUST be positioned as an **inner processing layer**, and its **upstream destination MUST be configured to point to LLM Switcher** at `http://127.0.0.1:3456`.
|
|
17
|
+
|
|
18
|
+
```
|
|
19
|
+
[Client CLI / Agent]
|
|
20
|
+
│
|
|
21
|
+
▼
|
|
22
|
+
[Token Compressor: RTK / Headroom / Ponytail] <-- Inner Layer
|
|
23
|
+
│
|
|
24
|
+
▼ (MANDATORY: Upstream MUST point to http://127.0.0.1:3456)
|
|
25
|
+
[LLM Switcher Edge Gateway (:3456)] <-- Outermost Gatekeeper
|
|
26
|
+
│
|
|
27
|
+
▼ (Heals schemas, maps 1M context, extracts thinking, converts protocol)
|
|
28
|
+
[Internet / intact (Recommended) / 9Router / Upstream LLM]
|
|
29
|
+
```
|
|
30
|
+
|
|
31
|
+
## 2. Why This is Mandatory
|
|
32
|
+
|
|
33
|
+
Direct outbound calls from token compressors will break the developer environment for several reasons:
|
|
34
|
+
|
|
35
|
+
1. **Schema Breakage & The Healer Engine:**
|
|
36
|
+
Token optimizers frequently prune conversation history to save tokens. In doing so, they often delete an `assistant` turn containing a `tool_use`, while leaving the subsequent `user` turn containing a `tool_result`.
|
|
37
|
+
- Anthropic and OpenAI APIs strictly reject this with `HTTP 400 (orphaned tool_use_id)`.
|
|
38
|
+
- **LLM Switcher** intercepts these malformed requests and runs its **Healer Engine**, automatically converting orphaned results into safe context text blocks and merging consecutive same-role turns so upstream providers accept the payload cleanly.
|
|
39
|
+
2. **Thinking / Reasoning Protection:**
|
|
40
|
+
Many token compressors naively strip `thinking` parameters or reasoning budgets to reduce token count.
|
|
41
|
+
- When routed through LLM Switcher, the gateway detects reasoning models (such as `ag/claude-opus-4-6-thinking` or `ag/gemini-3.8-flash`) and **automatically restores thinking parameters**, ensuring the model retains its full reasoning capabilities.
|
|
42
|
+
3. **Official Context Windows & Dynamic Model Discovery:**
|
|
43
|
+
Context windows follow the model you select without mutating permanent settings. The gateway auto-discovers official models from Anthropic and OpenAI, refreshing mappings on tool version upgrades (`switch models`).
|
|
44
|
+
4. **Server Gateway Synergy (intact / 9Router):**
|
|
45
|
+
LLM Switcher pairs directly with server-side gateways like **intact** (strongly recommended for credential proxying, account pooling, and contract lab self-healing) or 9Router, handling retries and quotas far better than any local tool.
|
|
46
|
+
|
|
47
|
+
## 3. Configuring Token Compression Tools
|
|
48
|
+
|
|
49
|
+
When setting up or assisting the user with token compression tools, apply these exact configurations:
|
|
50
|
+
|
|
51
|
+
### A. Headroom (Prompt Compression Proxy)
|
|
52
|
+
When configuring Headroom's upstream endpoint:
|
|
53
|
+
- In Headroom config (`headroom.yaml` or environment variables):
|
|
54
|
+
```bash
|
|
55
|
+
ANTHROPIC_BASE_URL="http://127.0.0.1:3456"
|
|
56
|
+
OPENAI_BASE_URL="http://127.0.0.1:3456/v1"
|
|
57
|
+
```
|
|
58
|
+
- Point your CLI to Headroom's listening port (e.g. `http://127.0.0.1:8787`). Headroom compresses the prompt and forwards it to LLM Switcher on port `3456`.
|
|
59
|
+
|
|
60
|
+
### B. RTK (Rust Token Killer / Repomix)
|
|
61
|
+
- In RTK configuration or execution flags:
|
|
62
|
+
```bash
|
|
63
|
+
# Ensure RTK's target upstream proxy is LLM Switcher
|
|
64
|
+
export UPSTREAM_URL="http://127.0.0.1:3456"
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
### C. Ponytail / Custom Proxies
|
|
68
|
+
- Always set the proxy's upstream host to:
|
|
69
|
+
`http://127.0.0.1:3456` (or `http://127.0.0.1:3456/v1` for OpenAI-compatible tools).
|
|
70
|
+
|
|
71
|
+
### D. Sub-Agents & Spawned Terminal Commands
|
|
72
|
+
When an agent spawns sub-processes that call LLM CLIs:
|
|
73
|
+
- **Never advise sourcing `env.sh` or `env.cmd`.** Those files carry no proxy variables any
|
|
74
|
+
more, so sourcing one changes nothing — and a stale variable an older version once wrote
|
|
75
|
+
would still point the tool at a port where nothing listens.
|
|
76
|
+
- Run `switch claude <profile>` or `switch codex <profile>` instead. The shims in
|
|
77
|
+
`~/.llm-switcher/bin` inject the proxy variables into the tool process itself and clear
|
|
78
|
+
anything stale first.
|
|
79
|
+
- `claude` and `codex` on `PATH` already are those shims: invoke them normally, and never
|
|
80
|
+
wrap them in another script.
|
|
81
|
+
|
|
82
|
+
## 4. Operational Rules for AI Agents
|
|
83
|
+
|
|
84
|
+
1. **NEVER edit `~/.claude/settings.json` directly.**
|
|
85
|
+
LLM Switcher uses zero-mutation launcher flags (`active.flag`). Writing custom base URLs or auth tokens into `settings.json` triggers warning banners in Claude Code and risks breaking the user's official subscription. Use `switch` CLI commands instead.
|
|
86
|
+
2. **Check Gateway Health Before Complex Operations:**
|
|
87
|
+
Run `switch status` or call the `switcher_audit` MCP tool to confirm:
|
|
88
|
+
- LLM Switcher is active on port `3456`.
|
|
89
|
+
- The active profile matches the intended CLI target (Claude Code, Codex, or OpenAI).
|
|
90
|
+
3. **Verify Routing When Errors Occur:**
|
|
91
|
+
If a tool fails with `HTTP 400`, `HTTP 502`, or connection errors:
|
|
92
|
+
- Run `switch doctor` to audit port collisions and environment variables.
|
|
93
|
+
- Inspect recent request logs via `http://127.0.0.1:3456/ui` (Tab 4: Live Inspector) to see if an intermediary tool mangled the payload.
|
package/switch
CHANGED
|
File without changes
|
package/switch.cmd
CHANGED
|
@@ -1,2 +1,2 @@
|
|
|
1
|
-
@echo off
|
|
2
|
-
node "%~dp0switch.mjs" %*
|
|
1
|
+
@echo off
|
|
2
|
+
node "%~dp0switch.mjs" %*
|
package/switch.mjs
CHANGED
|
@@ -20,7 +20,7 @@ import {
|
|
|
20
20
|
suggestedRcFiles, auditRunningProcesses
|
|
21
21
|
} from './shim.mjs';
|
|
22
22
|
import {
|
|
23
|
-
serviceEnv, systemdUnit, launchdPlist, scheduledTaskXml, decodeConsoleText, portFromServiceText, writeServiceFile
|
|
23
|
+
serviceEnv, systemdUnit, launchdPlist, scheduledTaskXml, decodeConsoleText, portFromServiceText, autoupdateFromServiceText, writeServiceFile
|
|
24
24
|
} from './service.mjs';
|
|
25
25
|
|
|
26
26
|
const proxyScript = path.join(ROOT_DIR, 'proxy.mjs');
|
|
@@ -272,6 +272,9 @@ async function changePort(newPortStr) {
|
|
|
272
272
|
const target = await probeGateway(p);
|
|
273
273
|
if (isHeld(target)) refuseForeignPort(p, 'this switcher', target);
|
|
274
274
|
const svc = installedService();
|
|
275
|
+
// The unit is rewritten below. It keeps the choice that `service install --no-autoupdate` made.
|
|
276
|
+
const svcText = svc ? serviceText(svc) : null;
|
|
277
|
+
const autoupdate = svcText === null || autoupdateFromServiceText(svcText);
|
|
275
278
|
const wasRunning = await checkProxyRunning(oldPort);
|
|
276
279
|
// Stop the old gateway even when a service is installed: it may be a copy started outside the unit.
|
|
277
280
|
if (wasRunning) {
|
|
@@ -301,7 +304,7 @@ async function changePort(newPortStr) {
|
|
|
301
304
|
} catch (err) {
|
|
302
305
|
console.error(`[Error] Failed to restore config.json during rollback: ${err.message}`);
|
|
303
306
|
}
|
|
304
|
-
if (svc) installService(oldPort);
|
|
307
|
+
if (svc) installService(oldPort, { autoupdate });
|
|
305
308
|
else if (wasRunning) startProxyBackground(oldPort);
|
|
306
309
|
let up = !svc && !wasRunning;
|
|
307
310
|
for (let i = 0; i < 20 && !up; i++) { await sleep(250); up = await checkProxyRunning(oldPort); }
|
|
@@ -311,7 +314,7 @@ async function changePort(newPortStr) {
|
|
|
311
314
|
if (svc) {
|
|
312
315
|
// The unit fixes the port on its command line, so it is rewritten and restarted, not fought.
|
|
313
316
|
console.log(`Reinstalling the ${svc} service on port ${p}...`);
|
|
314
|
-
if (!installService(p)) await rollBack(`The ${svc} service could not be installed for port ${p}.`);
|
|
317
|
+
if (!installService(p, { autoupdate })) await rollBack(`The ${svc} service could not be installed for port ${p}.`);
|
|
315
318
|
let up = false;
|
|
316
319
|
for (let i = 0; i < 20 && !up; i++) { await sleep(250); up = await checkProxyRunning(p); }
|
|
317
320
|
if (!up) await rollBack(`The ${svc} service did not come up on port ${p}. See ${proxyLogPath}.`);
|
|
@@ -669,16 +672,22 @@ function serviceStart(kind) {
|
|
|
669
672
|
} catch {}
|
|
670
673
|
}
|
|
671
674
|
|
|
672
|
-
/** The
|
|
673
|
-
function
|
|
675
|
+
/** The installed service definition, or null when it cannot be read. */
|
|
676
|
+
function serviceText(kind) {
|
|
674
677
|
try {
|
|
675
|
-
if (kind === 'systemd') return
|
|
676
|
-
if (kind === 'launchd') return
|
|
677
|
-
if (kind === 'schtasks') return
|
|
678
|
+
if (kind === 'systemd') return fs.readFileSync(SYSTEMD_UNIT, 'utf8');
|
|
679
|
+
if (kind === 'launchd') return fs.readFileSync(LAUNCHD_PLIST, 'utf8');
|
|
680
|
+
if (kind === 'schtasks') return decodeConsoleText(execFileSync('schtasks', ['/Query', '/TN', 'LLMSwitcher', '/XML']));
|
|
678
681
|
} catch {}
|
|
679
682
|
return null;
|
|
680
683
|
}
|
|
681
684
|
|
|
685
|
+
/** The port on the service's command line, or null when it cannot be read. */
|
|
686
|
+
function servicePort(kind) {
|
|
687
|
+
const text = serviceText(kind);
|
|
688
|
+
return text === null ? null : portFromServiceText(text);
|
|
689
|
+
}
|
|
690
|
+
|
|
682
691
|
function reportBackup(backup) {
|
|
683
692
|
if (backup) console.log(`[Service] The previous definition differed and is kept as ${backup}.`);
|
|
684
693
|
}
|
|
@@ -693,7 +702,7 @@ function serviceStop(kind) {
|
|
|
693
702
|
}
|
|
694
703
|
|
|
695
704
|
// Writes the service for `port` and (re)starts it, so a running unit picks up the new command line.
|
|
696
|
-
function installService(port) {
|
|
705
|
+
function installService(port, { autoupdate = true } = {}) {
|
|
697
706
|
const nodeBin = process.execPath;
|
|
698
707
|
const env = serviceEnv();
|
|
699
708
|
if (process.platform === 'win32') {
|
|
@@ -702,12 +711,12 @@ function installService(port) {
|
|
|
702
711
|
const userId = process.env.USERDOMAIN && process.env.USERNAME ? `${process.env.USERDOMAIN}\\${process.env.USERNAME}` : os.userInfo().username;
|
|
703
712
|
const xmlPath = path.join(dir, 'task.xml');
|
|
704
713
|
// Task Scheduler reads the XML as UTF-16, the encoding it declares.
|
|
705
|
-
fs.writeFileSync(xmlPath, Buffer.concat([Buffer.from([0xff, 0xfe]), Buffer.from(scheduledTaskXml({ nodeBin, script: proxyScript, port, userId }), 'utf16le')]));
|
|
714
|
+
fs.writeFileSync(xmlPath, Buffer.concat([Buffer.from([0xff, 0xfe]), Buffer.from(scheduledTaskXml({ nodeBin, script: proxyScript, port, userId, autoupdate }), 'utf16le')]));
|
|
706
715
|
execFileSync('schtasks', ['/Create', '/TN', 'LLMSwitcher', '/XML', xmlPath, '/F'], { stdio: 'inherit' });
|
|
707
716
|
try { execFileSync('schtasks', ['/End', '/TN', 'LLMSwitcher'], { stdio: 'ignore' }); } catch {}
|
|
708
717
|
execFileSync('schtasks', ['/Run', '/TN', 'LLMSwitcher'], { stdio: 'ignore' });
|
|
709
718
|
if (env.length) console.log(`[WARN] The scheduled task does not receive ${env.map(([k]) => k).join(', ')}. Set them as User environment variables.`);
|
|
710
|
-
console.log(
|
|
719
|
+
console.log(`[SUCCESS] Installed and started Windows Scheduled Task "LLMSwitcher" (${autoupdate ? 'auto-updates on logon and starts' : 'auto-starts on logon'}).`);
|
|
711
720
|
return true;
|
|
712
721
|
} catch (err) {
|
|
713
722
|
console.error('[Error] Failed to register the scheduled task:', err.message);
|
|
@@ -718,7 +727,7 @@ function installService(port) {
|
|
|
718
727
|
}
|
|
719
728
|
if (process.platform === 'darwin') {
|
|
720
729
|
try {
|
|
721
|
-
reportBackup(writeServiceFile(LAUNCHD_PLIST, launchdPlist({ nodeBin, script: proxyScript, port, logPath: proxyLogPath, env })));
|
|
730
|
+
reportBackup(writeServiceFile(LAUNCHD_PLIST, launchdPlist({ nodeBin, script: proxyScript, port, logPath: proxyLogPath, env, autoupdate })));
|
|
722
731
|
try { execFileSync('launchctl', ['unload', LAUNCHD_PLIST], { stdio: 'ignore' }); } catch {}
|
|
723
732
|
execFileSync('launchctl', ['load', LAUNCHD_PLIST], { stdio: 'inherit' });
|
|
724
733
|
console.log('[SUCCESS] Installed and started macOS launchd service.');
|
|
@@ -729,7 +738,7 @@ function installService(port) {
|
|
|
729
738
|
}
|
|
730
739
|
}
|
|
731
740
|
try {
|
|
732
|
-
reportBackup(writeServiceFile(SYSTEMD_UNIT, systemdUnit({ nodeBin, script: proxyScript, port, env })));
|
|
741
|
+
reportBackup(writeServiceFile(SYSTEMD_UNIT, systemdUnit({ nodeBin, script: proxyScript, port, env, autoupdate })));
|
|
733
742
|
systemctlUser(['daemon-reload'], { stdio: 'inherit' });
|
|
734
743
|
systemctlUser(['enable', 'llm-switcher'], { stdio: 'inherit' });
|
|
735
744
|
systemctlUser(['restart', 'llm-switcher'], { stdio: 'inherit' });
|
|
@@ -745,7 +754,8 @@ async function manageService(action) {
|
|
|
745
754
|
const port = getTargetPort();
|
|
746
755
|
|
|
747
756
|
if (action === 'install') {
|
|
748
|
-
|
|
757
|
+
const autoupdate = !process.argv.includes('--no-autoupdate');
|
|
758
|
+
if (!installService(port, { autoupdate })) process.exit(1);
|
|
749
759
|
return;
|
|
750
760
|
}
|
|
751
761
|
|
|
@@ -1106,6 +1116,78 @@ async function showVersion() {
|
|
|
1106
1116
|
else console.log('This is the latest version.');
|
|
1107
1117
|
}
|
|
1108
1118
|
|
|
1119
|
+
async function gatewayPid(port) {
|
|
1120
|
+
try {
|
|
1121
|
+
const r = await fetch(`http://127.0.0.1:${port}/health?challenge=update`, { signal: AbortSignal.timeout(2000) });
|
|
1122
|
+
return (await r.json()).pid ?? null;
|
|
1123
|
+
} catch {
|
|
1124
|
+
return null;
|
|
1125
|
+
}
|
|
1126
|
+
}
|
|
1127
|
+
|
|
1128
|
+
// A running gateway updates itself and hands over to the new code, so the CLI and the dashboard
|
|
1129
|
+
// share one update path. With no gateway up, the files are updated here and the next start runs them.
|
|
1130
|
+
async function runUpdate() {
|
|
1131
|
+
const port = getTargetPort();
|
|
1132
|
+
if (!(await checkProxyRunning(port))) {
|
|
1133
|
+
const { applyUpdate } = await import('./update.mjs');
|
|
1134
|
+
try {
|
|
1135
|
+
const r = await applyUpdate({ logger: (text) => console.log(` ${text}`) });
|
|
1136
|
+
console.log(r.updated ? `[SUCCESS] Updated v${r.from} -> v${r.to}.` : r.reason);
|
|
1137
|
+
} catch (err) {
|
|
1138
|
+
console.error(`[Error] ${err.message}`);
|
|
1139
|
+
process.exit(1);
|
|
1140
|
+
}
|
|
1141
|
+
return;
|
|
1142
|
+
}
|
|
1143
|
+
const oldPid = await gatewayPid(port);
|
|
1144
|
+
const res = await fetch(`http://127.0.0.1:${port}/api/update`, {
|
|
1145
|
+
method: 'POST',
|
|
1146
|
+
headers: { 'Content-Type': 'application/json', 'x-llm-switcher-token': readAdminToken() || '' },
|
|
1147
|
+
body: '{}'
|
|
1148
|
+
});
|
|
1149
|
+
if (!res.ok) {
|
|
1150
|
+
console.error(`[Error] ${(await res.json().catch(() => ({}))).error || `HTTP ${res.status}`}`);
|
|
1151
|
+
process.exit(1);
|
|
1152
|
+
}
|
|
1153
|
+
let result = null;
|
|
1154
|
+
let failed = null;
|
|
1155
|
+
let buf = '';
|
|
1156
|
+
const decoder = new TextDecoder();
|
|
1157
|
+
for await (const chunk of res.body) {
|
|
1158
|
+
buf += decoder.decode(chunk, { stream: true });
|
|
1159
|
+
let i;
|
|
1160
|
+
while ((i = buf.indexOf('\n\n')) !== -1) {
|
|
1161
|
+
const block = buf.slice(0, i);
|
|
1162
|
+
buf = buf.slice(i + 2);
|
|
1163
|
+
const event = /^event: (.*)$/m.exec(block)?.[1];
|
|
1164
|
+
const data = JSON.parse(/^data: (.*)$/m.exec(block)?.[1] || 'null');
|
|
1165
|
+
if (event === 'log') console.log(` ${show(data.text)}`);
|
|
1166
|
+
else if (event === 'result') result = data;
|
|
1167
|
+
else if (event === 'error') failed = data.message;
|
|
1168
|
+
}
|
|
1169
|
+
}
|
|
1170
|
+
if (failed || !result) {
|
|
1171
|
+
console.error(`[Error] ${show(failed || 'The gateway closed the update stream without a result.')}`);
|
|
1172
|
+
process.exit(1);
|
|
1173
|
+
}
|
|
1174
|
+
if (!result.updated) {
|
|
1175
|
+
console.log(show(result.reason));
|
|
1176
|
+
return;
|
|
1177
|
+
}
|
|
1178
|
+
console.log(`Updated v${show(result.from)} -> v${show(result.to)}. Waiting for the new gateway...`);
|
|
1179
|
+
for (let i = 0; i < 60; i++) {
|
|
1180
|
+
await sleep(250);
|
|
1181
|
+
const pid = await gatewayPid(port);
|
|
1182
|
+
if (pid && pid !== oldPid && (await checkProxyRunning(port))) {
|
|
1183
|
+
console.log(`[SUCCESS] The gateway on port ${port} runs v${show(result.to)}.`);
|
|
1184
|
+
return;
|
|
1185
|
+
}
|
|
1186
|
+
}
|
|
1187
|
+
console.error(`[Error] The new gateway did not answer on port ${port}. See ${proxyLogPath}.`);
|
|
1188
|
+
process.exit(1);
|
|
1189
|
+
}
|
|
1190
|
+
|
|
1109
1191
|
async function showModels(target) {
|
|
1110
1192
|
const { syncLocalCatalog, refreshCatalog } = await import('./catalog.mjs');
|
|
1111
1193
|
if (optionValue('--refresh') || process.argv.includes('--refresh')) {
|
|
@@ -1169,6 +1251,8 @@ if (cmd === 'off' || cmd === 'stop') {
|
|
|
1169
1251
|
await showModels(subArg);
|
|
1170
1252
|
} else if (cmd === 'status' || cmd === 'st') {
|
|
1171
1253
|
await showStatus();
|
|
1254
|
+
} else if (cmd === 'update' || cmd === 'upgrade') {
|
|
1255
|
+
await runUpdate();
|
|
1172
1256
|
} else if (cmd === 'on' || cmd === 'start') {
|
|
1173
1257
|
await turnOn(subArg);
|
|
1174
1258
|
} else if (cmd && findProfileKey(config, rawCmd)) {
|
|
@@ -1179,12 +1263,13 @@ if (cmd === 'off' || cmd === 'stop') {
|
|
|
1179
1263
|
console.log(' switch status # Show multi-CLI active status');
|
|
1180
1264
|
console.log(' switch version # Show the version and check for a newer one');
|
|
1181
1265
|
console.log(' switch doctor # Audit environment, settings & routing');
|
|
1266
|
+
console.log(' switch update # Install the newest release and restart the gateway');
|
|
1182
1267
|
console.log(' switch on [profile] # Start gateway & activate profile for all compatible targets');
|
|
1183
1268
|
console.log(' switch <profile> # Activate profile for all compatible targets');
|
|
1184
1269
|
console.log(' switch claude <profile> # Set active profile for Claude Code');
|
|
1185
1270
|
console.log(' switch codex <profile> # Set active profile for Codex');
|
|
1186
1271
|
console.log(' switch port <number> # Change gateway port');
|
|
1187
|
-
console.log(' switch service install # Install OS background autostart service');
|
|
1272
|
+
console.log(' switch service install # Install OS background autostart service (auto-updates on logon)');
|
|
1188
1273
|
console.log(' switch service uninstall # Uninstall background autostart service');
|
|
1189
1274
|
console.log(' switch shim install # Auto-inject env into resumed sessions (claude --resume)');
|
|
1190
1275
|
console.log(' switch shim status # Check shims + detect sessions bypassing the gateway');
|
|
@@ -1251,6 +1251,17 @@ test('Codex WS: conversation.item.create is echoed and response.cancel stops the
|
|
|
1251
1251
|
ws.socket.destroy();
|
|
1252
1252
|
});
|
|
1253
1253
|
|
|
1254
|
+
test('Test connection on a Bifrost model checks the model and the key, and sends no fake request', async () => {
|
|
1255
|
+
const before = received.length;
|
|
1256
|
+
const r = await post('/api/test-upstream', { baseURL: `http://127.0.0.1:${upstreamPort}/intact/v1`, apiKey: 'k', model: 'claude/claude-opus-5', mode: 'convert' });
|
|
1257
|
+
const j = await r.json();
|
|
1258
|
+
assert.equal(j.ok, true);
|
|
1259
|
+
assert.equal(j.outFormat, 'bifrost');
|
|
1260
|
+
assert.match(j.sample, /claude-cli\//);
|
|
1261
|
+
const sent = received.slice(before).map(x => x.url);
|
|
1262
|
+
assert.ok(!sent.some(u => u.includes('/messages') || u.includes('/chat/completions')), `sent ${sent}`);
|
|
1263
|
+
});
|
|
1264
|
+
|
|
1254
1265
|
test('Admin test-upstream reports latency and a sample from the upstream', async () => {
|
|
1255
1266
|
const r = await post('/api/test-upstream', { baseURL: `http://127.0.0.1:${upstreamPort}/chat/v1`, apiKey: 'k', model: 'm', mode: 'convert' });
|
|
1256
1267
|
assert.equal(r.status, 200);
|