llm-switcher 1.2.0 → 1.2.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +12 -0
- package/classifier.mjs +238 -0
- package/docs/TOKEN-OPTIMIZER-INTEROP.md +110 -110
- package/docs/response-matrix.json +1130 -1130
- package/formats.mjs +30 -1
- package/icons/antigravity.png +0 -0
- package/icons/claude.png +0 -0
- package/icons/codex.png +0 -0
- package/icons/deepseek.png +0 -0
- package/icons/gemini.png +0 -0
- package/icons/github.png +0 -0
- package/icons/groq.png +0 -0
- package/icons/intact.svg +1 -0
- package/icons/ollama.png +0 -0
- package/icons/openai.png +0 -0
- package/icons/openrouter.png +0 -0
- package/icons/qwen.png +0 -0
- package/icons/vertex.png +0 -0
- package/package.json +1 -1
- package/proxy.mjs +27 -15
- package/skills/llm-switcher/SKILL.md +93 -93
- package/switch +0 -0
- package/tests/classifier.test.mjs +210 -0
- package/tests/formats.test.mjs +30 -0
- package/tests/helpers.mjs +24 -24
- package/tests/live-optimizer-interop.mjs +205 -205
- package/ui.html +1700 -1623
package/formats.mjs
CHANGED
|
@@ -19,6 +19,7 @@
|
|
|
19
19
|
// params: { maxTokens, temperature, topP, topK, stop[],
|
|
20
20
|
// presencePenalty?, frequencyPenalty? },
|
|
21
21
|
// thinking: { type:'enabled'|'disabled'|'adaptive', budget?, effort? } | null,
|
|
22
|
+
// responseFormat: { type:'json_object' } | { type:'json_schema', name?, schema, strict? } | null,
|
|
22
23
|
// stream: bool
|
|
23
24
|
// }
|
|
24
25
|
// ============================================================
|
|
@@ -327,10 +328,23 @@ function baseIR() {
|
|
|
327
328
|
return {
|
|
328
329
|
model: '', system: '', messages: [], tools: [], toolChoice: null,
|
|
329
330
|
params: { maxTokens: null, temperature: null, topP: null, topK: null, stop: [] },
|
|
330
|
-
thinking: null, stream: false
|
|
331
|
+
thinking: null, responseFormat: null, stream: false
|
|
331
332
|
};
|
|
332
333
|
}
|
|
333
334
|
|
|
335
|
+
// Structured output (Anthropic output_config.format, Responses text.format) -> IR responseFormat.
|
|
336
|
+
// `text` asks for nothing.
|
|
337
|
+
function responseFormatIR(f) {
|
|
338
|
+
if (!f || typeof f !== 'object') return null;
|
|
339
|
+
if (f.type === 'json_object') return { type: 'json_object' };
|
|
340
|
+
if (f.type !== 'json_schema') return null;
|
|
341
|
+
if (!f.schema || typeof f.schema !== 'object') return { type: 'json_object' };
|
|
342
|
+
const out = { type: 'json_schema', schema: f.schema };
|
|
343
|
+
if (typeof f.name === 'string' && f.name) out.name = f.name;
|
|
344
|
+
if (typeof f.strict === 'boolean') out.strict = f.strict;
|
|
345
|
+
return out;
|
|
346
|
+
}
|
|
347
|
+
|
|
334
348
|
// Anthropic Messages API -> IR (logic ported from the old transformAnthropicToOpenAI).
|
|
335
349
|
function anthropicToIR(payload) {
|
|
336
350
|
const ir = baseIR();
|
|
@@ -423,6 +437,7 @@ function anthropicToIR(payload) {
|
|
|
423
437
|
if (Array.isArray(payload.stop_sequences)) ir.params.stop = stopList(payload.stop_sequences);
|
|
424
438
|
|
|
425
439
|
ir.thinking = thinkingFromAnthropicParam(payload.thinking, payload.output_config?.effort);
|
|
440
|
+
ir.responseFormat = responseFormatIR(payload.output_config?.format || payload.output_format);
|
|
426
441
|
return ir;
|
|
427
442
|
}
|
|
428
443
|
|
|
@@ -644,6 +659,7 @@ function responsesToIR(payload) {
|
|
|
644
659
|
if (payload.reasoning && typeof payload.reasoning === 'object') {
|
|
645
660
|
ir.thinking = thinkingFromReasoningParam(payload.reasoning);
|
|
646
661
|
}
|
|
662
|
+
ir.responseFormat = responseFormatIR(payload.text?.format);
|
|
647
663
|
return ir;
|
|
648
664
|
}
|
|
649
665
|
|
|
@@ -822,6 +838,13 @@ function irToChatBody(ir, model, opts = {}) {
|
|
|
822
838
|
if (typeof ir.params.frequencyPenalty === 'number') body.frequency_penalty = ir.params.frequencyPenalty;
|
|
823
839
|
if (ir.params.stop.length) body.stop = ir.params.stop;
|
|
824
840
|
if (ir.params.parallelToolCalls === false && body.tools) body.parallel_tool_calls = false;
|
|
841
|
+
if (ir.responseFormat?.type === 'json_schema') {
|
|
842
|
+
const js = { name: ir.responseFormat.name || 'response', schema: ir.responseFormat.schema };
|
|
843
|
+
if (typeof ir.responseFormat.strict === 'boolean') js.strict = ir.responseFormat.strict;
|
|
844
|
+
body.response_format = { type: 'json_schema', json_schema: js };
|
|
845
|
+
} else if (ir.responseFormat?.type === 'json_object') {
|
|
846
|
+
body.response_format = { type: 'json_object' };
|
|
847
|
+
}
|
|
825
848
|
|
|
826
849
|
if (wantsThinking && mode === 'native') {
|
|
827
850
|
body.reasoning_effort = ir.thinking?.effort && !['max', 'xhigh'].includes(ir.thinking.effort)
|
|
@@ -941,6 +964,8 @@ function irToAnthropicBody(ir, model) {
|
|
|
941
964
|
if (typeof ir.params.topP === 'number') body.top_p = ir.params.topP;
|
|
942
965
|
if (typeof ir.params.topK === 'number') body.top_k = ir.params.topK;
|
|
943
966
|
if (ir.params.stop.length) body.stop_sequences = ir.params.stop;
|
|
967
|
+
// Anthropic has no JSON mode without a schema.
|
|
968
|
+
if (ir.responseFormat?.type === 'json_schema') body.output_config = { format: { type: 'json_schema', schema: ir.responseFormat.schema } };
|
|
944
969
|
if (ir.thinking && ir.thinking.type === 'adaptive') {
|
|
945
970
|
body.thinking = { type: 'adaptive' };
|
|
946
971
|
} else if (ir.thinking && ir.thinking.type !== 'disabled' && body.max_tokens > 1024) {
|
|
@@ -1062,6 +1087,10 @@ function irToVertexBody(ir, model) {
|
|
|
1062
1087
|
gc.thinkingConfig.thinkingBudget = clampBudget(ir.thinking.budget ?? (ir.thinking.effort ? effortToBudget(ir.thinking.effort) : 2048));
|
|
1063
1088
|
}
|
|
1064
1089
|
}
|
|
1090
|
+
if (ir.responseFormat) {
|
|
1091
|
+
gc.responseMimeType = 'application/json';
|
|
1092
|
+
if (ir.responseFormat.type === 'json_schema') gc.responseSchema = toGeminiSchema(ir.responseFormat.schema);
|
|
1093
|
+
}
|
|
1065
1094
|
if (Object.keys(gc).length) body.generationConfig = gc;
|
|
1066
1095
|
return body;
|
|
1067
1096
|
}
|
|
Binary file
|
package/icons/claude.png
ADDED
|
Binary file
|
package/icons/codex.png
ADDED
|
Binary file
|
|
Binary file
|
package/icons/gemini.png
ADDED
|
Binary file
|
package/icons/github.png
ADDED
|
Binary file
|
package/icons/groq.png
ADDED
|
Binary file
|
package/icons/intact.svg
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 32 32"><defs><linearGradient id="lg" x1="0" y1="0" x2="1" y2="1"><stop offset="0" stop-color="#7c6cf6"/><stop offset="1" stop-color="#2dd4bf"/></linearGradient></defs><rect width="32" height="32" rx="9" fill="url(#lg)"/><path d="M16 6.5 24 9.6v6.1c0 5-3.4 8.6-8 10-4.6-1.4-8-5-8-10V9.6Z" fill="none" stroke="#fff" stroke-width="2.2" stroke-linejoin="round"/><path d="m12.4 16.2 2.6 2.6 4.8-5" fill="none" stroke="#fff" stroke-width="2.4" stroke-linecap="round" stroke-linejoin="round"/></svg>
|
package/icons/ollama.png
ADDED
|
Binary file
|
package/icons/openai.png
ADDED
|
Binary file
|
|
Binary file
|
package/icons/qwen.png
ADDED
|
Binary file
|
package/icons/vertex.png
ADDED
|
Binary file
|
package/package.json
CHANGED
package/proxy.mjs
CHANGED
|
@@ -1018,15 +1018,6 @@ function validateProfileInput(p) {
|
|
|
1018
1018
|
if (name !== '' && !isSafeModelName(name)) return `Invalid codexRoles.${slot} value "${name}"`;
|
|
1019
1019
|
}
|
|
1020
1020
|
}
|
|
1021
|
-
if (p.blindfoldPort !== undefined && p.blindfoldPort !== '' && !parsePort(p.blindfoldPort)) {
|
|
1022
|
-
return `Invalid blindfoldPort "${p.blindfoldPort}"`;
|
|
1023
|
-
}
|
|
1024
|
-
if (p.blindfoldHost !== undefined && p.blindfoldHost !== '' && !/^[A-Za-z0-9.-]{1,253}$/.test(p.blindfoldHost)) {
|
|
1025
|
-
return `Invalid blindfoldHost "${p.blindfoldHost}"`;
|
|
1026
|
-
}
|
|
1027
|
-
if (p.blindfoldPrefix !== undefined && p.blindfoldPrefix !== '' && !/^\/[A-Za-z0-9._~/-]{0,200}$/.test(p.blindfoldPrefix)) {
|
|
1028
|
-
return `Invalid blindfoldPrefix "${p.blindfoldPrefix}"`;
|
|
1029
|
-
}
|
|
1030
1021
|
return null;
|
|
1031
1022
|
}
|
|
1032
1023
|
|
|
@@ -1118,11 +1109,32 @@ async function route(req, res) {
|
|
|
1118
1109
|
res.writeHead(204, {
|
|
1119
1110
|
'Access-Control-Allow-Origin': req.headers.origin || `http://127.0.0.1:${PORT}`,
|
|
1120
1111
|
'Access-Control-Allow-Methods': 'GET, POST, OPTIONS',
|
|
1121
|
-
'Access-Control-Allow-Headers': 'Content-Type, Authorization'
|
|
1112
|
+
'Access-Control-Allow-Headers': 'Content-Type, Authorization, x-llm-switcher-token'
|
|
1122
1113
|
});
|
|
1123
1114
|
return res.end();
|
|
1124
1115
|
}
|
|
1125
1116
|
|
|
1117
|
+
// Serve static tool icons (cached, immutable, strictly image/png and image/svg+xml)
|
|
1118
|
+
if (method === 'GET' && pathname.startsWith('/icons/')) {
|
|
1119
|
+
const iconName = path.basename(pathname);
|
|
1120
|
+
if (!/\.(png|svg)$/i.test(iconName)) {
|
|
1121
|
+
res.writeHead(404, { 'Content-Type': 'text/plain' });
|
|
1122
|
+
return res.end('Icon not found');
|
|
1123
|
+
}
|
|
1124
|
+
const iconPath = path.join(__dirname, 'icons', iconName);
|
|
1125
|
+
if (fs.existsSync(iconPath)) {
|
|
1126
|
+
const ext = path.extname(iconName).toLowerCase();
|
|
1127
|
+
const contentType = ext === '.svg' ? 'image/svg+xml' : 'image/png';
|
|
1128
|
+
res.writeHead(200, {
|
|
1129
|
+
'Content-Type': contentType,
|
|
1130
|
+
'Cache-Control': 'public, max-age=604800, immutable'
|
|
1131
|
+
});
|
|
1132
|
+
return fs.createReadStream(iconPath).pipe(res);
|
|
1133
|
+
}
|
|
1134
|
+
res.writeHead(404, { 'Content-Type': 'text/plain' });
|
|
1135
|
+
return res.end('Icon not found');
|
|
1136
|
+
}
|
|
1137
|
+
|
|
1126
1138
|
// Serve Web UI (no-cache: always serve the latest version after file edits)
|
|
1127
1139
|
if (method === 'GET' && (pathname === '/' || pathname === '/ui')) {
|
|
1128
1140
|
if (fs.existsSync(uiHtmlPath)) {
|
|
@@ -1275,6 +1287,11 @@ async function routeApi(req, res, method, pathname) {
|
|
|
1275
1287
|
return sendJson(res, 200, { logs: requestLogs.slice().reverse() });
|
|
1276
1288
|
}
|
|
1277
1289
|
|
|
1290
|
+
// GET /api/catalog (Dynamic Model Discovery)
|
|
1291
|
+
if (method === 'GET' && pathname === '/api/catalog') {
|
|
1292
|
+
return sendJson(res, 200, loadCatalogCache(STATE_DIR));
|
|
1293
|
+
}
|
|
1294
|
+
|
|
1278
1295
|
if (method !== 'POST') {
|
|
1279
1296
|
req.resume();
|
|
1280
1297
|
return sendJson(res, 404, { error: `Not found: ${method} ${pathname}` });
|
|
@@ -1287,11 +1304,6 @@ async function routeApi(req, res, method, pathname) {
|
|
|
1287
1304
|
return sendJson(res, err.status || 400, { error: err.message });
|
|
1288
1305
|
}
|
|
1289
1306
|
|
|
1290
|
-
// GET /api/catalog
|
|
1291
|
-
if (method === 'GET' && pathname === '/api/catalog') {
|
|
1292
|
-
return sendJson(res, 200, loadCatalogCache(STATE_DIR));
|
|
1293
|
-
}
|
|
1294
|
-
|
|
1295
1307
|
// POST /api/catalog/refresh
|
|
1296
1308
|
if (method === 'POST' && pathname === '/api/catalog/refresh') {
|
|
1297
1309
|
const active = getActiveMap(requireConfig(res) || {});
|
|
@@ -1,93 +1,93 @@
|
|
|
1
|
-
---
|
|
2
|
-
name: llm-switcher
|
|
3
|
-
description: Mandatory guidance for AI coding agents and token compression tools (Headroom, RTK, Ponytail, Repomix). Enforces routing all LLM traffic through the local LLM Switcher edge gateway (http://127.0.0.1:3456) instead of directly calling provider endpoints.
|
|
4
|
-
---
|
|
5
|
-
|
|
6
|
-
# LLM Switcher Edge Gateway Skill
|
|
7
|
-
|
|
8
|
-
This skill governs how AI coding agents (Claude Code, Codex, Cursor, Windsurf, Opencode) and prompt optimization/compression tools (Headroom, RTK, Ponytail, Repomix) must interact with LLM providers on this workstation.
|
|
9
|
-
|
|
10
|
-
## 1. The Core Directive
|
|
11
|
-
|
|
12
|
-
> **ALL OUTBOUND LLM REQUESTS MUST ROUTE THROUGH LLM SWITCHER (`http://127.0.0.1:3456`).**
|
|
13
|
-
>
|
|
14
|
-
> Never configure any tool, script, sub-agent, or token compression proxy to call external provider endpoints directly (such as `https://api.anthropic.com`, `https://api.openai.com`, or `https://generativelanguage.googleapis.com`).
|
|
15
|
-
>
|
|
16
|
-
> If a token compression tool (Headroom, RTK, Ponytail) is used, it MUST be positioned as an **inner processing layer**, and its **upstream destination MUST be configured to point to LLM Switcher** at `http://127.0.0.1:3456`.
|
|
17
|
-
|
|
18
|
-
```
|
|
19
|
-
[Client CLI / Agent]
|
|
20
|
-
│
|
|
21
|
-
▼
|
|
22
|
-
[Token Compressor: RTK / Headroom / Ponytail] <-- Inner Layer
|
|
23
|
-
│
|
|
24
|
-
▼ (MANDATORY: Upstream MUST point to http://127.0.0.1:3456)
|
|
25
|
-
[LLM Switcher Edge Gateway (:3456)] <-- Outermost Gatekeeper
|
|
26
|
-
│
|
|
27
|
-
▼ (Heals schemas, maps 1M context, extracts thinking, converts protocol)
|
|
28
|
-
[Internet / intact (Recommended) / 9Router / Upstream LLM]
|
|
29
|
-
```
|
|
30
|
-
|
|
31
|
-
## 2. Why This is Mandatory
|
|
32
|
-
|
|
33
|
-
Direct outbound calls from token compressors will break the developer environment for several reasons:
|
|
34
|
-
|
|
35
|
-
1. **Schema Breakage & The Healer Engine:**
|
|
36
|
-
Token optimizers frequently prune conversation history to save tokens. In doing so, they often delete an `assistant` turn containing a `tool_use`, while leaving the subsequent `user` turn containing a `tool_result`.
|
|
37
|
-
- Anthropic and OpenAI APIs strictly reject this with `HTTP 400 (orphaned tool_use_id)`.
|
|
38
|
-
- **LLM Switcher** intercepts these malformed requests and runs its **Healer Engine**, automatically converting orphaned results into safe context text blocks and merging consecutive same-role turns so upstream providers accept the payload cleanly.
|
|
39
|
-
2. **Thinking / Reasoning Protection:**
|
|
40
|
-
Many token compressors naively strip `thinking` parameters or reasoning budgets to reduce token count.
|
|
41
|
-
- When routed through LLM Switcher, the gateway detects reasoning models (such as `ag/claude-opus-4-6-thinking` or `ag/gemini-3.8-flash`) and **automatically restores thinking parameters**, ensuring the model retains its full reasoning capabilities.
|
|
42
|
-
3. **Official Context Windows & Dynamic Model Discovery:**
|
|
43
|
-
Context windows follow the model you select without mutating permanent settings. The gateway auto-discovers official models from Anthropic and OpenAI, refreshing mappings on tool version upgrades (`switch models`).
|
|
44
|
-
4. **Server Gateway Synergy (intact / 9Router):**
|
|
45
|
-
LLM Switcher pairs directly with server-side gateways like **intact** (strongly recommended for credential proxying, account pooling, and contract lab self-healing) or 9Router, handling retries and quotas far better than any local tool.
|
|
46
|
-
|
|
47
|
-
## 3. Configuring Token Compression Tools
|
|
48
|
-
|
|
49
|
-
When setting up or assisting the user with token compression tools, apply these exact configurations:
|
|
50
|
-
|
|
51
|
-
### A. Headroom (Prompt Compression Proxy)
|
|
52
|
-
When configuring Headroom's upstream endpoint:
|
|
53
|
-
- In Headroom config (`headroom.yaml` or environment variables):
|
|
54
|
-
```bash
|
|
55
|
-
ANTHROPIC_BASE_URL="http://127.0.0.1:3456"
|
|
56
|
-
OPENAI_BASE_URL="http://127.0.0.1:3456/v1"
|
|
57
|
-
```
|
|
58
|
-
- Point your CLI to Headroom's listening port (e.g. `http://127.0.0.1:8787`). Headroom compresses the prompt and forwards it to LLM Switcher on port `3456`.
|
|
59
|
-
|
|
60
|
-
### B. RTK (Rust Token Killer / Repomix)
|
|
61
|
-
- In RTK configuration or execution flags:
|
|
62
|
-
```bash
|
|
63
|
-
# Ensure RTK's target upstream proxy is LLM Switcher
|
|
64
|
-
export UPSTREAM_URL="http://127.0.0.1:3456"
|
|
65
|
-
```
|
|
66
|
-
|
|
67
|
-
### C. Ponytail / Custom Proxies
|
|
68
|
-
- Always set the proxy's upstream host to:
|
|
69
|
-
`http://127.0.0.1:3456` (or `http://127.0.0.1:3456/v1` for OpenAI-compatible tools).
|
|
70
|
-
|
|
71
|
-
### D. Sub-Agents & Spawned Terminal Commands
|
|
72
|
-
When an agent spawns sub-processes that call LLM CLIs:
|
|
73
|
-
- **Never advise sourcing `env.sh` or `env.cmd`.** Those files carry no proxy variables any
|
|
74
|
-
more, so sourcing one changes nothing — and a stale variable an older version once wrote
|
|
75
|
-
would still point the tool at a port where nothing listens.
|
|
76
|
-
- Run `switch claude <profile>` or `switch codex <profile>` instead. The shims in
|
|
77
|
-
`~/.llm-switcher/bin` inject the proxy variables into the tool process itself and clear
|
|
78
|
-
anything stale first.
|
|
79
|
-
- `claude` and `codex` on `PATH` already are those shims: invoke them normally, and never
|
|
80
|
-
wrap them in another script.
|
|
81
|
-
|
|
82
|
-
## 4. Operational Rules for AI Agents
|
|
83
|
-
|
|
84
|
-
1. **NEVER edit `~/.claude/settings.json` directly.**
|
|
85
|
-
LLM Switcher uses zero-mutation launcher flags (`active.flag`). Writing custom base URLs or auth tokens into `settings.json` triggers warning banners in Claude Code and risks breaking the user's official subscription. Use `switch` CLI commands instead.
|
|
86
|
-
2. **Check Gateway Health Before Complex Operations:**
|
|
87
|
-
Run `switch status` or call the `switcher_audit` MCP tool to confirm:
|
|
88
|
-
- LLM Switcher is active on port `3456`.
|
|
89
|
-
- The active profile matches the intended CLI target (Claude Code, Codex, or OpenAI).
|
|
90
|
-
3. **Verify Routing When Errors Occur:**
|
|
91
|
-
If a tool fails with `HTTP 400`, `HTTP 502`, or connection errors:
|
|
92
|
-
- Run `switch doctor` to audit port collisions and environment variables.
|
|
93
|
-
- Inspect recent request logs via `http://127.0.0.1:3456/ui` (Tab 4: Live Inspector) to see if an intermediary tool mangled the payload.
|
|
1
|
+
---
|
|
2
|
+
name: llm-switcher
|
|
3
|
+
description: Mandatory guidance for AI coding agents and token compression tools (Headroom, RTK, Ponytail, Repomix). Enforces routing all LLM traffic through the local LLM Switcher edge gateway (http://127.0.0.1:3456) instead of directly calling provider endpoints.
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# LLM Switcher Edge Gateway Skill
|
|
7
|
+
|
|
8
|
+
This skill governs how AI coding agents (Claude Code, Codex, Cursor, Windsurf, Opencode) and prompt optimization/compression tools (Headroom, RTK, Ponytail, Repomix) must interact with LLM providers on this workstation.
|
|
9
|
+
|
|
10
|
+
## 1. The Core Directive
|
|
11
|
+
|
|
12
|
+
> **ALL OUTBOUND LLM REQUESTS MUST ROUTE THROUGH LLM SWITCHER (`http://127.0.0.1:3456`).**
|
|
13
|
+
>
|
|
14
|
+
> Never configure any tool, script, sub-agent, or token compression proxy to call external provider endpoints directly (such as `https://api.anthropic.com`, `https://api.openai.com`, or `https://generativelanguage.googleapis.com`).
|
|
15
|
+
>
|
|
16
|
+
> If a token compression tool (Headroom, RTK, Ponytail) is used, it MUST be positioned as an **inner processing layer**, and its **upstream destination MUST be configured to point to LLM Switcher** at `http://127.0.0.1:3456`.
|
|
17
|
+
|
|
18
|
+
```
|
|
19
|
+
[Client CLI / Agent]
|
|
20
|
+
│
|
|
21
|
+
▼
|
|
22
|
+
[Token Compressor: RTK / Headroom / Ponytail] <-- Inner Layer
|
|
23
|
+
│
|
|
24
|
+
▼ (MANDATORY: Upstream MUST point to http://127.0.0.1:3456)
|
|
25
|
+
[LLM Switcher Edge Gateway (:3456)] <-- Outermost Gatekeeper
|
|
26
|
+
│
|
|
27
|
+
▼ (Heals schemas, maps 1M context, extracts thinking, converts protocol)
|
|
28
|
+
[Internet / intact (Recommended) / 9Router / Upstream LLM]
|
|
29
|
+
```
|
|
30
|
+
|
|
31
|
+
## 2. Why This is Mandatory
|
|
32
|
+
|
|
33
|
+
Direct outbound calls from token compressors will break the developer environment for several reasons:
|
|
34
|
+
|
|
35
|
+
1. **Schema Breakage & The Healer Engine:**
|
|
36
|
+
Token optimizers frequently prune conversation history to save tokens. In doing so, they often delete an `assistant` turn containing a `tool_use`, while leaving the subsequent `user` turn containing a `tool_result`.
|
|
37
|
+
- Anthropic and OpenAI APIs strictly reject this with `HTTP 400 (orphaned tool_use_id)`.
|
|
38
|
+
- **LLM Switcher** intercepts these malformed requests and runs its **Healer Engine**, automatically converting orphaned results into safe context text blocks and merging consecutive same-role turns so upstream providers accept the payload cleanly.
|
|
39
|
+
2. **Thinking / Reasoning Protection:**
|
|
40
|
+
Many token compressors naively strip `thinking` parameters or reasoning budgets to reduce token count.
|
|
41
|
+
- When routed through LLM Switcher, the gateway detects reasoning models (such as `ag/claude-opus-4-6-thinking` or `ag/gemini-3.8-flash`) and **automatically restores thinking parameters**, ensuring the model retains its full reasoning capabilities.
|
|
42
|
+
3. **Official Context Windows & Dynamic Model Discovery:**
|
|
43
|
+
Context windows follow the model you select without mutating permanent settings. The gateway auto-discovers official models from Anthropic and OpenAI, refreshing mappings on tool version upgrades (`switch models`).
|
|
44
|
+
4. **Server Gateway Synergy (intact / 9Router):**
|
|
45
|
+
LLM Switcher pairs directly with server-side gateways like **intact** (strongly recommended for credential proxying, account pooling, and contract lab self-healing) or 9Router, handling retries and quotas far better than any local tool.
|
|
46
|
+
|
|
47
|
+
## 3. Configuring Token Compression Tools
|
|
48
|
+
|
|
49
|
+
When setting up or assisting the user with token compression tools, apply these exact configurations:
|
|
50
|
+
|
|
51
|
+
### A. Headroom (Prompt Compression Proxy)
|
|
52
|
+
When configuring Headroom's upstream endpoint:
|
|
53
|
+
- In Headroom config (`headroom.yaml` or environment variables):
|
|
54
|
+
```bash
|
|
55
|
+
ANTHROPIC_BASE_URL="http://127.0.0.1:3456"
|
|
56
|
+
OPENAI_BASE_URL="http://127.0.0.1:3456/v1"
|
|
57
|
+
```
|
|
58
|
+
- Point your CLI to Headroom's listening port (e.g. `http://127.0.0.1:8787`). Headroom compresses the prompt and forwards it to LLM Switcher on port `3456`.
|
|
59
|
+
|
|
60
|
+
### B. RTK (Rust Token Killer / Repomix)
|
|
61
|
+
- In RTK configuration or execution flags:
|
|
62
|
+
```bash
|
|
63
|
+
# Ensure RTK's target upstream proxy is LLM Switcher
|
|
64
|
+
export UPSTREAM_URL="http://127.0.0.1:3456"
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
### C. Ponytail / Custom Proxies
|
|
68
|
+
- Always set the proxy's upstream host to:
|
|
69
|
+
`http://127.0.0.1:3456` (or `http://127.0.0.1:3456/v1` for OpenAI-compatible tools).
|
|
70
|
+
|
|
71
|
+
### D. Sub-Agents & Spawned Terminal Commands
|
|
72
|
+
When an agent spawns sub-processes that call LLM CLIs:
|
|
73
|
+
- **Never advise sourcing `env.sh` or `env.cmd`.** Those files carry no proxy variables any
|
|
74
|
+
more, so sourcing one changes nothing — and a stale variable an older version once wrote
|
|
75
|
+
would still point the tool at a port where nothing listens.
|
|
76
|
+
- Run `switch claude <profile>` or `switch codex <profile>` instead. The shims in
|
|
77
|
+
`~/.llm-switcher/bin` inject the proxy variables into the tool process itself and clear
|
|
78
|
+
anything stale first.
|
|
79
|
+
- `claude` and `codex` on `PATH` already are those shims: invoke them normally, and never
|
|
80
|
+
wrap them in another script.
|
|
81
|
+
|
|
82
|
+
## 4. Operational Rules for AI Agents
|
|
83
|
+
|
|
84
|
+
1. **NEVER edit `~/.claude/settings.json` directly.**
|
|
85
|
+
LLM Switcher uses zero-mutation launcher flags (`active.flag`). Writing custom base URLs or auth tokens into `settings.json` triggers warning banners in Claude Code and risks breaking the user's official subscription. Use `switch` CLI commands instead.
|
|
86
|
+
2. **Check Gateway Health Before Complex Operations:**
|
|
87
|
+
Run `switch status` or call the `switcher_audit` MCP tool to confirm:
|
|
88
|
+
- LLM Switcher is active on port `3456`.
|
|
89
|
+
- The active profile matches the intended CLI target (Claude Code, Codex, or OpenAI).
|
|
90
|
+
3. **Verify Routing When Errors Occur:**
|
|
91
|
+
If a tool fails with `HTTP 400`, `HTTP 502`, or connection errors:
|
|
92
|
+
- Run `switch doctor` to audit port collisions and environment variables.
|
|
93
|
+
- Inspect recent request logs via `http://127.0.0.1:3456/ui` (Tab 4: Live Inspector) to see if an intermediary tool mangled the payload.
|
package/switch
CHANGED
|
File without changes
|
|
@@ -0,0 +1,210 @@
|
|
|
1
|
+
import { describe, it } from 'node:test';
|
|
2
|
+
import assert from 'node:assert/strict';
|
|
3
|
+
import http from 'node:http';
|
|
4
|
+
import {
|
|
5
|
+
heuristicClassify,
|
|
6
|
+
findJevKey,
|
|
7
|
+
classifyPrompt,
|
|
8
|
+
checkSemanticEquivalence,
|
|
9
|
+
DEFAULT_JEV_URL,
|
|
10
|
+
OPENROUTER_JEV_URL,
|
|
11
|
+
} from '../classifier.mjs';
|
|
12
|
+
|
|
13
|
+
function createServer(handler) {
|
|
14
|
+
const server = http.createServer(handler);
|
|
15
|
+
return new Promise((resolve, reject) => {
|
|
16
|
+
server.listen(0, '127.0.0.1', () => {
|
|
17
|
+
const addr = server.address();
|
|
18
|
+
resolve({
|
|
19
|
+
url: `http://127.0.0.1:${addr.port}`,
|
|
20
|
+
close: () => new Promise((res) => server.close(res)),
|
|
21
|
+
});
|
|
22
|
+
});
|
|
23
|
+
server.on('error', reject);
|
|
24
|
+
});
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
describe('classifier module', () => {
|
|
28
|
+
describe('heuristicClassify', () => {
|
|
29
|
+
it('classifies complex tasks to opus', () => {
|
|
30
|
+
assert.equal(heuristicClassify('Design the system architecture for high throughput'), 'opus');
|
|
31
|
+
assert.equal(heuristicClassify('Find race condition in concurrent worker pool'), 'opus');
|
|
32
|
+
assert.equal(heuristicClassify('Conduct a security audit of our auth boundary'), 'opus');
|
|
33
|
+
});
|
|
34
|
+
|
|
35
|
+
it('classifies simple short tasks to haiku', () => {
|
|
36
|
+
assert.equal(heuristicClassify('Fix typo in comment'), 'haiku');
|
|
37
|
+
assert.equal(heuristicClassify('Format json output to string'), 'haiku');
|
|
38
|
+
assert.equal(heuristicClassify('Extract date from this line'), 'haiku');
|
|
39
|
+
});
|
|
40
|
+
|
|
41
|
+
it('defaults general programming tasks to sonnet', () => {
|
|
42
|
+
assert.equal(heuristicClassify('Refactor this helper function to accept options object and write tests'), 'sonnet');
|
|
43
|
+
});
|
|
44
|
+
});
|
|
45
|
+
|
|
46
|
+
describe('findJevKey', () => {
|
|
47
|
+
it('resolves TYPESAFE_API_KEY first', () => {
|
|
48
|
+
const res = findJevKey({ TYPESAFE_API_KEY: 'ts_key', JEV_API_KEY: 'jev_key' });
|
|
49
|
+
assert.deepEqual(res, { key: 'ts_key', source: 'typesafe' });
|
|
50
|
+
});
|
|
51
|
+
|
|
52
|
+
it('resolves JEV_API_KEY when TYPESAFE_API_KEY is absent', () => {
|
|
53
|
+
const res = findJevKey({ JEV_API_KEY: 'jev_key' });
|
|
54
|
+
assert.deepEqual(res, { key: 'jev_key', source: 'openrouter' });
|
|
55
|
+
});
|
|
56
|
+
|
|
57
|
+
it('returns null on empty env', () => {
|
|
58
|
+
assert.equal(findJevKey({}), null);
|
|
59
|
+
});
|
|
60
|
+
});
|
|
61
|
+
|
|
62
|
+
describe('classifyPrompt', () => {
|
|
63
|
+
it('returns heuristic tier with reason no-key when no key is present', async () => {
|
|
64
|
+
const res = await classifyPrompt({
|
|
65
|
+
prompt: 'Design architecture for database',
|
|
66
|
+
env: {},
|
|
67
|
+
});
|
|
68
|
+
assert.equal(res.tier, 'opus');
|
|
69
|
+
assert.equal(res.source, 'heuristic');
|
|
70
|
+
assert.equal(res.reason, 'no-key');
|
|
71
|
+
});
|
|
72
|
+
|
|
73
|
+
it('calls Jev endpoint and parses choice response', async () => {
|
|
74
|
+
let receivedAuth = null;
|
|
75
|
+
let receivedBody = null;
|
|
76
|
+
|
|
77
|
+
const server = await createServer(async (req, res) => {
|
|
78
|
+
receivedAuth = req.headers.authorization;
|
|
79
|
+
const chunks = [];
|
|
80
|
+
for await (const chunk of req) chunks.push(chunk);
|
|
81
|
+
receivedBody = JSON.parse(Buffer.concat(chunks).toString('utf8'));
|
|
82
|
+
|
|
83
|
+
res.writeHead(200, { 'Content-Type': 'application/json' });
|
|
84
|
+
res.end(
|
|
85
|
+
JSON.stringify({
|
|
86
|
+
model: 'jev-latest',
|
|
87
|
+
answers: {
|
|
88
|
+
recommended_tier: {
|
|
89
|
+
choice: 'haiku',
|
|
90
|
+
confidence: 0.94,
|
|
91
|
+
distribution: { haiku: 0.94, sonnet: 0.05, opus: 0.01 },
|
|
92
|
+
},
|
|
93
|
+
},
|
|
94
|
+
}),
|
|
95
|
+
);
|
|
96
|
+
});
|
|
97
|
+
|
|
98
|
+
try {
|
|
99
|
+
const res = await classifyPrompt({
|
|
100
|
+
prompt: 'Translate hello to French',
|
|
101
|
+
apiKey: 'test-ts-key',
|
|
102
|
+
url: server.url,
|
|
103
|
+
});
|
|
104
|
+
|
|
105
|
+
assert.equal(res.tier, 'haiku');
|
|
106
|
+
assert.equal(res.confidence, 0.94);
|
|
107
|
+
assert.equal(res.source, 'jev');
|
|
108
|
+
assert.equal(res.model, 'jev-latest');
|
|
109
|
+
assert.equal(receivedAuth, 'Bearer test-ts-key');
|
|
110
|
+
assert.equal(receivedBody.state, 'Translate hello to French');
|
|
111
|
+
assert.ok('recommended_tier' in receivedBody.questions);
|
|
112
|
+
} finally {
|
|
113
|
+
await server.close();
|
|
114
|
+
}
|
|
115
|
+
});
|
|
116
|
+
|
|
117
|
+
it('falls back gracefully to heuristic on HTTP error', async () => {
|
|
118
|
+
const server = await createServer((req, res) => {
|
|
119
|
+
res.writeHead(500, { 'Content-Type': 'application/json' });
|
|
120
|
+
res.end(JSON.stringify({ error: 'internal error' }));
|
|
121
|
+
});
|
|
122
|
+
|
|
123
|
+
try {
|
|
124
|
+
const res = await classifyPrompt({
|
|
125
|
+
prompt: 'Fix typo',
|
|
126
|
+
apiKey: 'key',
|
|
127
|
+
url: server.url,
|
|
128
|
+
});
|
|
129
|
+
|
|
130
|
+
assert.equal(res.tier, 'haiku');
|
|
131
|
+
assert.equal(res.source, 'heuristic');
|
|
132
|
+
assert.equal(res.reason, 'http-500');
|
|
133
|
+
} finally {
|
|
134
|
+
await server.close();
|
|
135
|
+
}
|
|
136
|
+
});
|
|
137
|
+
|
|
138
|
+
it('falls back gracefully to heuristic on timeout', async () => {
|
|
139
|
+
const server = await createServer((req, res) => {
|
|
140
|
+
// Hang indefinitely
|
|
141
|
+
});
|
|
142
|
+
|
|
143
|
+
try {
|
|
144
|
+
const res = await classifyPrompt({
|
|
145
|
+
prompt: 'Fix typo',
|
|
146
|
+
apiKey: 'key',
|
|
147
|
+
url: server.url,
|
|
148
|
+
timeoutMs: 50,
|
|
149
|
+
});
|
|
150
|
+
|
|
151
|
+
assert.equal(res.tier, 'haiku');
|
|
152
|
+
assert.equal(res.source, 'heuristic');
|
|
153
|
+
assert.equal(res.reason, 'timeout');
|
|
154
|
+
} finally {
|
|
155
|
+
await server.close();
|
|
156
|
+
}
|
|
157
|
+
});
|
|
158
|
+
});
|
|
159
|
+
|
|
160
|
+
describe('checkSemanticEquivalence', () => {
|
|
161
|
+
it('returns exact-match immediately when strings match', async () => {
|
|
162
|
+
const res = await checkSemanticEquivalence({
|
|
163
|
+
candidate: 'What is Node.js?',
|
|
164
|
+
target: 'What is Node.js?',
|
|
165
|
+
});
|
|
166
|
+
assert.equal(res.equivalent, true);
|
|
167
|
+
assert.equal(res.confidence, 1.0);
|
|
168
|
+
assert.equal(res.reason, 'exact-match');
|
|
169
|
+
});
|
|
170
|
+
|
|
171
|
+
it('bypasses immediately with false when no key is configured', async () => {
|
|
172
|
+
const res = await checkSemanticEquivalence({
|
|
173
|
+
candidate: 'Say hello',
|
|
174
|
+
target: 'Greet me',
|
|
175
|
+
env: {},
|
|
176
|
+
});
|
|
177
|
+
assert.equal(res.equivalent, false);
|
|
178
|
+
assert.equal(res.reason, 'no-key-fallback-bypass');
|
|
179
|
+
});
|
|
180
|
+
|
|
181
|
+
it('evaluates Noul probability from Jev and confirms equivalence >= 0.85', async () => {
|
|
182
|
+
const server = await createServer(async (req, res) => {
|
|
183
|
+
res.writeHead(200, { 'Content-Type': 'application/json' });
|
|
184
|
+
res.end(
|
|
185
|
+
JSON.stringify({
|
|
186
|
+
model: 'jev-latest',
|
|
187
|
+
answers: {
|
|
188
|
+
is_equivalent: 0.92,
|
|
189
|
+
},
|
|
190
|
+
}),
|
|
191
|
+
);
|
|
192
|
+
});
|
|
193
|
+
|
|
194
|
+
try {
|
|
195
|
+
const res = await checkSemanticEquivalence({
|
|
196
|
+
candidate: 'How do I read a file in Node?',
|
|
197
|
+
target: 'Read file contents in NodeJS',
|
|
198
|
+
apiKey: 'test-key',
|
|
199
|
+
url: server.url,
|
|
200
|
+
});
|
|
201
|
+
|
|
202
|
+
assert.equal(res.equivalent, true);
|
|
203
|
+
assert.equal(res.confidence, 0.92);
|
|
204
|
+
assert.equal(res.reason, 'jev-confirmed');
|
|
205
|
+
} finally {
|
|
206
|
+
await server.close();
|
|
207
|
+
}
|
|
208
|
+
});
|
|
209
|
+
});
|
|
210
|
+
});
|
package/tests/formats.test.mjs
CHANGED
|
@@ -802,3 +802,33 @@ test('an empty tool schema is healed, not forwarded as {}', () => {
|
|
|
802
802
|
);
|
|
803
803
|
assert.deepEqual(body.tools[0].function.parameters, { type: 'object', properties: {} });
|
|
804
804
|
});
|
|
805
|
+
|
|
806
|
+
// Structured output: a JSON schema the client asks the answer to follow must reach every upstream.
|
|
807
|
+
const TITLE_SCHEMA = { type: 'object', properties: { title: { type: 'string' } }, required: ['title'], additionalProperties: false };
|
|
808
|
+
|
|
809
|
+
test('structured output: Anthropic output_config.format reaches chat, Anthropic and Vertex upstreams', () => {
|
|
810
|
+
const ir = anthropicToIR({ model: 'x', max_tokens: 100, messages: [{ role: 'user', content: 'hi' }], tools: [],
|
|
811
|
+
output_config: { format: { type: 'json_schema', schema: TITLE_SCHEMA } } });
|
|
812
|
+
assert.deepEqual(irToChatBody(ir, 'm').response_format,
|
|
813
|
+
{ type: 'json_schema', json_schema: { name: 'response', schema: TITLE_SCHEMA } });
|
|
814
|
+
assert.deepEqual(irToAnthropicBody(ir, 'm').output_config, { format: { type: 'json_schema', schema: TITLE_SCHEMA } });
|
|
815
|
+
const gc = irToVertexBody(ir, 'm').generationConfig;
|
|
816
|
+
assert.equal(gc.responseMimeType, 'application/json');
|
|
817
|
+
assert.deepEqual(gc.responseSchema, toGeminiSchema(TITLE_SCHEMA));
|
|
818
|
+
// The deprecated top-level field still works.
|
|
819
|
+
const old = anthropicToIR({ model: 'x', messages: [{ role: 'user', content: 'hi' }], output_format: { type: 'json_schema', schema: TITLE_SCHEMA } });
|
|
820
|
+
assert.equal(irToChatBody(old, 'm').response_format.json_schema.schema, TITLE_SCHEMA);
|
|
821
|
+
});
|
|
822
|
+
|
|
823
|
+
test('structured output: Responses text.format keeps name and strict; JSON mode and plain text', () => {
|
|
824
|
+
const resp = responsesToIR({ model: 'x', input: 'hi', text: { format: { type: 'json_schema', name: 'title', schema: TITLE_SCHEMA, strict: true } } });
|
|
825
|
+
assert.deepEqual(irToChatBody(resp, 'm').response_format,
|
|
826
|
+
{ type: 'json_schema', json_schema: { name: 'title', schema: TITLE_SCHEMA, strict: true } });
|
|
827
|
+
const jm = responsesToIR({ model: 'x', input: 'hi', text: { format: { type: 'json_object' } } });
|
|
828
|
+
assert.deepEqual(irToChatBody(jm, 'm').response_format, { type: 'json_object' });
|
|
829
|
+
assert.equal(irToVertexBody(jm, 'm').generationConfig.responseMimeType, 'application/json');
|
|
830
|
+
assert.equal(irToAnthropicBody(jm, 'm').output_config, undefined);
|
|
831
|
+
const txt = responsesToIR({ model: 'x', input: 'hi', text: { format: { type: 'text' } } });
|
|
832
|
+
assert.equal(irToChatBody(txt, 'm').response_format, undefined);
|
|
833
|
+
assert.equal(irToVertexBody(txt, 'm').generationConfig, undefined);
|
|
834
|
+
});
|