llm-switcher 1.2.0 → 1.2.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +12 -0
- package/classifier.mjs +238 -0
- package/docs/TOKEN-OPTIMIZER-INTEROP.md +110 -110
- package/docs/response-matrix.json +1130 -1130
- package/formats.mjs +30 -1
- package/icons/antigravity.png +0 -0
- package/icons/claude.png +0 -0
- package/icons/codex.png +0 -0
- package/icons/deepseek.png +0 -0
- package/icons/gemini.png +0 -0
- package/icons/github.png +0 -0
- package/icons/groq.png +0 -0
- package/icons/intact.svg +1 -0
- package/icons/ollama.png +0 -0
- package/icons/openai.png +0 -0
- package/icons/openrouter.png +0 -0
- package/icons/qwen.png +0 -0
- package/icons/vertex.png +0 -0
- package/package.json +1 -1
- package/proxy.mjs +27 -15
- package/skills/llm-switcher/SKILL.md +93 -93
- package/switch +0 -0
- package/tests/classifier.test.mjs +210 -0
- package/tests/formats.test.mjs +30 -0
- package/tests/helpers.mjs +24 -24
- package/tests/live-optimizer-interop.mjs +205 -205
- package/ui.html +1700 -1623
package/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,17 @@
|
|
|
1
1
|
# Changelog — LLM Switcher
|
|
2
2
|
|
|
3
|
+
## Release 1.2.2
|
|
4
|
+
|
|
5
|
+
- **Structured output reaches every upstream:** A request that asks for JSON that matches a schema now keeps that schema. Before this release, the gateway did not read `output_config.format` (Claude Code) or `text.format` (Codex), so the provider got a free-text request. The schema now goes to the provider as `response_format` (OpenAI Chat), `output_config.format` (Anthropic), or `responseMimeType` with `responseSchema` (Gemini and Vertex). A request for JSON without a schema reaches Anthropic as plain text, because Anthropic has no JSON mode without a schema.
|
|
6
|
+
|
|
7
|
+
## Release 1.2.1
|
|
8
|
+
|
|
9
|
+
- **Intact Visual Architecture & Hash Routing:** Fully restructured dashboard using Intact design system, pure CSS tokens, and hash-based client routing (`#/routes`, `#/models`, `#/logs`, `#/doctor`).
|
|
10
|
+
- **Official Brand Icon Assets:** Integrated official brand icons for Anthropic Claude, OpenAI Codex, Google Gemini, Intact, OpenRouter, and Ollama, served statically under `/icons/*` with immutable caching and PNG/SVG whitelisting.
|
|
11
|
+
- **Prefetched Navigation Counters:** Navigation badges prefetch model discovery and request inspector counts immediately on page load, eliminating the delay where badges showed zero until the tab was selected.
|
|
12
|
+
- **Hardened Profile Slot Mapping:** Synchronized Claude model tiers (`sonnet`, `opus`, `haiku`, `fable`) and Codex model roles (`main`, `review`, `subagent`), ensuring slot persistence when editing profiles and removing redundant role tabs for Claude-targeted profiles.
|
|
13
|
+
- **CORS & Endpoint Fixes:** Added `x-llm-switcher-token` to preflight `Access-Control-Allow-Headers` and resolved routing precedence for `GET /api/catalog`.
|
|
14
|
+
|
|
3
15
|
## Release 1.2.0
|
|
4
16
|
|
|
5
17
|
- **Zero-Mutation Interceptor Invariant:** The gateway now operates strictly in the network path via the blindfold interceptor, never modifying user configuration files like `~/.claude/settings.json` or `~/.codex/config.toml`. Loopback base URLs (`ANTHROPIC_BASE_URL` and `OPENAI_BASE_URL`) are scrubbed by shims so traffic routes through `HTTPS_PROXY` cleanly.
|
package/classifier.mjs
ADDED
|
@@ -0,0 +1,238 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Jev-powered Semantic Classifier & Router for llm-switcher
|
|
3
|
+
* Evaluates prompt complexity using TypeSafe System One (Jev) models
|
|
4
|
+
* and maps intent to optimal model tier (haiku, sonnet, opus).
|
|
5
|
+
*/
|
|
6
|
+
|
|
7
|
+
export const DEFAULT_JEV_URL = 'https://api.typesafe.ai/v1/systemone';
|
|
8
|
+
export const DEFAULT_JEV_MODEL = 'jev-latest';
|
|
9
|
+
export const OPENROUTER_JEV_URL = 'https://openrouter.ai/api/alpha/decisions';
|
|
10
|
+
export const OPENROUTER_JEV_MODEL = 'typesafe/jev-1.13';
|
|
11
|
+
|
|
12
|
+
export const TIER_CRITERIA = {
|
|
13
|
+
haiku: 'Mechanical task, template fill, single fact lookup, extraction, binary decision, short formatting',
|
|
14
|
+
sonnet: 'Multi-step coding, refactor, search across files, general programming, document drafting',
|
|
15
|
+
opus: 'Architecture design, security analysis, deep reasoning, race conditions, migration plans, complex synthesis',
|
|
16
|
+
};
|
|
17
|
+
|
|
18
|
+
/**
|
|
19
|
+
* Fast synchronous heuristic fallback when offline or no API key is set.
|
|
20
|
+
*/
|
|
21
|
+
export function heuristicClassify(prompt = '') {
|
|
22
|
+
const text = String(prompt || '').toLowerCase();
|
|
23
|
+
|
|
24
|
+
// Heavy reasoning or safety critical indicators -> opus
|
|
25
|
+
const opusKeywords = [
|
|
26
|
+
'architecture', 'security audit', 'race condition', 'deadlock', 'concurrency',
|
|
27
|
+
'threat model', 'penetration', 'migration plan', 'adversarial', 'formal verification'
|
|
28
|
+
];
|
|
29
|
+
if (opusKeywords.some((kw) => text.includes(kw))) {
|
|
30
|
+
return 'opus';
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
// Trivial or mechanical indicators -> haiku
|
|
34
|
+
const haikuKeywords = [
|
|
35
|
+
'translate', 'format json', 'extract', 'regex', 'rename', 'single word',
|
|
36
|
+
'yes or no', 'spell check', 'fix typo', 'convert to csv'
|
|
37
|
+
];
|
|
38
|
+
if (text.length < 120 && haikuKeywords.some((kw) => text.includes(kw))) {
|
|
39
|
+
return 'haiku';
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
// Default balanced tier
|
|
43
|
+
return 'sonnet';
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
/**
|
|
47
|
+
* Resolves Jev API key from environment.
|
|
48
|
+
*/
|
|
49
|
+
export function findJevKey(env = process.env) {
|
|
50
|
+
if (typeof env?.TYPESAFE_API_KEY === 'string' && env.TYPESAFE_API_KEY.trim().length > 0) {
|
|
51
|
+
return { key: env.TYPESAFE_API_KEY.trim(), source: 'typesafe' };
|
|
52
|
+
}
|
|
53
|
+
if (typeof env?.JEV_API_KEY === 'string' && env.JEV_API_KEY.trim().length > 0) {
|
|
54
|
+
return { key: env.JEV_API_KEY.trim(), source: 'openrouter' };
|
|
55
|
+
}
|
|
56
|
+
return null;
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
/**
|
|
60
|
+
* Classifies prompt into optimal model tier using TypeSafe Jev.
|
|
61
|
+
* Falls back safely to heuristic on missing key, timeout, or network error.
|
|
62
|
+
*/
|
|
63
|
+
export async function classifyPrompt({
|
|
64
|
+
prompt,
|
|
65
|
+
apiKey,
|
|
66
|
+
url,
|
|
67
|
+
model,
|
|
68
|
+
timeoutMs = 5000,
|
|
69
|
+
env = process.env,
|
|
70
|
+
fetchImpl = globalThis.fetch,
|
|
71
|
+
} = {}) {
|
|
72
|
+
const fallbackTier = heuristicClassify(prompt);
|
|
73
|
+
|
|
74
|
+
const keyInfo = apiKey ? { key: apiKey, source: 'custom' } : findJevKey(env);
|
|
75
|
+
if (!keyInfo) {
|
|
76
|
+
return {
|
|
77
|
+
tier: fallbackTier,
|
|
78
|
+
confidence: 0.5,
|
|
79
|
+
source: 'heuristic',
|
|
80
|
+
reason: 'no-key',
|
|
81
|
+
};
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
const endpointUrl = url || (keyInfo.source === 'openrouter' ? OPENROUTER_JEV_URL : DEFAULT_JEV_URL);
|
|
85
|
+
const modelName = model || (keyInfo.source === 'openrouter' ? OPENROUTER_JEV_MODEL : DEFAULT_JEV_MODEL);
|
|
86
|
+
|
|
87
|
+
const controller = new AbortController();
|
|
88
|
+
const timer = setTimeout(() => controller.abort(), timeoutMs);
|
|
89
|
+
|
|
90
|
+
try {
|
|
91
|
+
const payload = {
|
|
92
|
+
state: String(prompt || ''),
|
|
93
|
+
model: modelName,
|
|
94
|
+
questions: {
|
|
95
|
+
recommended_tier: {
|
|
96
|
+
type: 'choice',
|
|
97
|
+
instructions: 'Determine the optimal model tier for this task according to complexity and reasoning depth required.',
|
|
98
|
+
criteria: TIER_CRITERIA,
|
|
99
|
+
},
|
|
100
|
+
},
|
|
101
|
+
};
|
|
102
|
+
|
|
103
|
+
const res = await fetchImpl(endpointUrl, {
|
|
104
|
+
method: 'POST',
|
|
105
|
+
headers: {
|
|
106
|
+
Authorization: `Bearer ${keyInfo.key}`,
|
|
107
|
+
'Content-Type': 'application/json',
|
|
108
|
+
},
|
|
109
|
+
body: JSON.stringify(payload),
|
|
110
|
+
signal: controller.signal,
|
|
111
|
+
});
|
|
112
|
+
|
|
113
|
+
clearTimeout(timer);
|
|
114
|
+
|
|
115
|
+
if (!res.ok) {
|
|
116
|
+
return {
|
|
117
|
+
tier: fallbackTier,
|
|
118
|
+
confidence: 0.5,
|
|
119
|
+
source: 'heuristic',
|
|
120
|
+
reason: `http-${res.status}`,
|
|
121
|
+
};
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
const data = await res.json();
|
|
125
|
+
const ans = data?.answers?.recommended_tier;
|
|
126
|
+
if (!ans || typeof ans.choice !== 'string') {
|
|
127
|
+
return {
|
|
128
|
+
tier: fallbackTier,
|
|
129
|
+
confidence: 0.5,
|
|
130
|
+
source: 'heuristic',
|
|
131
|
+
reason: 'bad-response',
|
|
132
|
+
};
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
const tier = ['haiku', 'sonnet', 'opus'].includes(ans.choice) ? ans.choice : fallbackTier;
|
|
136
|
+
const confidence = typeof ans.confidence === 'number' ? Number(ans.confidence.toFixed(3)) : 0.8;
|
|
137
|
+
|
|
138
|
+
return {
|
|
139
|
+
tier,
|
|
140
|
+
confidence,
|
|
141
|
+
distribution: ans.distribution || null,
|
|
142
|
+
source: 'jev',
|
|
143
|
+
model: data.model || modelName,
|
|
144
|
+
};
|
|
145
|
+
} catch (err) {
|
|
146
|
+
clearTimeout(timer);
|
|
147
|
+
const reason = err?.name === 'AbortError' ? 'timeout' : 'network';
|
|
148
|
+
return {
|
|
149
|
+
tier: fallbackTier,
|
|
150
|
+
confidence: 0.5,
|
|
151
|
+
source: 'heuristic',
|
|
152
|
+
reason,
|
|
153
|
+
};
|
|
154
|
+
}
|
|
155
|
+
}
|
|
156
|
+
|
|
157
|
+
/**
|
|
158
|
+
* Checks semantic equivalence between candidate text and cached target using Jev Noul.
|
|
159
|
+
* Optional gate: returns false immediately if Jev is not configured or unavailable.
|
|
160
|
+
*/
|
|
161
|
+
export async function checkSemanticEquivalence({
|
|
162
|
+
candidate,
|
|
163
|
+
target,
|
|
164
|
+
apiKey,
|
|
165
|
+
url,
|
|
166
|
+
model,
|
|
167
|
+
timeoutMs = 3000,
|
|
168
|
+
env = process.env,
|
|
169
|
+
fetchImpl = globalThis.fetch,
|
|
170
|
+
} = {}) {
|
|
171
|
+
if (!candidate || !target) return { equivalent: false, confidence: 0, reason: 'empty-input' };
|
|
172
|
+
if (candidate.trim() === target.trim()) {
|
|
173
|
+
return { equivalent: true, confidence: 1.0, reason: 'exact-match' };
|
|
174
|
+
}
|
|
175
|
+
|
|
176
|
+
const keyInfo = apiKey ? { key: apiKey, source: 'custom' } : findJevKey(env);
|
|
177
|
+
if (!keyInfo) {
|
|
178
|
+
return { equivalent: false, confidence: 0, reason: 'no-key-fallback-bypass' };
|
|
179
|
+
}
|
|
180
|
+
|
|
181
|
+
const endpointUrl = url || (keyInfo.source === 'openrouter' ? OPENROUTER_JEV_URL : DEFAULT_JEV_URL);
|
|
182
|
+
const modelName = model || (keyInfo.source === 'openrouter' ? OPENROUTER_JEV_MODEL : DEFAULT_JEV_MODEL);
|
|
183
|
+
|
|
184
|
+
const controller = new AbortController();
|
|
185
|
+
const timer = setTimeout(() => controller.abort(), timeoutMs);
|
|
186
|
+
|
|
187
|
+
try {
|
|
188
|
+
const payload = {
|
|
189
|
+
state: `REQUEST_A:\n${candidate}\n\nREQUEST_B:\n${target}`,
|
|
190
|
+
model: modelName,
|
|
191
|
+
questions: {
|
|
192
|
+
is_equivalent: {
|
|
193
|
+
type: 'noul',
|
|
194
|
+
instructions: 'Do REQUEST_A and REQUEST_B ask for the exact same semantic task or answer, such that the response to B fully satisfies A?',
|
|
195
|
+
},
|
|
196
|
+
},
|
|
197
|
+
};
|
|
198
|
+
|
|
199
|
+
const res = await fetchImpl(endpointUrl, {
|
|
200
|
+
method: 'POST',
|
|
201
|
+
headers: {
|
|
202
|
+
Authorization: `Bearer ${keyInfo.key}`,
|
|
203
|
+
'Content-Type': 'application/json',
|
|
204
|
+
},
|
|
205
|
+
body: JSON.stringify(payload),
|
|
206
|
+
signal: controller.signal,
|
|
207
|
+
});
|
|
208
|
+
|
|
209
|
+
clearTimeout(timer);
|
|
210
|
+
if (!res.ok) {
|
|
211
|
+
return { equivalent: false, confidence: 0, reason: `http-${res.status}` };
|
|
212
|
+
}
|
|
213
|
+
|
|
214
|
+
const data = await res.json();
|
|
215
|
+
const ans = data?.answers?.is_equivalent;
|
|
216
|
+
if (ans == null) {
|
|
217
|
+
return { equivalent: false, confidence: 0, reason: 'bad-response' };
|
|
218
|
+
}
|
|
219
|
+
|
|
220
|
+
// Noul returns either probability number directly or object with probability
|
|
221
|
+
const prob = typeof ans === 'number' ? ans : (typeof ans?.probability === 'number' ? ans.probability : 0);
|
|
222
|
+
const equivalent = prob >= 0.85;
|
|
223
|
+
|
|
224
|
+
return {
|
|
225
|
+
equivalent,
|
|
226
|
+
confidence: Number(prob.toFixed(3)),
|
|
227
|
+
reason: equivalent ? 'jev-confirmed' : 'below-threshold',
|
|
228
|
+
model: data.model || modelName,
|
|
229
|
+
};
|
|
230
|
+
} catch (err) {
|
|
231
|
+
clearTimeout(timer);
|
|
232
|
+
return {
|
|
233
|
+
equivalent: false,
|
|
234
|
+
confidence: 0,
|
|
235
|
+
reason: err?.name === 'AbortError' ? 'timeout' : 'network-error',
|
|
236
|
+
};
|
|
237
|
+
}
|
|
238
|
+
}
|
|
@@ -1,110 +1,110 @@
|
|
|
1
|
-
# Token Optimizer Interoperability & Failure Mode Report
|
|
2
|
-
|
|
3
|
-
**How LLM Switcher acts as the protective outermost edge gateway for aggressive prompt/token optimizers (Headroom, RTK, Ponytail).**
|
|
4
|
-
|
|
5
|
-
---
|
|
6
|
-
|
|
7
|
-
## Executive Summary
|
|
8
|
-
|
|
9
|
-
Third-party prompt optimizers and token compressors — such as **Headroom**, **RTK (Rust Token Killer)**, and **Ponytail** — attempt to reduce LLM input tokens by aggressively pruning message history, truncating command stdout, or forcing extreme prompt brevity.
|
|
10
|
-
|
|
11
|
-
While these tools can reduce raw token counts in simple scenarios, **they frequently break complex agentic coding workflows** by corrupting message graphs, orphaning tool calls, and stripping reasoning parameters. When these pruned payloads hit upstream APIs directly (such as Anthropic, OpenAI, or 9Router), the provider immediately throws fatal `HTTP 400 Bad Request` errors or severely degrades reasoning depth.
|
|
12
|
-
|
|
13
|
-
**LLM Switcher solves this by acting as the outermost edge gatekeeper (`127.0.0.1:3456`).** It intercepts the pruned payload before it leaves your machine, runs its built-in **Healer Engine** to repair message graphs and restore reasoning parameters, unlocks 1M context windows, and safely converts the protocol to your upstream provider.
|
|
14
|
-
|
|
15
|
-
---
|
|
16
|
-
|
|
17
|
-
## Tool Breakdown: What They Do & How They Break Payloads
|
|
18
|
-
|
|
19
|
-
### 1. Headroom (`headroomlabs-ai/headroom`)
|
|
20
|
-
- **Mechanism:** Runs as a local proxy on `:8787` (or wraps CLI agents). Compresses conversation history, RAG chunks, and tool outputs using SmartCrusher (JSON), CodeCompressor (AST), and Kompress-v2-base. Also attempts "effort routing" to dial down thinking budgets.
|
|
21
|
-
- **Critical Failure Points:**
|
|
22
|
-
- **Orphaned `tool_result` blocks:** When pruning historical turns, Headroom often discards the `assistant` turn containing a `tool_use`, while retaining the subsequent `user` turn containing the `tool_result`. Anthropic's API strictly validates tool use IDs and crashes with:
|
|
23
|
-
```
|
|
24
|
-
HTTP 400 invalid_request_error: "tool_use_id 'xxx' does not correspond to any tool_use"
|
|
25
|
-
```
|
|
26
|
-
- **Consecutive `user` turns:** Dropping intermediary assistant turns causes multiple user messages to sit adjacent to each other. Anthropic strictly throws:
|
|
27
|
-
```
|
|
28
|
-
HTTP 400 invalid_request_error: "roles must alternate between 'user' and 'assistant'"
|
|
29
|
-
```
|
|
30
|
-
- **Reasoning Suppression:** Its "effort routing" dials down `thinking.budget_tokens` on routine tool turns. On complex models (Claude Opus, Gemini Flash), this prevents the model from formulating multi-step reasoning before acting.
|
|
31
|
-
|
|
32
|
-
### 2. RTK (`rtk-ai/rtk` - Rust Token Killer)
|
|
33
|
-
- **Mechanism:** A single Rust binary that hooks into shell tool execution (e.g. `PreToolUse` in Claude Code / Cursor) and rewrites CLI commands (`git`, `ls`, `cat`, `grep`, `pytest`) to filter out noise, truncate lines, and inject recall tokens (`[full output: rtk recall xxx]`).
|
|
34
|
-
- **Critical Failure Points:**
|
|
35
|
-
- **Corrupted Structural Data:** When an agent invokes a tool expecting machine-readable JSON or exact AST formatting, RTK's heuristic summaries can alter structural delimiters, causing downstream tool call parsing errors.
|
|
36
|
-
- **Custom Tracking Headers:** RTK and associated tracing proxies inject headers (`x-rtk-*`, `traceparent`, `x-optimizer-id`) that some strict upstream endpoints reject if not cleanly forwarded.
|
|
37
|
-
|
|
38
|
-
### 3. Ponytail (`DietrichGebert/ponytail`)
|
|
39
|
-
- **Mechanism:** A behavioral prompt engineering plugin/ruleset that injects extreme conciseness instructions ("write one line, it works, YAGNI") into agent system prompts across 20+ coding tools.
|
|
40
|
-
- **Critical Failure Points:**
|
|
41
|
-
- **Premature Execution without Reasoning:** By commanding the model to be maximally brief and avoid planning, reasoning models are discouraged from spending thinking tokens. The model outputs untested single-liners that often fail type checks and test suites.
|
|
42
|
-
- **System Prompt Prefix Invalidation:** Injected rules alter the leading system prompt bytes, busting provider prompt caches unless carefully aligned.
|
|
43
|
-
|
|
44
|
-
---
|
|
45
|
-
|
|
46
|
-
## The Healer Engine: How LLM Switcher Protects the Workflow
|
|
47
|
-
|
|
48
|
-
LLM Switcher sits between the optimizer tool and the upstream LLM:
|
|
49
|
-
|
|
50
|
-
```
|
|
51
|
-
[CLI Agent] ──> [Optimizer: Headroom / RTK] ──> [LLM Switcher :3456] ──> [Upstream / 9Router]
|
|
52
|
-
│
|
|
53
|
-
├── 1. Heal Orphaned tool_results
|
|
54
|
-
├── 2. Merge Consecutive Turns
|
|
55
|
-
├── 3. Restore Stripped Thinking
|
|
56
|
-
└── 4. Enforce 1M Context
|
|
57
|
-
```
|
|
58
|
-
|
|
59
|
-
### Protection Matrix: Before vs. After
|
|
60
|
-
|
|
61
|
-
| Scenario | Direct to Upstream (Without Switcher) | Through LLM Switcher (Healer Engine) |
|
|
62
|
-
|---|---|---|
|
|
63
|
-
| **Orphaned `tool_result` turn** | ❌ **HTTP 400 Crash**: `tool_use_id does not correspond to any tool_use` | ✅ **HTTP 200 OK**: Heals orphaned result into contextual text block `[Tool Result (id)]: ...` |
|
|
64
|
-
| **Consecutive `user` turns** | ❌ **HTTP 400 Crash**: `roles must alternate` | ✅ **HTTP 200 OK**: Merges consecutive turns into a single valid turn seamlessly |
|
|
65
|
-
| **Stripped `thinking` parameters** | ⚠️ **Degraded AI**: Reasoning disabled, model outputs shallow single-liners | ✅ **HTTP 200 OK**: Detects reasoning models and automatically restores safe thinking budget |
|
|
66
|
-
| **Orphaned `tool` role in Chat API** | ❌ **HTTP 400 Crash**: `tool role must respond to tool_calls` | ✅ **HTTP 200 OK**: Converts orphaned tool message into user context |
|
|
67
|
-
| **Custom tracking headers** | ⚠️ Connection dropped / unrecognized header warnings | ✅ **HTTP 200 OK**: Cleanly passes through `traceparent`, `x-request-id`, `x-rtk-*` |
|
|
68
|
-
|
|
69
|
-
---
|
|
70
|
-
|
|
71
|
-
## Test Methodology & Verification Suite
|
|
72
|
-
|
|
73
|
-
We created an automated verification suite in [`tests/live-optimizer-interop.mjs`](../tests/live-optimizer-interop.mjs) that systematically replicates the failure modes of each tool:
|
|
74
|
-
|
|
75
|
-
### Running the Test Suite
|
|
76
|
-
```bash
|
|
77
|
-
node tests/live-optimizer-interop.mjs
|
|
78
|
-
```
|
|
79
|
-
|
|
80
|
-
### Live Test Results
|
|
81
|
-
```text
|
|
82
|
-
================================================================
|
|
83
|
-
LLM SWITCHER — TOKEN OPTIMIZER INTEROPERABILITY TEST SUITE
|
|
84
|
-
Simulating failure modes from Headroom, RTK, and Ponytail
|
|
85
|
-
================================================================
|
|
86
|
-
|
|
87
|
-
[TEST] Headroom Simulation: Orphaned tool_result turn... PASS (8840ms)
|
|
88
|
-
↳ Healed orphaned tool_result. Response HTTP 200: "It looks like you've shared a fragment of context ..."
|
|
89
|
-
[TEST] Headroom Simulation: Consecutive User turns (Role alternation violation)... PASS (2683ms)
|
|
90
|
-
↳ Merged consecutive turns seamlessly. Stop reason: end_turn
|
|
91
|
-
[TEST] Thinking Guard: Restoring stripped thinking parameter on reasoning models... PASS (5372ms)
|
|
92
|
-
↳ Automatically restored thinking: thinking_delta=true, signature_delta=true, text_delta=true
|
|
93
|
-
[TEST] RTK Intermediary: Custom headers and traceparent passthrough... PASS (2453ms)
|
|
94
|
-
↳ Headers accepted cleanly with HTTP 200 OK
|
|
95
|
-
[TEST] OpenAI Chat Healer: Orphaned role tool without preceding assistant tool_calls... PASS (1561ms)
|
|
96
|
-
↳ Chat Healer rescued orphaned tool role. Stop reason: length
|
|
97
|
-
|
|
98
|
-
================================================================
|
|
99
|
-
TEST RESULTS: 5 PASSED / 0 FAILED
|
|
100
|
-
================================================================
|
|
101
|
-
```
|
|
102
|
-
|
|
103
|
-
---
|
|
104
|
-
|
|
105
|
-
## Recommended User Setup
|
|
106
|
-
|
|
107
|
-
For developers using prompt optimization tools:
|
|
108
|
-
1. Keep the optimizer installed in your CLI tool as usual.
|
|
109
|
-
2. In the optimizer's configuration (e.g. `headroom.yaml` or RTK upstream settings), set the upstream target URL to **LLM Switcher** (`http://127.0.0.1:3456`).
|
|
110
|
-
3. Enjoy prompt compression savings without worrying about broken conversation graphs, HTTP 400 crashes, or lost reasoning depth.
|
|
1
|
+
# Token Optimizer Interoperability & Failure Mode Report
|
|
2
|
+
|
|
3
|
+
**How LLM Switcher acts as the protective outermost edge gateway for aggressive prompt/token optimizers (Headroom, RTK, Ponytail).**
|
|
4
|
+
|
|
5
|
+
---
|
|
6
|
+
|
|
7
|
+
## Executive Summary
|
|
8
|
+
|
|
9
|
+
Third-party prompt optimizers and token compressors — such as **Headroom**, **RTK (Rust Token Killer)**, and **Ponytail** — attempt to reduce LLM input tokens by aggressively pruning message history, truncating command stdout, or forcing extreme prompt brevity.
|
|
10
|
+
|
|
11
|
+
While these tools can reduce raw token counts in simple scenarios, **they frequently break complex agentic coding workflows** by corrupting message graphs, orphaning tool calls, and stripping reasoning parameters. When these pruned payloads hit upstream APIs directly (such as Anthropic, OpenAI, or 9Router), the provider immediately throws fatal `HTTP 400 Bad Request` errors or severely degrades reasoning depth.
|
|
12
|
+
|
|
13
|
+
**LLM Switcher solves this by acting as the outermost edge gatekeeper (`127.0.0.1:3456`).** It intercepts the pruned payload before it leaves your machine, runs its built-in **Healer Engine** to repair message graphs and restore reasoning parameters, unlocks 1M context windows, and safely converts the protocol to your upstream provider.
|
|
14
|
+
|
|
15
|
+
---
|
|
16
|
+
|
|
17
|
+
## Tool Breakdown: What They Do & How They Break Payloads
|
|
18
|
+
|
|
19
|
+
### 1. Headroom (`headroomlabs-ai/headroom`)
|
|
20
|
+
- **Mechanism:** Runs as a local proxy on `:8787` (or wraps CLI agents). Compresses conversation history, RAG chunks, and tool outputs using SmartCrusher (JSON), CodeCompressor (AST), and Kompress-v2-base. Also attempts "effort routing" to dial down thinking budgets.
|
|
21
|
+
- **Critical Failure Points:**
|
|
22
|
+
- **Orphaned `tool_result` blocks:** When pruning historical turns, Headroom often discards the `assistant` turn containing a `tool_use`, while retaining the subsequent `user` turn containing the `tool_result`. Anthropic's API strictly validates tool use IDs and crashes with:
|
|
23
|
+
```
|
|
24
|
+
HTTP 400 invalid_request_error: "tool_use_id 'xxx' does not correspond to any tool_use"
|
|
25
|
+
```
|
|
26
|
+
- **Consecutive `user` turns:** Dropping intermediary assistant turns causes multiple user messages to sit adjacent to each other. Anthropic strictly throws:
|
|
27
|
+
```
|
|
28
|
+
HTTP 400 invalid_request_error: "roles must alternate between 'user' and 'assistant'"
|
|
29
|
+
```
|
|
30
|
+
- **Reasoning Suppression:** Its "effort routing" dials down `thinking.budget_tokens` on routine tool turns. On complex models (Claude Opus, Gemini Flash), this prevents the model from formulating multi-step reasoning before acting.
|
|
31
|
+
|
|
32
|
+
### 2. RTK (`rtk-ai/rtk` - Rust Token Killer)
|
|
33
|
+
- **Mechanism:** A single Rust binary that hooks into shell tool execution (e.g. `PreToolUse` in Claude Code / Cursor) and rewrites CLI commands (`git`, `ls`, `cat`, `grep`, `pytest`) to filter out noise, truncate lines, and inject recall tokens (`[full output: rtk recall xxx]`).
|
|
34
|
+
- **Critical Failure Points:**
|
|
35
|
+
- **Corrupted Structural Data:** When an agent invokes a tool expecting machine-readable JSON or exact AST formatting, RTK's heuristic summaries can alter structural delimiters, causing downstream tool call parsing errors.
|
|
36
|
+
- **Custom Tracking Headers:** RTK and associated tracing proxies inject headers (`x-rtk-*`, `traceparent`, `x-optimizer-id`) that some strict upstream endpoints reject if not cleanly forwarded.
|
|
37
|
+
|
|
38
|
+
### 3. Ponytail (`DietrichGebert/ponytail`)
|
|
39
|
+
- **Mechanism:** A behavioral prompt engineering plugin/ruleset that injects extreme conciseness instructions ("write one line, it works, YAGNI") into agent system prompts across 20+ coding tools.
|
|
40
|
+
- **Critical Failure Points:**
|
|
41
|
+
- **Premature Execution without Reasoning:** By commanding the model to be maximally brief and avoid planning, reasoning models are discouraged from spending thinking tokens. The model outputs untested single-liners that often fail type checks and test suites.
|
|
42
|
+
- **System Prompt Prefix Invalidation:** Injected rules alter the leading system prompt bytes, busting provider prompt caches unless carefully aligned.
|
|
43
|
+
|
|
44
|
+
---
|
|
45
|
+
|
|
46
|
+
## The Healer Engine: How LLM Switcher Protects the Workflow
|
|
47
|
+
|
|
48
|
+
LLM Switcher sits between the optimizer tool and the upstream LLM:
|
|
49
|
+
|
|
50
|
+
```
|
|
51
|
+
[CLI Agent] ──> [Optimizer: Headroom / RTK] ──> [LLM Switcher :3456] ──> [Upstream / 9Router]
|
|
52
|
+
│
|
|
53
|
+
├── 1. Heal Orphaned tool_results
|
|
54
|
+
├── 2. Merge Consecutive Turns
|
|
55
|
+
├── 3. Restore Stripped Thinking
|
|
56
|
+
└── 4. Enforce 1M Context
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
### Protection Matrix: Before vs. After
|
|
60
|
+
|
|
61
|
+
| Scenario | Direct to Upstream (Without Switcher) | Through LLM Switcher (Healer Engine) |
|
|
62
|
+
|---|---|---|
|
|
63
|
+
| **Orphaned `tool_result` turn** | ❌ **HTTP 400 Crash**: `tool_use_id does not correspond to any tool_use` | ✅ **HTTP 200 OK**: Heals orphaned result into contextual text block `[Tool Result (id)]: ...` |
|
|
64
|
+
| **Consecutive `user` turns** | ❌ **HTTP 400 Crash**: `roles must alternate` | ✅ **HTTP 200 OK**: Merges consecutive turns into a single valid turn seamlessly |
|
|
65
|
+
| **Stripped `thinking` parameters** | ⚠️ **Degraded AI**: Reasoning disabled, model outputs shallow single-liners | ✅ **HTTP 200 OK**: Detects reasoning models and automatically restores safe thinking budget |
|
|
66
|
+
| **Orphaned `tool` role in Chat API** | ❌ **HTTP 400 Crash**: `tool role must respond to tool_calls` | ✅ **HTTP 200 OK**: Converts orphaned tool message into user context |
|
|
67
|
+
| **Custom tracking headers** | ⚠️ Connection dropped / unrecognized header warnings | ✅ **HTTP 200 OK**: Cleanly passes through `traceparent`, `x-request-id`, `x-rtk-*` |
|
|
68
|
+
|
|
69
|
+
---
|
|
70
|
+
|
|
71
|
+
## Test Methodology & Verification Suite
|
|
72
|
+
|
|
73
|
+
We created an automated verification suite in [`tests/live-optimizer-interop.mjs`](../tests/live-optimizer-interop.mjs) that systematically replicates the failure modes of each tool:
|
|
74
|
+
|
|
75
|
+
### Running the Test Suite
|
|
76
|
+
```bash
|
|
77
|
+
node tests/live-optimizer-interop.mjs
|
|
78
|
+
```
|
|
79
|
+
|
|
80
|
+
### Live Test Results
|
|
81
|
+
```text
|
|
82
|
+
================================================================
|
|
83
|
+
LLM SWITCHER — TOKEN OPTIMIZER INTEROPERABILITY TEST SUITE
|
|
84
|
+
Simulating failure modes from Headroom, RTK, and Ponytail
|
|
85
|
+
================================================================
|
|
86
|
+
|
|
87
|
+
[TEST] Headroom Simulation: Orphaned tool_result turn... PASS (8840ms)
|
|
88
|
+
↳ Healed orphaned tool_result. Response HTTP 200: "It looks like you've shared a fragment of context ..."
|
|
89
|
+
[TEST] Headroom Simulation: Consecutive User turns (Role alternation violation)... PASS (2683ms)
|
|
90
|
+
↳ Merged consecutive turns seamlessly. Stop reason: end_turn
|
|
91
|
+
[TEST] Thinking Guard: Restoring stripped thinking parameter on reasoning models... PASS (5372ms)
|
|
92
|
+
↳ Automatically restored thinking: thinking_delta=true, signature_delta=true, text_delta=true
|
|
93
|
+
[TEST] RTK Intermediary: Custom headers and traceparent passthrough... PASS (2453ms)
|
|
94
|
+
↳ Headers accepted cleanly with HTTP 200 OK
|
|
95
|
+
[TEST] OpenAI Chat Healer: Orphaned role tool without preceding assistant tool_calls... PASS (1561ms)
|
|
96
|
+
↳ Chat Healer rescued orphaned tool role. Stop reason: length
|
|
97
|
+
|
|
98
|
+
================================================================
|
|
99
|
+
TEST RESULTS: 5 PASSED / 0 FAILED
|
|
100
|
+
================================================================
|
|
101
|
+
```
|
|
102
|
+
|
|
103
|
+
---
|
|
104
|
+
|
|
105
|
+
## Recommended User Setup
|
|
106
|
+
|
|
107
|
+
For developers using prompt optimization tools:
|
|
108
|
+
1. Keep the optimizer installed in your CLI tool as usual.
|
|
109
|
+
2. In the optimizer's configuration (e.g. `headroom.yaml` or RTK upstream settings), set the upstream target URL to **LLM Switcher** (`http://127.0.0.1:3456`).
|
|
110
|
+
3. Enjoy prompt compression savings without worrying about broken conversation graphs, HTTP 400 crashes, or lost reasoning depth.
|