llm-switcher 1.1.10 → 1.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +51 -0
- package/README.md +202 -257
- package/README.vi.md +200 -256
- package/blindfold/blindfold.mjs +200 -53
- package/blindfold/make-certs.sh +26 -7
- package/catalog.mjs +246 -0
- package/config.example.json +12 -34
- package/docs/TOKEN-OPTIMIZER-INTEROP.md +110 -110
- package/docs/codex-blindfold.md +28 -17
- package/docs/cross-platform.md +16 -7
- package/docs/diagrams/ir-healer-pipeline.mmd +16 -0
- package/docs/diagrams/ir-healer-pipeline.png +0 -0
- package/docs/diagrams/ir-healer-pipeline.svg +90 -0
- package/docs/diagrams/ir-translation-pipeline.html +14925 -0
- package/docs/diagrams/ir-translation-pipeline.sequence.json +31 -0
- package/docs/diagrams/ir-translation-pipeline.svg +5128 -0
- package/docs/diagrams/system-architecture.architecture.json +76 -0
- package/docs/diagrams/system-architecture.html +14978 -0
- package/docs/diagrams/system-architecture.svg +5147 -0
- package/docs/diagrams/system-topology.mmd +30 -0
- package/docs/diagrams/system-topology.png +0 -0
- package/docs/diagrams/system-topology.svg +125 -0
- package/docs/response-matrix.json +1130 -1130
- package/ensure-ca-bundle.mjs +28 -0
- package/formats.mjs +13 -155
- package/mcp.mjs +39 -11
- package/package.json +1 -1
- package/proxy.mjs +92 -27
- package/shim.mjs +200 -57
- package/skills/llm-switcher/SKILL.md +93 -88
- package/state.mjs +1100 -191
- package/switch +0 -0
- package/switch.cmd +2 -2
- package/switch.mjs +228 -53
- package/tests/blindfold-e2e.test.mjs +380 -0
- package/tests/blindfold-task5.test.mjs +429 -0
- package/tests/blindfold.task3.test.mjs +700 -0
- package/tests/blindfold.test.mjs +10 -5
- package/tests/catalog.test.mjs +147 -0
- package/tests/contract-lab.test.mjs +22 -7
- package/tests/formats.test.mjs +33 -46
- package/tests/gateway.e2e.test.mjs +136 -36
- package/tests/helpers.mjs +24 -24
- package/tests/lifecycle.test.mjs +16 -10
- package/tests/live-optimizer-interop.mjs +205 -205
- package/tests/mcp.test.mjs +78 -2
- package/tests/real-user-sim.test.mjs +464 -0
- package/tests/shim.test.mjs +159 -66
- package/tests/state.test.mjs +975 -193
- package/tests/switch.test.mjs +446 -2
- package/ui.html +61 -154
package/catalog.mjs
ADDED
|
@@ -0,0 +1,246 @@
|
|
|
1
|
+
// ============================================================
|
|
2
|
+
// catalog.mjs — dynamic model catalog discovery and slot mapping (Claude Code & Codex)
|
|
3
|
+
//
|
|
4
|
+
// Automatically fetches official model lists from Anthropic / OpenAI or custom
|
|
5
|
+
// upstream endpoints to ensure mapping tables never go stale when providers release
|
|
6
|
+
// new model variants (e.g. gpt-6-sol, claude-opus-5-5).
|
|
7
|
+
// ============================================================
|
|
8
|
+
|
|
9
|
+
import fs from 'node:fs';
|
|
10
|
+
import path from 'node:path';
|
|
11
|
+
|
|
12
|
+
export const OFFICIAL_MODEL_URLS = {
|
|
13
|
+
claude: 'https://api.anthropic.com/v1/models',
|
|
14
|
+
codex: 'https://api.openai.com/v1/models'
|
|
15
|
+
};
|
|
16
|
+
|
|
17
|
+
export const BASELINE_MODELS = {
|
|
18
|
+
claude: {
|
|
19
|
+
opus: ['claude-opus-5-5', 'claude-opus-4-6', 'claude-3-opus-20240229'],
|
|
20
|
+
sonnet: ['claude-sonnet-4', 'claude-3-7-sonnet-20250219', 'claude-3-5-sonnet-20241022'],
|
|
21
|
+
haiku: ['claude-haiku-4', 'claude-3-5-haiku-20241022'],
|
|
22
|
+
fable: ['claude-fable-4']
|
|
23
|
+
},
|
|
24
|
+
codex: {
|
|
25
|
+
main: ['gpt-6-sol', 'gpt-5.6-sol', 'gpt-5.2', 'o3', 'gpt-4o'],
|
|
26
|
+
review: ['gpt-6-terra', 'gpt-5.6-terra', 'o3-mini'],
|
|
27
|
+
subagent: ['gpt-6-luna', 'gpt-5.6-luna', 'gpt-5.3-codex', 'gpt-5.2-codex']
|
|
28
|
+
}
|
|
29
|
+
};
|
|
30
|
+
|
|
31
|
+
/** Classifies a Claude model name into its corresponding tier slot */
|
|
32
|
+
export function classifyClaudeTier(modelId) {
|
|
33
|
+
const m = String(modelId || '').toLowerCase();
|
|
34
|
+
if (m.includes('fable')) return 'fable';
|
|
35
|
+
if (m.includes('opus')) return 'opus';
|
|
36
|
+
if (m.includes('haiku')) return 'haiku';
|
|
37
|
+
return 'sonnet';
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
/** Classifies a Codex model name into its corresponding role slot */
|
|
41
|
+
export function classifyCodexRole(modelId) {
|
|
42
|
+
const m = String(modelId || '').toLowerCase();
|
|
43
|
+
if (m.includes('terra') || m.includes('review')) return 'review';
|
|
44
|
+
if (m.includes('luna') || m.includes('subagent')) return 'subagent';
|
|
45
|
+
return 'main';
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
function catalogCachePath(stateDir) {
|
|
49
|
+
return path.join(stateDir, 'catalog-cache.json');
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
/** Loads cached catalog from disk, falling back to built-in baseline models */
|
|
53
|
+
export function loadCatalogCache(stateDir) {
|
|
54
|
+
const p = catalogCachePath(stateDir);
|
|
55
|
+
try {
|
|
56
|
+
if (fs.existsSync(p)) {
|
|
57
|
+
const data = JSON.parse(fs.readFileSync(p, 'utf8'));
|
|
58
|
+
if (data && typeof data === 'object') return data;
|
|
59
|
+
}
|
|
60
|
+
} catch {}
|
|
61
|
+
return {
|
|
62
|
+
updatedAt: 0,
|
|
63
|
+
claude: {
|
|
64
|
+
models: Object.entries(BASELINE_MODELS.claude).flatMap(([tier, ids]) => ids.map(id => ({ id, tier })))
|
|
65
|
+
},
|
|
66
|
+
codex: {
|
|
67
|
+
models: Object.entries(BASELINE_MODELS.codex).flatMap(([role, ids]) => ids.map(id => ({ id, role })))
|
|
68
|
+
}
|
|
69
|
+
};
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
/** Atomically writes the model catalog cache to disk */
|
|
73
|
+
export function saveCatalogCache(stateDir, catalog) {
|
|
74
|
+
const p = catalogCachePath(stateDir);
|
|
75
|
+
const tmp = `${p}.${process.pid}.tmp`;
|
|
76
|
+
try {
|
|
77
|
+
fs.writeFileSync(tmp, JSON.stringify(catalog, null, 2), 'utf8');
|
|
78
|
+
fs.renameSync(tmp, p);
|
|
79
|
+
return true;
|
|
80
|
+
} catch {
|
|
81
|
+
try { fs.rmSync(tmp, { force: true }); } catch {}
|
|
82
|
+
return false;
|
|
83
|
+
}
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
/**
|
|
87
|
+
* Fetches the latest models for a tool from its official endpoint or a custom URL.
|
|
88
|
+
* Never throws: falls back safely to baseline models if offline or unreachable.
|
|
89
|
+
*/
|
|
90
|
+
export async function fetchToolModels(tool, { url, apiKey, timeout = 3000 } = {}) {
|
|
91
|
+
const targetUrl = url || OFFICIAL_MODEL_URLS[tool];
|
|
92
|
+
if (!targetUrl) return { ok: false, error: `Unknown tool: ${tool}`, models: [] };
|
|
93
|
+
|
|
94
|
+
const headers = {
|
|
95
|
+
'user-agent': 'llm-switcher/catalog',
|
|
96
|
+
'accept': 'application/json'
|
|
97
|
+
};
|
|
98
|
+
if (tool === 'claude') {
|
|
99
|
+
headers['anthropic-version'] = '2023-06-01';
|
|
100
|
+
if (apiKey) headers['x-api-key'] = apiKey;
|
|
101
|
+
} else if (apiKey) {
|
|
102
|
+
headers['authorization'] = `Bearer ${apiKey}`;
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
try {
|
|
106
|
+
const res = await fetch(targetUrl, {
|
|
107
|
+
method: 'GET',
|
|
108
|
+
headers,
|
|
109
|
+
signal: AbortSignal.timeout(timeout)
|
|
110
|
+
});
|
|
111
|
+
if (!res.ok) {
|
|
112
|
+
return { ok: false, status: res.status, error: `HTTP ${res.status}`, models: [] };
|
|
113
|
+
}
|
|
114
|
+
const json = await res.json();
|
|
115
|
+
const rawList = Array.isArray(json.data) ? json.data : Array.isArray(json.models) ? json.models : [];
|
|
116
|
+
|
|
117
|
+
if (tool === 'claude') {
|
|
118
|
+
const models = rawList
|
|
119
|
+
.filter(m => m && (m.id || typeof m === 'string'))
|
|
120
|
+
.map(m => {
|
|
121
|
+
const id = typeof m === 'string' ? m : m.id;
|
|
122
|
+
return { id, display_name: m.display_name || id, tier: classifyClaudeTier(id) };
|
|
123
|
+
});
|
|
124
|
+
return { ok: true, models };
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
if (tool === 'codex') {
|
|
128
|
+
const models = rawList
|
|
129
|
+
.filter(m => m && (m.id || typeof m === 'string'))
|
|
130
|
+
.map(m => {
|
|
131
|
+
const id = typeof m === 'string' ? m : m.id;
|
|
132
|
+
return { id, role: classifyCodexRole(id) };
|
|
133
|
+
})
|
|
134
|
+
.filter(m => /^(gpt|o\d|codex)/i.test(m.id));
|
|
135
|
+
return { ok: true, models };
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
return { ok: true, models: rawList };
|
|
139
|
+
} catch (err) {
|
|
140
|
+
return { ok: false, error: err.message, models: [] };
|
|
141
|
+
}
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
/**
|
|
145
|
+
* Refreshes the local catalog cache with models from both official endpoints
|
|
146
|
+
* and saves to the state directory.
|
|
147
|
+
*/
|
|
148
|
+
export async function refreshCatalog(stateDir, { claudeKey, codexKey } = {}) {
|
|
149
|
+
const cache = loadCatalogCache(stateDir);
|
|
150
|
+
|
|
151
|
+
const [claudeRes, codexRes] = await Promise.all([
|
|
152
|
+
fetchToolModels('claude', { apiKey: claudeKey }),
|
|
153
|
+
fetchToolModels('codex', { apiKey: codexKey })
|
|
154
|
+
]);
|
|
155
|
+
|
|
156
|
+
if (claudeRes.ok && claudeRes.models.length > 0) {
|
|
157
|
+
const existing = new Set(claudeRes.models.map(m => m.id));
|
|
158
|
+
// Keep any baseline models that might be absent from the API
|
|
159
|
+
for (const [tier, ids] of Object.entries(BASELINE_MODELS.claude)) {
|
|
160
|
+
for (const id of ids) {
|
|
161
|
+
if (!existing.has(id)) claudeRes.models.push({ id, tier });
|
|
162
|
+
}
|
|
163
|
+
}
|
|
164
|
+
cache.claude = { models: claudeRes.models };
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
if (codexRes.ok && codexRes.models.length > 0) {
|
|
168
|
+
const existing = new Set(codexRes.models.map(m => m.id));
|
|
169
|
+
for (const [role, ids] of Object.entries(BASELINE_MODELS.codex)) {
|
|
170
|
+
for (const id of ids) {
|
|
171
|
+
if (!existing.has(id)) codexRes.models.push({ id, role });
|
|
172
|
+
}
|
|
173
|
+
}
|
|
174
|
+
cache.codex = { models: codexRes.models };
|
|
175
|
+
}
|
|
176
|
+
|
|
177
|
+
cache.updatedAt = Date.now();
|
|
178
|
+
saveCatalogCache(stateDir, cache);
|
|
179
|
+
return cache;
|
|
180
|
+
}
|
|
181
|
+
|
|
182
|
+
/** Extracts tool version string from client User-Agent or headers */
|
|
183
|
+
export function detectToolVersion(headers = {}, expectedTool) {
|
|
184
|
+
const ua = headers['user-agent'] || headers['User-Agent'] || '';
|
|
185
|
+
const first = String(ua || '').split(' ')[0];
|
|
186
|
+
const slash = first.indexOf('/');
|
|
187
|
+
if (slash >= 0) {
|
|
188
|
+
const prefix = first.slice(0, slash).toLowerCase();
|
|
189
|
+
if (prefix.includes('claude') || prefix.includes('codex') || (expectedTool && prefix.includes(expectedTool))) {
|
|
190
|
+
const version = first.slice(slash + 1);
|
|
191
|
+
if (/^\d{1,5}\.\d{1,5}/.test(version)) return version;
|
|
192
|
+
}
|
|
193
|
+
}
|
|
194
|
+
const m = /(?:claude(?:-cli|-code)?|codex(?:-cli)?)[/v\s]+([0-9]+\.[0-9]+(?:\.[0-9]+)?)/i.exec(ua);
|
|
195
|
+
return m ? m[1] : '';
|
|
196
|
+
}
|
|
197
|
+
|
|
198
|
+
const refreshingTools = new Set();
|
|
199
|
+
|
|
200
|
+
/**
|
|
201
|
+
* Version-triggered auto-poll:
|
|
202
|
+
* When a request arrives with a new tool version not yet seen in cache,
|
|
203
|
+
* immediately records the new version and kicks off an asynchronous background refresh.
|
|
204
|
+
* Subsequent requests with the same version do zero network calls.
|
|
205
|
+
*/
|
|
206
|
+
export function checkVersionAndRefresh(tool, headers, stateDir, apiKey) {
|
|
207
|
+
const version = detectToolVersion(headers, tool);
|
|
208
|
+
if (!version) return;
|
|
209
|
+
|
|
210
|
+
const cache = loadCatalogCache(stateDir);
|
|
211
|
+
const toolEntry = cache[tool] || {};
|
|
212
|
+
const lastVersion = toolEntry.lastSeenVersion || '';
|
|
213
|
+
|
|
214
|
+
if (version !== lastVersion) {
|
|
215
|
+
toolEntry.lastSeenVersion = version;
|
|
216
|
+
cache[tool] = toolEntry;
|
|
217
|
+
saveCatalogCache(stateDir, cache);
|
|
218
|
+
|
|
219
|
+
if (!refreshingTools.has(tool)) {
|
|
220
|
+
refreshingTools.add(tool);
|
|
221
|
+
fetchToolModels(tool, { apiKey })
|
|
222
|
+
.then(res => {
|
|
223
|
+
if (res.ok && res.models.length > 0) {
|
|
224
|
+
const fresh = loadCatalogCache(stateDir);
|
|
225
|
+
const existing = new Set(res.models.map(m => m.id));
|
|
226
|
+
const baseline = BASELINE_MODELS[tool] || {};
|
|
227
|
+
for (const [, ids] of Object.entries(baseline)) {
|
|
228
|
+
for (const id of ids) {
|
|
229
|
+
if (!existing.has(id)) {
|
|
230
|
+
res.models.push({ id, ...(tool === 'claude' ? { tier: classifyClaudeTier(id) } : { role: classifyCodexRole(id) }) });
|
|
231
|
+
}
|
|
232
|
+
}
|
|
233
|
+
}
|
|
234
|
+
fresh[tool] = { lastSeenVersion: version, models: res.models };
|
|
235
|
+
fresh.updatedAt = Date.now();
|
|
236
|
+
saveCatalogCache(stateDir, fresh);
|
|
237
|
+
console.log(`[llm-switcher:catalog] Detected ${tool} version update to ${version} -> refreshed model catalog (${res.models.length} models)`);
|
|
238
|
+
}
|
|
239
|
+
})
|
|
240
|
+
.catch(() => {})
|
|
241
|
+
.finally(() => {
|
|
242
|
+
refreshingTools.delete(tool);
|
|
243
|
+
});
|
|
244
|
+
}
|
|
245
|
+
}
|
|
246
|
+
}
|
package/config.example.json
CHANGED
|
@@ -1,19 +1,19 @@
|
|
|
1
1
|
{
|
|
2
2
|
"port": 3456,
|
|
3
|
-
"activeProfile": "9router",
|
|
4
3
|
"activeProfiles": {
|
|
5
|
-
"
|
|
6
|
-
"
|
|
7
|
-
|
|
8
|
-
|
|
4
|
+
"claude": "claude-default",
|
|
5
|
+
"codex": "codex-default"
|
|
6
|
+
},
|
|
7
|
+
"blindfold": {
|
|
8
|
+
"port": 3457
|
|
9
9
|
},
|
|
10
10
|
"profiles": {
|
|
11
|
-
"
|
|
12
|
-
"name": "
|
|
11
|
+
"claude-default": {
|
|
12
|
+
"name": "Claude default (Claude client -> router)",
|
|
13
13
|
"mode": "convert",
|
|
14
|
-
"
|
|
14
|
+
"tool": "claude",
|
|
15
15
|
"outFormat": "openai-chat",
|
|
16
|
-
"baseURL": "https://YOUR-
|
|
16
|
+
"baseURL": "https://YOUR-ROUTER-HOST/v1",
|
|
17
17
|
"apiKey": "sk-REPLACE-ME",
|
|
18
18
|
"defaultModels": {
|
|
19
19
|
"opus": "ag/claude-opus-4-6-thinking",
|
|
@@ -28,30 +28,10 @@
|
|
|
28
28
|
"fable": true
|
|
29
29
|
}
|
|
30
30
|
},
|
|
31
|
-
"
|
|
32
|
-
"name": "
|
|
33
|
-
"mode": "direct",
|
|
34
|
-
"inFormat": "auto",
|
|
35
|
-
"outFormat": "anthropic",
|
|
36
|
-
"baseURL": "https://api.anthropic.com/v1",
|
|
37
|
-
"apiKey": "sk-ant-REPLACE-ME",
|
|
38
|
-
"defaultModels": {
|
|
39
|
-
"opus": "claude-opus-4-7",
|
|
40
|
-
"sonnet": "claude-sonnet-4-6",
|
|
41
|
-
"haiku": "claude-haiku-4-5-20251001",
|
|
42
|
-
"fable": "claude-haiku-4-5-20251001"
|
|
43
|
-
},
|
|
44
|
-
"model1M": {
|
|
45
|
-
"opus": false,
|
|
46
|
-
"sonnet": false,
|
|
47
|
-
"haiku": false,
|
|
48
|
-
"fable": false
|
|
49
|
-
}
|
|
50
|
-
},
|
|
51
|
-
"example-codex-vertex": {
|
|
52
|
-
"name": "Example Codex client -> Vertex upstream",
|
|
31
|
+
"codex-default": {
|
|
32
|
+
"name": "Codex default (Codex client -> Vertex upstream)",
|
|
53
33
|
"mode": "convert",
|
|
54
|
-
"
|
|
34
|
+
"tool": "codex",
|
|
55
35
|
"outFormat": "vertex",
|
|
56
36
|
"baseURL": "https://YOUR-VERTEX-GATEWAY/v1",
|
|
57
37
|
"apiKey": "REPLACE-ME",
|
|
@@ -61,8 +41,6 @@
|
|
|
61
41
|
"review": "gpt-5.6-terra",
|
|
62
42
|
"subagent": "gpt-5.6-luna"
|
|
63
43
|
},
|
|
64
|
-
"blindfold": false,
|
|
65
|
-
"blindfoldPort": 3457,
|
|
66
44
|
"defaultModels": {
|
|
67
45
|
"main": "gemini-3.8-flash",
|
|
68
46
|
"review": "gemini-3.7-flash-medium",
|
|
@@ -1,110 +1,110 @@
|
|
|
1
|
-
# Token Optimizer Interoperability & Failure Mode Report
|
|
2
|
-
|
|
3
|
-
**How LLM Switcher acts as the protective outermost edge gateway for aggressive prompt/token optimizers (Headroom, RTK, Ponytail).**
|
|
4
|
-
|
|
5
|
-
---
|
|
6
|
-
|
|
7
|
-
## Executive Summary
|
|
8
|
-
|
|
9
|
-
Third-party prompt optimizers and token compressors — such as **Headroom**, **RTK (Rust Token Killer)**, and **Ponytail** — attempt to reduce LLM input tokens by aggressively pruning message history, truncating command stdout, or forcing extreme prompt brevity.
|
|
10
|
-
|
|
11
|
-
While these tools can reduce raw token counts in simple scenarios, **they frequently break complex agentic coding workflows** by corrupting message graphs, orphaning tool calls, and stripping reasoning parameters. When these pruned payloads hit upstream APIs directly (such as Anthropic, OpenAI, or 9Router), the provider immediately throws fatal `HTTP 400 Bad Request` errors or severely degrades reasoning depth.
|
|
12
|
-
|
|
13
|
-
**LLM Switcher solves this by acting as the outermost edge gatekeeper (`127.0.0.1:3456`).** It intercepts the pruned payload before it leaves your machine, runs its built-in **Healer Engine** to repair message graphs and restore reasoning parameters, unlocks 1M context windows, and safely converts the protocol to your upstream provider.
|
|
14
|
-
|
|
15
|
-
---
|
|
16
|
-
|
|
17
|
-
## Tool Breakdown: What They Do & How They Break Payloads
|
|
18
|
-
|
|
19
|
-
### 1. Headroom (`headroomlabs-ai/headroom`)
|
|
20
|
-
- **Mechanism:** Runs as a local proxy on `:8787` (or wraps CLI agents). Compresses conversation history, RAG chunks, and tool outputs using SmartCrusher (JSON), CodeCompressor (AST), and Kompress-v2-base. Also attempts "effort routing" to dial down thinking budgets.
|
|
21
|
-
- **Critical Failure Points:**
|
|
22
|
-
- **Orphaned `tool_result` blocks:** When pruning historical turns, Headroom often discards the `assistant` turn containing a `tool_use`, while retaining the subsequent `user` turn containing the `tool_result`. Anthropic's API strictly validates tool use IDs and crashes with:
|
|
23
|
-
```
|
|
24
|
-
HTTP 400 invalid_request_error: "tool_use_id 'xxx' does not correspond to any tool_use"
|
|
25
|
-
```
|
|
26
|
-
- **Consecutive `user` turns:** Dropping intermediary assistant turns causes multiple user messages to sit adjacent to each other. Anthropic strictly throws:
|
|
27
|
-
```
|
|
28
|
-
HTTP 400 invalid_request_error: "roles must alternate between 'user' and 'assistant'"
|
|
29
|
-
```
|
|
30
|
-
- **Reasoning Suppression:** Its "effort routing" dials down `thinking.budget_tokens` on routine tool turns. On complex models (Claude Opus, Gemini Flash), this prevents the model from formulating multi-step reasoning before acting.
|
|
31
|
-
|
|
32
|
-
### 2. RTK (`rtk-ai/rtk` - Rust Token Killer)
|
|
33
|
-
- **Mechanism:** A single Rust binary that hooks into shell tool execution (e.g. `PreToolUse` in Claude Code / Cursor) and rewrites CLI commands (`git`, `ls`, `cat`, `grep`, `pytest`) to filter out noise, truncate lines, and inject recall tokens (`[full output: rtk recall xxx]`).
|
|
34
|
-
- **Critical Failure Points:**
|
|
35
|
-
- **Corrupted Structural Data:** When an agent invokes a tool expecting machine-readable JSON or exact AST formatting, RTK's heuristic summaries can alter structural delimiters, causing downstream tool call parsing errors.
|
|
36
|
-
- **Custom Tracking Headers:** RTK and associated tracing proxies inject headers (`x-rtk-*`, `traceparent`, `x-optimizer-id`) that some strict upstream endpoints reject if not cleanly forwarded.
|
|
37
|
-
|
|
38
|
-
### 3. Ponytail (`DietrichGebert/ponytail`)
|
|
39
|
-
- **Mechanism:** A behavioral prompt engineering plugin/ruleset that injects extreme conciseness instructions ("write one line, it works, YAGNI") into agent system prompts across 20+ coding tools.
|
|
40
|
-
- **Critical Failure Points:**
|
|
41
|
-
- **Premature Execution without Reasoning:** By commanding the model to be maximally brief and avoid planning, reasoning models are discouraged from spending thinking tokens. The model outputs untested single-liners that often fail type checks and test suites.
|
|
42
|
-
- **System Prompt Prefix Invalidation:** Injected rules alter the leading system prompt bytes, busting provider prompt caches unless carefully aligned.
|
|
43
|
-
|
|
44
|
-
---
|
|
45
|
-
|
|
46
|
-
## The Healer Engine: How LLM Switcher Protects the Workflow
|
|
47
|
-
|
|
48
|
-
LLM Switcher sits between the optimizer tool and the upstream LLM:
|
|
49
|
-
|
|
50
|
-
```
|
|
51
|
-
[CLI Agent] ──> [Optimizer: Headroom / RTK] ──> [LLM Switcher :3456] ──> [Upstream / 9Router]
|
|
52
|
-
│
|
|
53
|
-
├── 1. Heal Orphaned tool_results
|
|
54
|
-
├── 2. Merge Consecutive Turns
|
|
55
|
-
├── 3. Restore Stripped Thinking
|
|
56
|
-
└── 4. Enforce 1M Context
|
|
57
|
-
```
|
|
58
|
-
|
|
59
|
-
### Protection Matrix: Before vs. After
|
|
60
|
-
|
|
61
|
-
| Scenario | Direct to Upstream (Without Switcher) | Through LLM Switcher (Healer Engine) |
|
|
62
|
-
|---|---|---|
|
|
63
|
-
| **Orphaned `tool_result` turn** | ❌ **HTTP 400 Crash**: `tool_use_id does not correspond to any tool_use` | ✅ **HTTP 200 OK**: Heals orphaned result into contextual text block `[Tool Result (id)]: ...` |
|
|
64
|
-
| **Consecutive `user` turns** | ❌ **HTTP 400 Crash**: `roles must alternate` | ✅ **HTTP 200 OK**: Merges consecutive turns into a single valid turn seamlessly |
|
|
65
|
-
| **Stripped `thinking` parameters** | ⚠️ **Degraded AI**: Reasoning disabled, model outputs shallow single-liners | ✅ **HTTP 200 OK**: Detects reasoning models and automatically restores safe thinking budget |
|
|
66
|
-
| **Orphaned `tool` role in Chat API** | ❌ **HTTP 400 Crash**: `tool role must respond to tool_calls` | ✅ **HTTP 200 OK**: Converts orphaned tool message into user context |
|
|
67
|
-
| **Custom tracking headers** | ⚠️ Connection dropped / unrecognized header warnings | ✅ **HTTP 200 OK**: Cleanly passes through `traceparent`, `x-request-id`, `x-rtk-*` |
|
|
68
|
-
|
|
69
|
-
---
|
|
70
|
-
|
|
71
|
-
## Test Methodology & Verification Suite
|
|
72
|
-
|
|
73
|
-
We created an automated verification suite in [`tests/live-optimizer-interop.mjs`](../tests/live-optimizer-interop.mjs) that systematically replicates the failure modes of each tool:
|
|
74
|
-
|
|
75
|
-
### Running the Test Suite
|
|
76
|
-
```bash
|
|
77
|
-
node tests/live-optimizer-interop.mjs
|
|
78
|
-
```
|
|
79
|
-
|
|
80
|
-
### Live Test Results
|
|
81
|
-
```text
|
|
82
|
-
================================================================
|
|
83
|
-
LLM SWITCHER — TOKEN OPTIMIZER INTEROPERABILITY TEST SUITE
|
|
84
|
-
Simulating failure modes from Headroom, RTK, and Ponytail
|
|
85
|
-
================================================================
|
|
86
|
-
|
|
87
|
-
[TEST] Headroom Simulation: Orphaned tool_result turn... PASS (8840ms)
|
|
88
|
-
↳ Healed orphaned tool_result. Response HTTP 200: "It looks like you've shared a fragment of context ..."
|
|
89
|
-
[TEST] Headroom Simulation: Consecutive User turns (Role alternation violation)... PASS (2683ms)
|
|
90
|
-
↳ Merged consecutive turns seamlessly. Stop reason: end_turn
|
|
91
|
-
[TEST] Thinking Guard: Restoring stripped thinking parameter on reasoning models... PASS (5372ms)
|
|
92
|
-
↳ Automatically restored thinking: thinking_delta=true, signature_delta=true, text_delta=true
|
|
93
|
-
[TEST] RTK Intermediary: Custom headers and traceparent passthrough... PASS (2453ms)
|
|
94
|
-
↳ Headers accepted cleanly with HTTP 200 OK
|
|
95
|
-
[TEST] OpenAI Chat Healer: Orphaned role tool without preceding assistant tool_calls... PASS (1561ms)
|
|
96
|
-
↳ Chat Healer rescued orphaned tool role. Stop reason: length
|
|
97
|
-
|
|
98
|
-
================================================================
|
|
99
|
-
TEST RESULTS: 5 PASSED / 0 FAILED
|
|
100
|
-
================================================================
|
|
101
|
-
```
|
|
102
|
-
|
|
103
|
-
---
|
|
104
|
-
|
|
105
|
-
## Recommended User Setup
|
|
106
|
-
|
|
107
|
-
For developers using prompt optimization tools:
|
|
108
|
-
1. Keep the optimizer installed in your CLI tool as usual.
|
|
109
|
-
2. In the optimizer's configuration (e.g. `headroom.yaml` or RTK upstream settings), set the upstream target URL to **LLM Switcher** (`http://127.0.0.1:3456`).
|
|
110
|
-
3. Enjoy prompt compression savings without worrying about broken conversation graphs, HTTP 400 crashes, or lost reasoning depth.
|
|
1
|
+
# Token Optimizer Interoperability & Failure Mode Report
|
|
2
|
+
|
|
3
|
+
**How LLM Switcher acts as the protective outermost edge gateway for aggressive prompt/token optimizers (Headroom, RTK, Ponytail).**
|
|
4
|
+
|
|
5
|
+
---
|
|
6
|
+
|
|
7
|
+
## Executive Summary
|
|
8
|
+
|
|
9
|
+
Third-party prompt optimizers and token compressors — such as **Headroom**, **RTK (Rust Token Killer)**, and **Ponytail** — attempt to reduce LLM input tokens by aggressively pruning message history, truncating command stdout, or forcing extreme prompt brevity.
|
|
10
|
+
|
|
11
|
+
While these tools can reduce raw token counts in simple scenarios, **they frequently break complex agentic coding workflows** by corrupting message graphs, orphaning tool calls, and stripping reasoning parameters. When these pruned payloads hit upstream APIs directly (such as Anthropic, OpenAI, or 9Router), the provider immediately throws fatal `HTTP 400 Bad Request` errors or severely degrades reasoning depth.
|
|
12
|
+
|
|
13
|
+
**LLM Switcher solves this by acting as the outermost edge gatekeeper (`127.0.0.1:3456`).** It intercepts the pruned payload before it leaves your machine, runs its built-in **Healer Engine** to repair message graphs and restore reasoning parameters, unlocks 1M context windows, and safely converts the protocol to your upstream provider.
|
|
14
|
+
|
|
15
|
+
---
|
|
16
|
+
|
|
17
|
+
## Tool Breakdown: What They Do & How They Break Payloads
|
|
18
|
+
|
|
19
|
+
### 1. Headroom (`headroomlabs-ai/headroom`)
|
|
20
|
+
- **Mechanism:** Runs as a local proxy on `:8787` (or wraps CLI agents). Compresses conversation history, RAG chunks, and tool outputs using SmartCrusher (JSON), CodeCompressor (AST), and Kompress-v2-base. Also attempts "effort routing" to dial down thinking budgets.
|
|
21
|
+
- **Critical Failure Points:**
|
|
22
|
+
- **Orphaned `tool_result` blocks:** When pruning historical turns, Headroom often discards the `assistant` turn containing a `tool_use`, while retaining the subsequent `user` turn containing the `tool_result`. Anthropic's API strictly validates tool use IDs and crashes with:
|
|
23
|
+
```
|
|
24
|
+
HTTP 400 invalid_request_error: "tool_use_id 'xxx' does not correspond to any tool_use"
|
|
25
|
+
```
|
|
26
|
+
- **Consecutive `user` turns:** Dropping intermediary assistant turns causes multiple user messages to sit adjacent to each other. Anthropic strictly throws:
|
|
27
|
+
```
|
|
28
|
+
HTTP 400 invalid_request_error: "roles must alternate between 'user' and 'assistant'"
|
|
29
|
+
```
|
|
30
|
+
- **Reasoning Suppression:** Its "effort routing" dials down `thinking.budget_tokens` on routine tool turns. On complex models (Claude Opus, Gemini Flash), this prevents the model from formulating multi-step reasoning before acting.
|
|
31
|
+
|
|
32
|
+
### 2. RTK (`rtk-ai/rtk` - Rust Token Killer)
|
|
33
|
+
- **Mechanism:** A single Rust binary that hooks into shell tool execution (e.g. `PreToolUse` in Claude Code / Cursor) and rewrites CLI commands (`git`, `ls`, `cat`, `grep`, `pytest`) to filter out noise, truncate lines, and inject recall tokens (`[full output: rtk recall xxx]`).
|
|
34
|
+
- **Critical Failure Points:**
|
|
35
|
+
- **Corrupted Structural Data:** When an agent invokes a tool expecting machine-readable JSON or exact AST formatting, RTK's heuristic summaries can alter structural delimiters, causing downstream tool call parsing errors.
|
|
36
|
+
- **Custom Tracking Headers:** RTK and associated tracing proxies inject headers (`x-rtk-*`, `traceparent`, `x-optimizer-id`) that some strict upstream endpoints reject if not cleanly forwarded.
|
|
37
|
+
|
|
38
|
+
### 3. Ponytail (`DietrichGebert/ponytail`)
|
|
39
|
+
- **Mechanism:** A behavioral prompt engineering plugin/ruleset that injects extreme conciseness instructions ("write one line, it works, YAGNI") into agent system prompts across 20+ coding tools.
|
|
40
|
+
- **Critical Failure Points:**
|
|
41
|
+
- **Premature Execution without Reasoning:** By commanding the model to be maximally brief and avoid planning, reasoning models are discouraged from spending thinking tokens. The model outputs untested single-liners that often fail type checks and test suites.
|
|
42
|
+
- **System Prompt Prefix Invalidation:** Injected rules alter the leading system prompt bytes, busting provider prompt caches unless carefully aligned.
|
|
43
|
+
|
|
44
|
+
---
|
|
45
|
+
|
|
46
|
+
## The Healer Engine: How LLM Switcher Protects the Workflow
|
|
47
|
+
|
|
48
|
+
LLM Switcher sits between the optimizer tool and the upstream LLM:
|
|
49
|
+
|
|
50
|
+
```
|
|
51
|
+
[CLI Agent] ──> [Optimizer: Headroom / RTK] ──> [LLM Switcher :3456] ──> [Upstream / 9Router]
|
|
52
|
+
│
|
|
53
|
+
├── 1. Heal Orphaned tool_results
|
|
54
|
+
├── 2. Merge Consecutive Turns
|
|
55
|
+
├── 3. Restore Stripped Thinking
|
|
56
|
+
└── 4. Enforce 1M Context
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
### Protection Matrix: Before vs. After
|
|
60
|
+
|
|
61
|
+
| Scenario | Direct to Upstream (Without Switcher) | Through LLM Switcher (Healer Engine) |
|
|
62
|
+
|---|---|---|
|
|
63
|
+
| **Orphaned `tool_result` turn** | ❌ **HTTP 400 Crash**: `tool_use_id does not correspond to any tool_use` | ✅ **HTTP 200 OK**: Heals orphaned result into contextual text block `[Tool Result (id)]: ...` |
|
|
64
|
+
| **Consecutive `user` turns** | ❌ **HTTP 400 Crash**: `roles must alternate` | ✅ **HTTP 200 OK**: Merges consecutive turns into a single valid turn seamlessly |
|
|
65
|
+
| **Stripped `thinking` parameters** | ⚠️ **Degraded AI**: Reasoning disabled, model outputs shallow single-liners | ✅ **HTTP 200 OK**: Detects reasoning models and automatically restores safe thinking budget |
|
|
66
|
+
| **Orphaned `tool` role in Chat API** | ❌ **HTTP 400 Crash**: `tool role must respond to tool_calls` | ✅ **HTTP 200 OK**: Converts orphaned tool message into user context |
|
|
67
|
+
| **Custom tracking headers** | ⚠️ Connection dropped / unrecognized header warnings | ✅ **HTTP 200 OK**: Cleanly passes through `traceparent`, `x-request-id`, `x-rtk-*` |
|
|
68
|
+
|
|
69
|
+
---
|
|
70
|
+
|
|
71
|
+
## Test Methodology & Verification Suite
|
|
72
|
+
|
|
73
|
+
We created an automated verification suite in [`tests/live-optimizer-interop.mjs`](../tests/live-optimizer-interop.mjs) that systematically replicates the failure modes of each tool:
|
|
74
|
+
|
|
75
|
+
### Running the Test Suite
|
|
76
|
+
```bash
|
|
77
|
+
node tests/live-optimizer-interop.mjs
|
|
78
|
+
```
|
|
79
|
+
|
|
80
|
+
### Live Test Results
|
|
81
|
+
```text
|
|
82
|
+
================================================================
|
|
83
|
+
LLM SWITCHER — TOKEN OPTIMIZER INTEROPERABILITY TEST SUITE
|
|
84
|
+
Simulating failure modes from Headroom, RTK, and Ponytail
|
|
85
|
+
================================================================
|
|
86
|
+
|
|
87
|
+
[TEST] Headroom Simulation: Orphaned tool_result turn... PASS (8840ms)
|
|
88
|
+
↳ Healed orphaned tool_result. Response HTTP 200: "It looks like you've shared a fragment of context ..."
|
|
89
|
+
[TEST] Headroom Simulation: Consecutive User turns (Role alternation violation)... PASS (2683ms)
|
|
90
|
+
↳ Merged consecutive turns seamlessly. Stop reason: end_turn
|
|
91
|
+
[TEST] Thinking Guard: Restoring stripped thinking parameter on reasoning models... PASS (5372ms)
|
|
92
|
+
↳ Automatically restored thinking: thinking_delta=true, signature_delta=true, text_delta=true
|
|
93
|
+
[TEST] RTK Intermediary: Custom headers and traceparent passthrough... PASS (2453ms)
|
|
94
|
+
↳ Headers accepted cleanly with HTTP 200 OK
|
|
95
|
+
[TEST] OpenAI Chat Healer: Orphaned role tool without preceding assistant tool_calls... PASS (1561ms)
|
|
96
|
+
↳ Chat Healer rescued orphaned tool role. Stop reason: length
|
|
97
|
+
|
|
98
|
+
================================================================
|
|
99
|
+
TEST RESULTS: 5 PASSED / 0 FAILED
|
|
100
|
+
================================================================
|
|
101
|
+
```
|
|
102
|
+
|
|
103
|
+
---
|
|
104
|
+
|
|
105
|
+
## Recommended User Setup
|
|
106
|
+
|
|
107
|
+
For developers using prompt optimization tools:
|
|
108
|
+
1. Keep the optimizer installed in your CLI tool as usual.
|
|
109
|
+
2. In the optimizer's configuration (e.g. `headroom.yaml` or RTK upstream settings), set the upstream target URL to **LLM Switcher** (`http://127.0.0.1:3456`).
|
|
110
|
+
3. Enjoy prompt compression savings without worrying about broken conversation graphs, HTTP 400 crashes, or lost reasoning depth.
|
package/docs/codex-blindfold.md
CHANGED
|
@@ -28,18 +28,25 @@ Blindfold mode does that. Codex keeps its official endpoint. The switcher interc
|
|
|
28
28
|
|
|
29
29
|
Codex reads the `HTTPS_PROXY` variable. In blindfold mode the switcher points that variable at `blindfold/blindfold.mjs`. Codex then sends `CONNECT chatgpt.com:443` to that process.
|
|
30
30
|
|
|
31
|
-
The process answers the CONNECT itself. It ends the TLS session with a leaf certificate for the target host
|
|
31
|
+
The process answers the CONNECT itself. It ends the TLS session with a leaf certificate for the target host. One interceptor serves both tools: what each request reaches is decided by the host of the CONNECT request and by its path, and by nothing else.
|
|
32
32
|
|
|
33
|
-
|
|
|
33
|
+
| CONNECT host | Destination |
|
|
34
34
|
| --- | --- |
|
|
35
|
-
|
|
|
36
|
-
|
|
|
35
|
+
| `api.anthropic.com`, path `/v1/messages` or under it | The local gateway, path unchanged |
|
|
36
|
+
| `api.anthropic.com`, any other path | `api.anthropic.com`, over a new TLS session |
|
|
37
|
+
| `api.openai.com`, path `/v1/responses` or `/v1/models` (or under them) | The local gateway, path unchanged |
|
|
38
|
+
| `api.openai.com`, any other path | `api.openai.com`, over a new TLS session |
|
|
39
|
+
| `chatgpt.com`, path `/backend-api/codex/` or under it | The local gateway, `/backend-api/codex` rewritten to `/v1` |
|
|
40
|
+
| `chatgpt.com`, any other path | `chatgpt.com`, over a new TLS session |
|
|
37
41
|
| Another public host | A raw tunnel. The process never reads the bytes. |
|
|
38
42
|
| A local or private address | Refused. |
|
|
39
43
|
|
|
44
|
+
There is no `--host` and no `--prefix` any more: this table is the routing, and it cannot be
|
|
45
|
+
changed from a profile or from the command line.
|
|
46
|
+
|
|
40
47
|
The routing decision reads the **normalized** path, not the text the client sent. Node hands over the request target exactly as written, but the gateway resolves it with `new URL(...)`. A raw-text test would therefore accept a string that the gateway later reads as a different path: `/backend-api/codex/%2e%2e/api/logs` becomes `/api/logs`, which is the gateway's admin API. The decision also requires a segment boundary, so `/backend-api/codex-usage` stays with the host it belongs to.
|
|
41
48
|
|
|
42
|
-
Sign-in, token refresh and the usage page keep working, because they do not use the
|
|
49
|
+
Sign-in, token refresh and the usage page keep working, because they do not use a path in the table. A request whose `Host` header names a different host than its CONNECT target gets `421 Misdirected Request` and opens no upstream connection.
|
|
43
50
|
|
|
44
51
|
## Why no system change is necessary
|
|
45
52
|
|
|
@@ -81,14 +88,17 @@ CAUTION: Do not build these certificates with `New-SelfSignedCertificate` in Pow
|
|
|
81
88
|
|
|
82
89
|
### 2. Turn on blindfold mode in the profile
|
|
83
90
|
|
|
84
|
-
Add
|
|
91
|
+
Add the interceptor port at the top level of `config.json`, next to `port`:
|
|
85
92
|
|
|
86
93
|
```json
|
|
87
|
-
"blindfold":
|
|
88
|
-
"blindfoldPort": 3457
|
|
94
|
+
"blindfold": { "port": 3457 }
|
|
89
95
|
```
|
|
90
96
|
|
|
91
|
-
`
|
|
97
|
+
`blindfold.port` is optional. The default is 3457, and it must differ from the gateway port
|
|
98
|
+
in `port`. There is no per-profile `blindfold`, `blindfoldPort`, `blindfoldHost` or
|
|
99
|
+
`blindfoldPrefix` any more: the interceptor runs whenever at least one tool is active, and
|
|
100
|
+
what it routes is the fixed host table in [`cross-platform.md`](cross-platform.md), not a
|
|
101
|
+
profile setting.
|
|
92
102
|
|
|
93
103
|
### 3. Activate the profile
|
|
94
104
|
|
|
@@ -98,7 +108,7 @@ switch codex <your-profile>
|
|
|
98
108
|
|
|
99
109
|
This one command starts the gateway and writes the environment files. The gateway then starts the interceptor on the configured port. `switch off` stops both and removes the generated files.
|
|
100
110
|
|
|
101
|
-
The gateway owns the interceptor. It brings the interceptor in line with `config.json` when it starts, after every change in the dashboard, and after every `switch` command. A service start, a `switch port`, and a change of
|
|
111
|
+
The gateway owns the interceptor. It brings the interceptor in line with `config.json` when it starts, after every change in the dashboard, and after every `switch` command. A service start, a `switch port`, and a change of port in the dashboard therefore never leave `HTTPS_PROXY` pointing at a port where nothing listens. When blindfold mode is turned off for the Codex target, the gateway stops the interceptor, and a running Codex session must be restarted.
|
|
102
112
|
|
|
103
113
|
`switch` refuses the activation and writes no file in these cases:
|
|
104
114
|
|
|
@@ -155,13 +165,14 @@ re-originated to the real host. A compressed body is decoded first, because a
|
|
|
155
165
|
client asks for `gzip` and the bytes on the wire are not readable text. The
|
|
156
166
|
forwarded response keeps its original bytes; only the copy in the file is decoded.
|
|
157
167
|
|
|
158
|
-
|
|
159
|
-
|
|
168
|
+
Both tools are recorded by the same process, because one interceptor holds the whole
|
|
169
|
+
table. To capture into a directory of your own, pass `--capture` to a second interceptor
|
|
170
|
+
on a port of its own:
|
|
160
171
|
|
|
161
172
|
```bash
|
|
162
|
-
bash blindfold/make-certs.sh
|
|
163
|
-
node blindfold/blindfold.mjs --
|
|
164
|
-
--
|
|
173
|
+
bash blindfold/make-certs.sh
|
|
174
|
+
node blindfold/blindfold.mjs --port 3458 --gateway-port 3456 \
|
|
175
|
+
--certs blindfold/certs --capture ~/.llm-switcher/captures
|
|
165
176
|
```
|
|
166
177
|
|
|
167
178
|
Use directories that you own. `make-certs.sh` and `--capture` refuse a directory that another
|
|
@@ -201,13 +212,13 @@ The switcher writes `LLM_SWITCHER_CODEX_BASE_URL` again, the gateway stops the i
|
|
|
201
212
|
|
|
202
213
|
Read this section before you turn blindfold mode on.
|
|
203
214
|
|
|
204
|
-
**The CA is trusted for every host that Codex contacts.** `CODEX_CA_CERTIFICATE` adds this authority to the trust store that Codex uses for all of its HTTPS calls. A CA that `make-certs.sh` builds now carries
|
|
215
|
+
**The CA is trusted for every host that Codex contacts.** `CODEX_CA_CERTIFICATE` adds this authority to the trust store that Codex uses for all of its HTTPS calls. A CA that `make-certs.sh` builds now carries name constraints: it can sign only for `api.anthropic.com`, `api.openai.com` and `chatgpt.com`. A client that obeys name constraints refuses any other certificate from this CA. OpenSSL, rustls and the macOS and Windows verifiers obey them. A CA that an older version built has no constraint. Anybody who can read its `ca.key` can forge a certificate for any host. Run `make-certs.sh` again to replace it, and keep `blindfold/certs/` private in both cases.
|
|
205
216
|
|
|
206
217
|
**All traffic to the intercepted host is decrypted by this process.** That includes sign-in and token refresh, because the CONNECT for the whole host is terminated locally. Requests outside the Codex API path are re-originated to the real host over a new TLS session; they are forwarded, not tunneled. The `--verbose` flag prints the method and path of every such request.
|
|
207
218
|
|
|
208
219
|
**Directory permissions are weaker on Windows.** `make-certs.sh` calls `chmod` on the certificate directory, and that call does nothing under Git Bash for Windows. On that platform the keys are protected only by the account that owns the directory.
|
|
209
220
|
|
|
210
|
-
**
|
|
221
|
+
**Exactly three hosts are intercepted.** The leaf names `api.anthropic.com`, `api.openai.com` and `chatgpt.com`, so a Codex that signs in with an API key instead of a ChatGPT account reaches the gateway on `api.openai.com` with no reconfiguration. Any other host is tunneled without being read.
|
|
211
222
|
|
|
212
223
|
**The interceptor is a proxy.** It binds to loopback, so a remote machine cannot use it, but every process on this machine can. It refuses a CONNECT to a local or private address, so it cannot be used to reach a service that listens only on this machine. The interceptor resolves the name first and checks every address in the answer, so a spelling such as `2130706433` or a name that resolves to `127.0.0.1` is also refused. It then connects to the address that it checked. It prints each refused target once, without `--verbose`. A VPN or split DNS can resolve a public name to a private address, and that message shows the cause.
|
|
213
224
|
|