llm-switcher 1.2.7 → 1.2.8
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.impeccable/hook.cache.json +1 -0
- package/CHANGELOG.md +45 -0
- package/README.md +74 -4
- package/README.vi.md +75 -4
- package/blindfold/blindfold.mjs +2 -1
- package/docs/TOKEN-OPTIMIZER-INTEROP.md +110 -110
- package/docs/cross-platform.md +3 -1
- package/docs/response-matrix.json +1130 -1130
- package/hook-status.mjs +51 -0
- package/notify-route.ps1 +46 -0
- package/package.json +2 -2
- package/plugin.mjs +204 -0
- package/service.mjs +1 -1
- package/shim.mjs +28 -2
- package/skills/llm-switcher/SKILL.md +93 -93
- package/state.mjs +41 -5
- package/switch +0 -0
- package/switch.cmd +2 -2
- package/switch.mjs +53 -2
- package/tests/blindfold.test.mjs +1 -1
- package/tests/cdp.mjs +21 -2
- package/tests/dashboard.e2e.test.mjs +116 -0
- package/tests/gateway.e2e.test.mjs +1 -1
- package/tests/helpers.mjs +24 -24
- package/tests/hook-status.test.mjs +119 -0
- package/tests/live-optimizer-interop.mjs +205 -205
- package/tests/plugin.test.mjs +132 -0
- package/tests/real-user-sim.test.mjs +2 -2
- package/tests/service.test.mjs +2 -0
- package/tests/shim.test.mjs +112 -0
- package/tests/state.test.mjs +130 -5
- package/ui.html +85 -6
package/hook-status.mjs
ADDED
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// Reports the switcher route to a coding tool at the start of a session.
|
|
3
|
+
//
|
|
4
|
+
// The shim toast fires only when the shim runs, and that needs the shim directory first on PATH.
|
|
5
|
+
// This runs inside the tool, so it reports even when the launcher is the person's own script. Claude
|
|
6
|
+
// Code and Codex both read one JSON object from stdout and show `systemMessage` to the person.
|
|
7
|
+
//
|
|
8
|
+
// Two rules hold, whatever happens:
|
|
9
|
+
// - It prints exactly one JSON object and exits 0. A hook must never fail a session.
|
|
10
|
+
// - A tool with no profile gets an empty object. Silence for a tool on its official endpoint.
|
|
11
|
+
//
|
|
12
|
+
// Usage: node hook-status.mjs <claude|codex>
|
|
13
|
+
import fs from 'node:fs';
|
|
14
|
+
|
|
15
|
+
const TOOLS = ['claude', 'codex'];
|
|
16
|
+
|
|
17
|
+
const readOrEmpty = (file) => {
|
|
18
|
+
try { return fs.readFileSync(file, 'utf8'); } catch { return ''; }
|
|
19
|
+
};
|
|
20
|
+
|
|
21
|
+
async function message(tool) {
|
|
22
|
+
if (!TOOLS.includes(tool)) return {};
|
|
23
|
+
// Imported here, not at the top: a failure to load the state module must still print an object.
|
|
24
|
+
const s = await import('./state.mjs');
|
|
25
|
+
|
|
26
|
+
// The route file is written with the env file of the same tool, so it is empty exactly when that
|
|
27
|
+
// tool is not routed. One read answers both "is the switcher on" and "is this tool on".
|
|
28
|
+
const route = readOrEmpty(tool === 'codex' ? s.paths.routeCodex : s.paths.routeClaude).trim();
|
|
29
|
+
if (!route || !fs.existsSync(s.paths.activeFlag)) return {};
|
|
30
|
+
|
|
31
|
+
const port = s.resolvePort([], s.loadConfig() || {});
|
|
32
|
+
const state = await s.probeGateway(port);
|
|
33
|
+
if (state === 'ours') {
|
|
34
|
+
return { systemMessage: `LLM Switcher is ON. ${route}. This tool does not reach its official endpoint.` };
|
|
35
|
+
}
|
|
36
|
+
// The dangerous state: the launch files route the tool, and nothing answers. The tool fails on
|
|
37
|
+
// its first request with a connection error that names no cause.
|
|
38
|
+
const held = state === 'free'
|
|
39
|
+
? `the gateway on port ${port} does not answer`
|
|
40
|
+
: `port ${port} does not answer as this switcher (another program holds it)`;
|
|
41
|
+
return {
|
|
42
|
+
systemMessage: `WARNING: LLM Switcher is set to route this tool (${route}), but ${held}. `
|
|
43
|
+
+ 'The tool cannot reach the provider. Run `switch on` to start the gateway, '
|
|
44
|
+
+ 'or `switch off` to use the official endpoint.'
|
|
45
|
+
};
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
message(process.argv[2])
|
|
49
|
+
.then((out) => process.stdout.write(JSON.stringify(out)))
|
|
50
|
+
.catch(() => process.stdout.write('{}'))
|
|
51
|
+
.finally(() => process.exit(0));
|
package/notify-route.ps1
ADDED
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
# Raises one Windows toast that says the switcher took this tool's traffic.
|
|
2
|
+
#
|
|
3
|
+
# The launcher shim calls this detached, so it must never block and never fail loudly: a coding
|
|
4
|
+
# tool must start even when the notification does not. It reads one argument, the route file of
|
|
5
|
+
# the tool. An absent or empty file means the tool is not routed, and then nothing is shown.
|
|
6
|
+
#
|
|
7
|
+
# No module is installed. WinRT is used through the PowerShell application id, which exists on
|
|
8
|
+
# every Windows 10 and 11 machine.
|
|
9
|
+
param([Parameter(Mandatory = $true)][string]$RouteFile)
|
|
10
|
+
|
|
11
|
+
$ErrorActionPreference = 'Stop'
|
|
12
|
+
trap { exit 0 }
|
|
13
|
+
|
|
14
|
+
if (-not (Test-Path -LiteralPath $RouteFile)) { exit 0 }
|
|
15
|
+
$line = (Get-Content -LiteralPath $RouteFile -Raw -ErrorAction SilentlyContinue)
|
|
16
|
+
if ($null -eq $line) { exit 0 }
|
|
17
|
+
$line = $line.Trim()
|
|
18
|
+
if ($line.Length -eq 0) { exit 0 }
|
|
19
|
+
|
|
20
|
+
# The line is data, so it is escaped before it enters the toast XML.
|
|
21
|
+
$body = [System.Security.SecurityElement]::Escape($line)
|
|
22
|
+
|
|
23
|
+
[Windows.UI.Notifications.ToastNotificationManager, Windows.UI.Notifications, ContentType = WindowsRuntime] | Out-Null
|
|
24
|
+
[Windows.Data.Xml.Dom.XmlDocument, Windows.Data.Xml.Dom, ContentType = WindowsRuntime] | Out-Null
|
|
25
|
+
|
|
26
|
+
$appId = '{1AC14E77-02E7-4E5D-B744-2EB1AE5198B7}\WindowsPowerShell\v1.0\powershell.exe'
|
|
27
|
+
$xml = @"
|
|
28
|
+
<toast activationType="protocol" launch="http://127.0.0.1:3456/ui">
|
|
29
|
+
<visual>
|
|
30
|
+
<binding template="ToastGeneric">
|
|
31
|
+
<text>LLM Switcher is ON</text>
|
|
32
|
+
<text>$body</text>
|
|
33
|
+
<text placement="attribution">This tool does not reach its official endpoint.</text>
|
|
34
|
+
</binding>
|
|
35
|
+
</visual>
|
|
36
|
+
</toast>
|
|
37
|
+
"@
|
|
38
|
+
|
|
39
|
+
$doc = [Windows.Data.Xml.Dom.XmlDocument]::new()
|
|
40
|
+
$doc.LoadXml($xml)
|
|
41
|
+
$toast = [Windows.UI.Notifications.ToastNotification]::new($doc)
|
|
42
|
+
# One tag per tool: a second launch of the same tool replaces its notice instead of stacking.
|
|
43
|
+
$toast.Tag = 'llm-switcher-route'
|
|
44
|
+
$toast.Group = 'llm-switcher'
|
|
45
|
+
[Windows.UI.Notifications.ToastNotificationManager]::CreateToastNotifier($appId).Show($toast)
|
|
46
|
+
exit 0
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "llm-switcher",
|
|
3
|
-
"version": "1.2.
|
|
4
|
-
"description": "Zero-dependency multi-protocol edge gateway & provider switcher for Claude Code
|
|
3
|
+
"version": "1.2.8",
|
|
4
|
+
"description": "Zero-dependency multi-protocol edge gateway & provider switcher for Claude Code and Codex, with OpenAI-compatible, Anthropic and Vertex upstreams",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"llm",
|
|
7
7
|
"gateway",
|
package/plugin.mjs
ADDED
|
@@ -0,0 +1,204 @@
|
|
|
1
|
+
// ============================================================
|
|
2
|
+
// plugin.mjs - installs the launch hook into the two coding tools, and edits no configuration.
|
|
3
|
+
//
|
|
4
|
+
// WHY A HOOK AT ALL
|
|
5
|
+
// The shim raises a notice at launch, but only when the shim runs, and that needs the shim
|
|
6
|
+
// directory first on PATH. A person who keeps their own `claude` wrapper never sees it. A hook
|
|
7
|
+
// runs inside the tool, so it reports whatever the launcher is.
|
|
8
|
+
//
|
|
9
|
+
// WHY NOTHING IS EDITED
|
|
10
|
+
// Claude Code loads any folder under a skills directory that holds `.claude-plugin/plugin.json`
|
|
11
|
+
// as a plugin, on the next session, with no marketplace and no install step. Codex loads
|
|
12
|
+
// `$CODEX_HOME/hooks.json` by itself. So each tool reads a file of its own, and `settings.json`
|
|
13
|
+
// and `config.toml` are never opened.
|
|
14
|
+
//
|
|
15
|
+
// WHY IT IS OPT-IN
|
|
16
|
+
// The notice is a reminder, not a part of the routing. A person who does not want it installs
|
|
17
|
+
// nothing and loses nothing: the traffic still goes through the gateway, and the shim toast
|
|
18
|
+
// still fires.
|
|
19
|
+
// ============================================================
|
|
20
|
+
|
|
21
|
+
import fs from 'node:fs';
|
|
22
|
+
import path from 'node:path';
|
|
23
|
+
import os from 'node:os';
|
|
24
|
+
import { fileURLToPath } from 'node:url';
|
|
25
|
+
|
|
26
|
+
const HOOK_SCRIPT = path.join(path.dirname(fileURLToPath(import.meta.url)), 'hook-status.mjs');
|
|
27
|
+
|
|
28
|
+
export const PLUGIN_NAME = 'llm-switcher-status';
|
|
29
|
+
// Written into every file this module owns, and the only thing that authorizes an overwrite.
|
|
30
|
+
const MARK = 'Generated by LLM Switcher - safe to delete';
|
|
31
|
+
// A fresh start and a resumed session. A resume is the case the switcher exists for, and compact
|
|
32
|
+
// is left out on purpose: the notice belongs at the start of a session, not in the middle of one.
|
|
33
|
+
const SESSION_SOURCES = 'startup|resume';
|
|
34
|
+
|
|
35
|
+
export function claudeSkillsDir(env = process.env, home = os.homedir()) {
|
|
36
|
+
return path.join(env.CLAUDE_CONFIG_DIR || path.join(home, '.claude'), 'skills');
|
|
37
|
+
}
|
|
38
|
+
export function codexHome(env = process.env, home = os.homedir()) {
|
|
39
|
+
return env.CODEX_HOME || path.join(home, '.codex');
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
export function renderClaudePlugin(version = 'dev') {
|
|
43
|
+
return `${JSON.stringify({
|
|
44
|
+
name: PLUGIN_NAME,
|
|
45
|
+
version,
|
|
46
|
+
description: 'Reports at the start of a session when LLM Switcher routes this tool to another provider.',
|
|
47
|
+
author: { name: 'LLM Switcher' }
|
|
48
|
+
}, null, 2)}\n`;
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
// The args form, not one command string: a Windows path holds backslashes and spaces, and this
|
|
52
|
+
// way no shell has to quote it.
|
|
53
|
+
export function renderClaudeHooks() {
|
|
54
|
+
return `${JSON.stringify({
|
|
55
|
+
description: MARK,
|
|
56
|
+
hooks: {
|
|
57
|
+
SessionStart: [{
|
|
58
|
+
matcher: SESSION_SOURCES,
|
|
59
|
+
hooks: [{ type: 'command', command: 'node', args: [HOOK_SCRIPT, 'claude'], timeout: 10 }]
|
|
60
|
+
}]
|
|
61
|
+
}
|
|
62
|
+
}, null, 2)}\n`;
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
// Codex takes one command string. It is quoted here, once, for the same reason.
|
|
66
|
+
export function renderCodexHooks() {
|
|
67
|
+
return `${JSON.stringify({ description: MARK, hooks: { SessionStart: [codexEntry()] } }, null, 2)}\n`;
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
function codexEntry() {
|
|
71
|
+
return {
|
|
72
|
+
matcher: SESSION_SOURCES,
|
|
73
|
+
hooks: [{ type: 'command', command: `node "${HOOK_SCRIPT}" codex`, timeout: 10 }]
|
|
74
|
+
};
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
// Ours is the entry that runs our script. Nothing else in the file is touched, by any operation.
|
|
78
|
+
const isOurs = (group) => JSON.stringify(group).includes('hook-status.mjs');
|
|
79
|
+
|
|
80
|
+
function writeAtomic(file, content) {
|
|
81
|
+
const tmp = `${file}.${process.pid}.tmp`;
|
|
82
|
+
fs.mkdirSync(path.dirname(file), { recursive: true });
|
|
83
|
+
try {
|
|
84
|
+
fs.writeFileSync(tmp, content, 'utf8');
|
|
85
|
+
fs.renameSync(tmp, file);
|
|
86
|
+
} catch (err) {
|
|
87
|
+
fs.rmSync(tmp, { force: true });
|
|
88
|
+
throw err;
|
|
89
|
+
}
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
function readJsonFile(file) {
|
|
93
|
+
if (!fs.existsSync(file)) return { missing: true };
|
|
94
|
+
const raw = fs.readFileSync(file, 'utf8');
|
|
95
|
+
try { return { value: JSON.parse(raw) }; } catch (err) { return { error: err.message }; }
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
// ---- Claude Code -----------------------------------------------------------------------------
|
|
99
|
+
|
|
100
|
+
function pluginDir(opts) {
|
|
101
|
+
return path.join(opts.claudeSkillsDir || claudeSkillsDir(), PLUGIN_NAME);
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
function installClaude(opts, version) {
|
|
105
|
+
const dir = pluginDir(opts);
|
|
106
|
+
const manifest = path.join(dir, '.claude-plugin', 'plugin.json');
|
|
107
|
+
// A folder with a manifest that is not ours belongs to someone else, whatever its name is.
|
|
108
|
+
if (fs.existsSync(manifest)) {
|
|
109
|
+
const cur = readJsonFile(manifest);
|
|
110
|
+
if (cur.error) return { ok: false, reason: `${manifest} does not parse: ${cur.error}` };
|
|
111
|
+
if (cur.value?.name !== PLUGIN_NAME) {
|
|
112
|
+
return { ok: false, reason: `${dir} holds another plugin (${cur.value?.name}), so nothing was changed` };
|
|
113
|
+
}
|
|
114
|
+
}
|
|
115
|
+
writeAtomic(manifest, renderClaudePlugin(version));
|
|
116
|
+
writeAtomic(path.join(dir, 'hooks', 'hooks.json'), renderClaudeHooks());
|
|
117
|
+
return { ok: true, path: dir };
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
function uninstallClaude(opts) {
|
|
121
|
+
const dir = pluginDir(opts);
|
|
122
|
+
const manifest = path.join(dir, '.claude-plugin', 'plugin.json');
|
|
123
|
+
if (!fs.existsSync(manifest)) return { ok: true, path: dir, absent: true };
|
|
124
|
+
const cur = readJsonFile(manifest);
|
|
125
|
+
if (cur.value?.name !== PLUGIN_NAME) return { ok: false, reason: `${dir} is not our plugin, so nothing was removed` };
|
|
126
|
+
fs.rmSync(dir, { recursive: true, force: true });
|
|
127
|
+
return { ok: true, path: dir };
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
// ---- Codex -----------------------------------------------------------------------------------
|
|
131
|
+
|
|
132
|
+
function codexHooksPath(opts) {
|
|
133
|
+
return path.join(opts.codexHome || codexHome(), 'hooks.json');
|
|
134
|
+
}
|
|
135
|
+
|
|
136
|
+
// The file may already hold the hooks of this machine, so it is merged, never replaced. A file
|
|
137
|
+
// that does not parse is left exactly as it is: it can hold work that no backup would return.
|
|
138
|
+
function installCodex(opts) {
|
|
139
|
+
const file = codexHooksPath(opts);
|
|
140
|
+
const cur = readJsonFile(file);
|
|
141
|
+
if (cur.error) return { ok: false, reason: `${file} does not parse: ${cur.error}, so nothing was changed` };
|
|
142
|
+
if (cur.missing) {
|
|
143
|
+
writeAtomic(file, renderCodexHooks());
|
|
144
|
+
return { ok: true, path: file };
|
|
145
|
+
}
|
|
146
|
+
const doc = cur.value && typeof cur.value === 'object' ? cur.value : {};
|
|
147
|
+
doc.hooks = doc.hooks && typeof doc.hooks === 'object' ? doc.hooks : {};
|
|
148
|
+
const groups = Array.isArray(doc.hooks.SessionStart) ? doc.hooks.SessionStart : [];
|
|
149
|
+
doc.hooks.SessionStart = [...groups.filter(g => !isOurs(g)), codexEntry()];
|
|
150
|
+
writeAtomic(file, `${JSON.stringify(doc, null, 2)}\n`);
|
|
151
|
+
return { ok: true, path: file };
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
function uninstallCodex(opts) {
|
|
155
|
+
const file = codexHooksPath(opts);
|
|
156
|
+
const cur = readJsonFile(file);
|
|
157
|
+
if (cur.missing) return { ok: true, path: file, absent: true };
|
|
158
|
+
if (cur.error) return { ok: false, reason: `${file} does not parse: ${cur.error}, so nothing was changed` };
|
|
159
|
+
const doc = cur.value && typeof cur.value === 'object' ? cur.value : {};
|
|
160
|
+
const groups = Array.isArray(doc.hooks?.SessionStart) ? doc.hooks.SessionStart : [];
|
|
161
|
+
const kept = groups.filter(g => !isOurs(g));
|
|
162
|
+
if (kept.length) doc.hooks.SessionStart = kept;
|
|
163
|
+
else delete doc.hooks?.SessionStart;
|
|
164
|
+
// Only a file that holds nothing but our own mark is removed. Anything else is written back.
|
|
165
|
+
const empty = doc.hooks && Object.keys(doc.hooks).length === 0;
|
|
166
|
+
if (empty && doc.description === MARK) {
|
|
167
|
+
fs.rmSync(file, { force: true });
|
|
168
|
+
return { ok: true, path: file };
|
|
169
|
+
}
|
|
170
|
+
writeAtomic(file, `${JSON.stringify(doc, null, 2)}\n`);
|
|
171
|
+
return { ok: true, path: file };
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
// ---- The three commands ----------------------------------------------------------------------
|
|
175
|
+
|
|
176
|
+
function summarize(results) {
|
|
177
|
+
const out = { installed: [], failed: [], paths: {} };
|
|
178
|
+
for (const [tool, r] of Object.entries(results)) {
|
|
179
|
+
if (r.ok) out.installed.push(tool); else out.failed.push({ tool, reason: r.reason });
|
|
180
|
+
if (r.path) out.paths[tool] = r.path;
|
|
181
|
+
}
|
|
182
|
+
return out;
|
|
183
|
+
}
|
|
184
|
+
|
|
185
|
+
/** { installed: ['claude','codex'], failed: [{ tool, reason }], paths: { claude, codex } } */
|
|
186
|
+
export function installPlugin(opts = {}, version = 'dev') {
|
|
187
|
+
return summarize({ claude: installClaude(opts, version), codex: installCodex(opts) });
|
|
188
|
+
}
|
|
189
|
+
|
|
190
|
+
export function uninstallPlugin(opts = {}) {
|
|
191
|
+
return summarize({ claude: uninstallClaude(opts), codex: uninstallCodex(opts) });
|
|
192
|
+
}
|
|
193
|
+
|
|
194
|
+
export function pluginStatus(opts = {}) {
|
|
195
|
+
const dir = pluginDir(opts);
|
|
196
|
+
const claudeManifest = readJsonFile(path.join(dir, '.claude-plugin', 'plugin.json'));
|
|
197
|
+
const file = codexHooksPath(opts);
|
|
198
|
+
const codexDoc = readJsonFile(file);
|
|
199
|
+
const codexGroups = Array.isArray(codexDoc.value?.hooks?.SessionStart) ? codexDoc.value.hooks.SessionStart : [];
|
|
200
|
+
return {
|
|
201
|
+
claude: { installed: claudeManifest.value?.name === PLUGIN_NAME, path: dir, id: `${PLUGIN_NAME}@skills-dir` },
|
|
202
|
+
codex: { installed: codexGroups.some(isOurs), path: file }
|
|
203
|
+
};
|
|
204
|
+
}
|
package/service.mjs
CHANGED
|
@@ -5,7 +5,7 @@ import path from 'node:path';
|
|
|
5
5
|
|
|
6
6
|
// A service does not inherit the installing shell. Without these it would read another config.json
|
|
7
7
|
// or clean another settings.json than the shell that installed it.
|
|
8
|
-
const SERVICE_ENV_KEYS = ['CLAUDE_CONFIG_DIR', 'LLM_SWITCHER_CONFIG', 'LLM_SWITCHER_STATE_DIR', 'LLM_SWITCHER_BLINDFOLD_CERTS'];
|
|
8
|
+
const SERVICE_ENV_KEYS = ['LLM_SWITCHER_HOME', 'CLAUDE_CONFIG_DIR', 'LLM_SWITCHER_CONFIG', 'LLM_SWITCHER_STATE_DIR', 'LLM_SWITCHER_BLINDFOLD_CERTS'];
|
|
9
9
|
|
|
10
10
|
export function serviceEnv(env = process.env) {
|
|
11
11
|
return SERVICE_ENV_KEYS.filter(k => env[k]).map(k => [k, env[k]]);
|
package/shim.mjs
CHANGED
|
@@ -36,15 +36,18 @@ export const SHIM_DIR = path.join(os.homedir(), '.llm-switcher', 'bin');
|
|
|
36
36
|
// `node -e`, so neither cmd.exe quoting nor URL encoding can corrupt it: pathToFileURL turns a
|
|
37
37
|
// space in the checkout into %20, which cmd.exe then reads as the (undefined) argument %2.
|
|
38
38
|
const STATE_HELPER = path.join(path.dirname(fileURLToPath(import.meta.url)), 'ensure-ca-bundle.mjs');
|
|
39
|
+
const ROUTE_NOTIFIER = path.join(path.dirname(fileURLToPath(import.meta.url)), 'notify-route.ps1');
|
|
39
40
|
|
|
40
41
|
// CLIs to wrap. `claude` is the most important case (--resume), codex included for completeness.
|
|
41
42
|
export const SHIMMED = ['claude', 'codex'];
|
|
42
43
|
|
|
43
44
|
// One tool, one env file. Sourcing the shared one is how one tool used to capture the other.
|
|
44
45
|
const envFileOf = (name) => (name === 'codex' ? 'env-codex.sh' : 'env-claude.sh');
|
|
46
|
+
const routeFileOf = (name) => (name === 'codex' ? 'route-codex.txt' : 'route-claude.txt');
|
|
45
47
|
|
|
46
48
|
const POSIX_TEMPLATE = (name) => {
|
|
47
49
|
const envFile = envFileOf(name);
|
|
50
|
+
const routeFile = routeFileOf(name);
|
|
48
51
|
const caBlock = name === 'codex' ? '' : `
|
|
49
52
|
# R2: one bundle holds the user's own CA and the switcher CA, so trusting the interceptor never
|
|
50
53
|
# costs a certificate the user already had. Decided here, at launch, because only now is it known
|
|
@@ -85,6 +88,13 @@ else
|
|
|
85
88
|
TOOL_ACTIVE=0
|
|
86
89
|
fi
|
|
87
90
|
|
|
91
|
+
# Nothing on screen tells the person that the switcher took this tool's traffic, so say it once,
|
|
92
|
+
# here. Only while THIS tool is routed: an empty route file means the tool is off, so the tool
|
|
93
|
+
# reaches its official endpoint and there is nothing to warn about.
|
|
94
|
+
if [ "$TOOL_ACTIVE" = "1" ] && [ -s "$SWITCHER_DIR/${routeFile}" ]; then
|
|
95
|
+
printf '[llm-switcher] %s\\n' "$(cat "$SWITCHER_DIR/${routeFile}")" >&2
|
|
96
|
+
fi
|
|
97
|
+
|
|
88
98
|
# R8: a variable is removed ONLY when its value is one an older switcher wrote. The user's own
|
|
89
99
|
# proxy, base URL, model or LLM_SWITCHER_* setting is theirs and is kept.
|
|
90
100
|
GW_PORT="$( { [ -O "$SWITCHER_DIR/gateway.port" ] && tr -d '[:space:]' < "$SWITCHER_DIR/gateway.port"; } 2>/dev/null )"
|
|
@@ -167,6 +177,7 @@ exec "$REAL" "$@"
|
|
|
167
177
|
|
|
168
178
|
const WINDOWS_TEMPLATE = (name) => {
|
|
169
179
|
const envFile = envFileOf(name).replace(/\.sh$/, '.cmd');
|
|
180
|
+
const routeFile = routeFileOf(name);
|
|
170
181
|
const caBlock = name === 'codex' ? '' : `
|
|
171
182
|
REM R2: one bundle holds the user's own CA and the switcher CA, so trusting the interceptor
|
|
172
183
|
REM never costs a certificate the user already had. Decided here, at launch, because only now
|
|
@@ -202,6 +213,11 @@ if exist "%SWITCHER_DIR%\\active.flag" if exist "%SWITCHER_DIR%\\${envFile}" (
|
|
|
202
213
|
)
|
|
203
214
|
if "%TOOL_ACTIVE%"=="1" call "%SWITCHER_DIR%\\${envFile}"
|
|
204
215
|
|
|
216
|
+
REM Nothing on screen tells the person that the switcher took this tool's traffic, and both TUIs
|
|
217
|
+
REM can claim the whole screen, so a printed line would be hidden. A toast is not. It is detached:
|
|
218
|
+
REM the tool never waits for it. Only while THIS tool is routed (%TOOL_ACTIVE%).
|
|
219
|
+
if "%TOOL_ACTIVE%"=="1" if exist "%SWITCHER_DIR%\\${routeFile}" start "" /b powershell -NoProfile -ExecutionPolicy Bypass -File "${ROUTE_NOTIFIER}" "%SWITCHER_DIR%\\${routeFile}" >nul 2>&1
|
|
220
|
+
|
|
205
221
|
REM R8: a variable is removed ONLY when its value is one an older switcher wrote. The user's own
|
|
206
222
|
REM proxy, base URL, model or LLM_SWITCHER_* setting is theirs and is kept.
|
|
207
223
|
set "_GWPORT="
|
|
@@ -359,10 +375,13 @@ export function shimStatus(names = SHIMMED) {
|
|
|
359
375
|
/** Line to add to the shell rc so the shim comes before the real binary. */
|
|
360
376
|
// Windows: setx truncates at 1024 characters and would copy the merged system+user PATH into the
|
|
361
377
|
// user key, and PowerShell does not expand %PATH%. Prepend to the User-scope Path only.
|
|
362
|
-
export function pathExportLine(platform = process.platform) {
|
|
378
|
+
export function pathExportLine(platform = process.platform, shell = currentShell()) {
|
|
363
379
|
if (platform === 'win32') {
|
|
364
380
|
return `powershell -NoProfile -Command "[Environment]::SetEnvironmentVariable('Path', '${SHIM_DIR};' + [Environment]::GetEnvironmentVariable('Path', 'User'), 'User')"`;
|
|
365
381
|
}
|
|
382
|
+
// fish has no `export`, and it prepends through the fish_user_paths universal variable.
|
|
383
|
+
// `-m` moves the directory to the front when it is on the path already.
|
|
384
|
+
if (shell === 'fish') return `fish_add_path -m "${SHIM_DIR}"`;
|
|
366
385
|
return `export PATH="${SHIM_DIR}:$PATH"`;
|
|
367
386
|
}
|
|
368
387
|
|
|
@@ -370,12 +389,19 @@ export function pathExportLine(platform = process.platform) {
|
|
|
370
389
|
export function suggestedRcFiles(platform = process.platform) {
|
|
371
390
|
if (platform === 'win32') return [];
|
|
372
391
|
const home = os.homedir();
|
|
373
|
-
const shell =
|
|
392
|
+
const shell = currentShell();
|
|
374
393
|
if (shell === 'zsh') return [path.join(home, '.zshrc'), path.join(home, '.zprofile')];
|
|
375
394
|
if (shell === 'bash') return [path.join(home, '.bashrc'), path.join(home, '.bash_profile')];
|
|
395
|
+
// fish reads neither .profile nor any POSIX rc file.
|
|
396
|
+
if (shell === 'fish') return [path.join(home, '.config', 'fish', 'config.fish')];
|
|
376
397
|
return [path.join(home, '.profile')];
|
|
377
398
|
}
|
|
378
399
|
|
|
400
|
+
/** The basename of the login shell, which decides both the file to edit and the syntax of the line. */
|
|
401
|
+
function currentShell() {
|
|
402
|
+
return path.basename(process.env.SHELL || '');
|
|
403
|
+
}
|
|
404
|
+
|
|
379
405
|
// The shim is on PATH but behind the real binary: a line that runs later prepends another directory
|
|
380
406
|
// (npm-global, Homebrew). Login shells read the profile file, interactive shells the rc file.
|
|
381
407
|
export function pathOrderHint(platform = process.platform) {
|
|
@@ -1,93 +1,93 @@
|
|
|
1
|
-
---
|
|
2
|
-
name: llm-switcher
|
|
3
|
-
description: Mandatory guidance for AI coding agents and token compression tools (Headroom, RTK, Ponytail, Repomix). Enforces routing all LLM traffic through the local LLM Switcher edge gateway (http://127.0.0.1:3456) instead of directly calling provider endpoints.
|
|
4
|
-
---
|
|
5
|
-
|
|
6
|
-
# LLM Switcher Edge Gateway Skill
|
|
7
|
-
|
|
8
|
-
This skill governs how AI coding agents (Claude Code, Codex, Cursor, Windsurf, Opencode) and prompt optimization/compression tools (Headroom, RTK, Ponytail, Repomix) must interact with LLM providers on this workstation.
|
|
9
|
-
|
|
10
|
-
## 1. The Core Directive
|
|
11
|
-
|
|
12
|
-
> **ALL OUTBOUND LLM REQUESTS MUST ROUTE THROUGH LLM SWITCHER (`http://127.0.0.1:3456`).**
|
|
13
|
-
>
|
|
14
|
-
> Never configure any tool, script, sub-agent, or token compression proxy to call external provider endpoints directly (such as `https://api.anthropic.com`, `https://api.openai.com`, or `https://generativelanguage.googleapis.com`).
|
|
15
|
-
>
|
|
16
|
-
> If a token compression tool (Headroom, RTK, Ponytail) is used, it MUST be positioned as an **inner processing layer**, and its **upstream destination MUST be configured to point to LLM Switcher** at `http://127.0.0.1:3456`.
|
|
17
|
-
|
|
18
|
-
```
|
|
19
|
-
[Client CLI / Agent]
|
|
20
|
-
│
|
|
21
|
-
▼
|
|
22
|
-
[Token Compressor: RTK / Headroom / Ponytail] <-- Inner Layer
|
|
23
|
-
│
|
|
24
|
-
▼ (MANDATORY: Upstream MUST point to http://127.0.0.1:3456)
|
|
25
|
-
[LLM Switcher Edge Gateway (:3456)] <-- Outermost Gatekeeper
|
|
26
|
-
│
|
|
27
|
-
▼ (Heals schemas, maps 1M context, extracts thinking, converts protocol)
|
|
28
|
-
[Internet / intact (Recommended) / 9Router / Upstream LLM]
|
|
29
|
-
```
|
|
30
|
-
|
|
31
|
-
## 2. Why This is Mandatory
|
|
32
|
-
|
|
33
|
-
Direct outbound calls from token compressors will break the developer environment for several reasons:
|
|
34
|
-
|
|
35
|
-
1. **Schema Breakage & The Healer Engine:**
|
|
36
|
-
Token optimizers frequently prune conversation history to save tokens. In doing so, they often delete an `assistant` turn containing a `tool_use`, while leaving the subsequent `user` turn containing a `tool_result`.
|
|
37
|
-
- Anthropic and OpenAI APIs strictly reject this with `HTTP 400 (orphaned tool_use_id)`.
|
|
38
|
-
- **LLM Switcher** intercepts these malformed requests and runs its **Healer Engine**, automatically converting orphaned results into safe context text blocks and merging consecutive same-role turns so upstream providers accept the payload cleanly.
|
|
39
|
-
2. **Thinking / Reasoning Protection:**
|
|
40
|
-
Many token compressors naively strip `thinking` parameters or reasoning budgets to reduce token count.
|
|
41
|
-
- When routed through LLM Switcher, the gateway detects reasoning models (such as `ag/claude-opus-4-6-thinking` or `ag/gemini-3.8-flash`) and **automatically restores thinking parameters**, ensuring the model retains its full reasoning capabilities.
|
|
42
|
-
3. **Official Context Windows & Dynamic Model Discovery:**
|
|
43
|
-
Context windows follow the model you select without mutating permanent settings. The gateway auto-discovers official models from Anthropic and OpenAI, refreshing mappings on tool version upgrades (`switch models`).
|
|
44
|
-
4. **Server Gateway Synergy (intact / 9Router):**
|
|
45
|
-
LLM Switcher pairs directly with server-side gateways like **intact** (strongly recommended for credential proxying, account pooling, and contract lab self-healing) or 9Router, handling retries and quotas far better than any local tool.
|
|
46
|
-
|
|
47
|
-
## 3. Configuring Token Compression Tools
|
|
48
|
-
|
|
49
|
-
When setting up or assisting the user with token compression tools, apply these exact configurations:
|
|
50
|
-
|
|
51
|
-
### A. Headroom (Prompt Compression Proxy)
|
|
52
|
-
When configuring Headroom's upstream endpoint:
|
|
53
|
-
- In Headroom config (`headroom.yaml` or environment variables):
|
|
54
|
-
```bash
|
|
55
|
-
ANTHROPIC_BASE_URL="http://127.0.0.1:3456"
|
|
56
|
-
OPENAI_BASE_URL="http://127.0.0.1:3456/v1"
|
|
57
|
-
```
|
|
58
|
-
- Point your CLI to Headroom's listening port (e.g. `http://127.0.0.1:8787`). Headroom compresses the prompt and forwards it to LLM Switcher on port `3456`.
|
|
59
|
-
|
|
60
|
-
### B. RTK (Rust Token Killer / Repomix)
|
|
61
|
-
- In RTK configuration or execution flags:
|
|
62
|
-
```bash
|
|
63
|
-
# Ensure RTK's target upstream proxy is LLM Switcher
|
|
64
|
-
export UPSTREAM_URL="http://127.0.0.1:3456"
|
|
65
|
-
```
|
|
66
|
-
|
|
67
|
-
### C. Ponytail / Custom Proxies
|
|
68
|
-
- Always set the proxy's upstream host to:
|
|
69
|
-
`http://127.0.0.1:3456` (or `http://127.0.0.1:3456/v1` for OpenAI-compatible tools).
|
|
70
|
-
|
|
71
|
-
### D. Sub-Agents & Spawned Terminal Commands
|
|
72
|
-
When an agent spawns sub-processes that call LLM CLIs:
|
|
73
|
-
- **Never advise sourcing `env.sh` or `env.cmd`.** Those files carry no proxy variables any
|
|
74
|
-
more, so sourcing one changes nothing — and a stale variable an older version once wrote
|
|
75
|
-
would still point the tool at a port where nothing listens.
|
|
76
|
-
- Run `switch claude <profile>` or `switch codex <profile>` instead. The shims in
|
|
77
|
-
`~/.llm-switcher/bin` inject the proxy variables into the tool process itself and clear
|
|
78
|
-
anything stale first.
|
|
79
|
-
- `claude` and `codex` on `PATH` already are those shims: invoke them normally, and never
|
|
80
|
-
wrap them in another script.
|
|
81
|
-
|
|
82
|
-
## 4. Operational Rules for AI Agents
|
|
83
|
-
|
|
84
|
-
1. **NEVER edit `~/.claude/settings.json` directly.**
|
|
85
|
-
LLM Switcher uses zero-mutation launcher flags (`active.flag`). Writing custom base URLs or auth tokens into `settings.json` triggers warning banners in Claude Code and risks breaking the user's official subscription. Use `switch` CLI commands instead.
|
|
86
|
-
2. **Check Gateway Health Before Complex Operations:**
|
|
87
|
-
Run `switch status` or call the `switcher_audit` MCP tool to confirm:
|
|
88
|
-
- LLM Switcher is active on port `3456`.
|
|
89
|
-
- The active profile matches the intended CLI target (Claude Code, Codex, or OpenAI).
|
|
90
|
-
3. **Verify Routing When Errors Occur:**
|
|
91
|
-
If a tool fails with `HTTP 400`, `HTTP 502`, or connection errors:
|
|
92
|
-
- Run `switch doctor` to audit port collisions and environment variables.
|
|
93
|
-
- Inspect recent request logs via `http://127.0.0.1:3456/ui` (Tab 4: Live Inspector) to see if an intermediary tool mangled the payload.
|
|
1
|
+
---
|
|
2
|
+
name: llm-switcher
|
|
3
|
+
description: Mandatory guidance for AI coding agents and token compression tools (Headroom, RTK, Ponytail, Repomix). Enforces routing all LLM traffic through the local LLM Switcher edge gateway (http://127.0.0.1:3456) instead of directly calling provider endpoints.
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# LLM Switcher Edge Gateway Skill
|
|
7
|
+
|
|
8
|
+
This skill governs how AI coding agents (Claude Code, Codex, Cursor, Windsurf, Opencode) and prompt optimization/compression tools (Headroom, RTK, Ponytail, Repomix) must interact with LLM providers on this workstation.
|
|
9
|
+
|
|
10
|
+
## 1. The Core Directive
|
|
11
|
+
|
|
12
|
+
> **ALL OUTBOUND LLM REQUESTS MUST ROUTE THROUGH LLM SWITCHER (`http://127.0.0.1:3456`).**
|
|
13
|
+
>
|
|
14
|
+
> Never configure any tool, script, sub-agent, or token compression proxy to call external provider endpoints directly (such as `https://api.anthropic.com`, `https://api.openai.com`, or `https://generativelanguage.googleapis.com`).
|
|
15
|
+
>
|
|
16
|
+
> If a token compression tool (Headroom, RTK, Ponytail) is used, it MUST be positioned as an **inner processing layer**, and its **upstream destination MUST be configured to point to LLM Switcher** at `http://127.0.0.1:3456`.
|
|
17
|
+
|
|
18
|
+
```
|
|
19
|
+
[Client CLI / Agent]
|
|
20
|
+
│
|
|
21
|
+
▼
|
|
22
|
+
[Token Compressor: RTK / Headroom / Ponytail] <-- Inner Layer
|
|
23
|
+
│
|
|
24
|
+
▼ (MANDATORY: Upstream MUST point to http://127.0.0.1:3456)
|
|
25
|
+
[LLM Switcher Edge Gateway (:3456)] <-- Outermost Gatekeeper
|
|
26
|
+
│
|
|
27
|
+
▼ (Heals schemas, maps 1M context, extracts thinking, converts protocol)
|
|
28
|
+
[Internet / intact (Recommended) / 9Router / Upstream LLM]
|
|
29
|
+
```
|
|
30
|
+
|
|
31
|
+
## 2. Why This is Mandatory
|
|
32
|
+
|
|
33
|
+
Direct outbound calls from token compressors will break the developer environment for several reasons:
|
|
34
|
+
|
|
35
|
+
1. **Schema Breakage & The Healer Engine:**
|
|
36
|
+
Token optimizers frequently prune conversation history to save tokens. In doing so, they often delete an `assistant` turn containing a `tool_use`, while leaving the subsequent `user` turn containing a `tool_result`.
|
|
37
|
+
- Anthropic and OpenAI APIs strictly reject this with `HTTP 400 (orphaned tool_use_id)`.
|
|
38
|
+
- **LLM Switcher** intercepts these malformed requests and runs its **Healer Engine**, automatically converting orphaned results into safe context text blocks and merging consecutive same-role turns so upstream providers accept the payload cleanly.
|
|
39
|
+
2. **Thinking / Reasoning Protection:**
|
|
40
|
+
Many token compressors naively strip `thinking` parameters or reasoning budgets to reduce token count.
|
|
41
|
+
- When routed through LLM Switcher, the gateway detects reasoning models (such as `ag/claude-opus-4-6-thinking` or `ag/gemini-3.8-flash`) and **automatically restores thinking parameters**, ensuring the model retains its full reasoning capabilities.
|
|
42
|
+
3. **Official Context Windows & Dynamic Model Discovery:**
|
|
43
|
+
Context windows follow the model you select without mutating permanent settings. The gateway auto-discovers official models from Anthropic and OpenAI, refreshing mappings on tool version upgrades (`switch models`).
|
|
44
|
+
4. **Server Gateway Synergy (intact / 9Router):**
|
|
45
|
+
LLM Switcher pairs directly with server-side gateways like **intact** (strongly recommended for credential proxying, account pooling, and contract lab self-healing) or 9Router, handling retries and quotas far better than any local tool.
|
|
46
|
+
|
|
47
|
+
## 3. Configuring Token Compression Tools
|
|
48
|
+
|
|
49
|
+
When setting up or assisting the user with token compression tools, apply these exact configurations:
|
|
50
|
+
|
|
51
|
+
### A. Headroom (Prompt Compression Proxy)
|
|
52
|
+
When configuring Headroom's upstream endpoint:
|
|
53
|
+
- In Headroom config (`headroom.yaml` or environment variables):
|
|
54
|
+
```bash
|
|
55
|
+
ANTHROPIC_BASE_URL="http://127.0.0.1:3456"
|
|
56
|
+
OPENAI_BASE_URL="http://127.0.0.1:3456/v1"
|
|
57
|
+
```
|
|
58
|
+
- Point your CLI to Headroom's listening port (e.g. `http://127.0.0.1:8787`). Headroom compresses the prompt and forwards it to LLM Switcher on port `3456`.
|
|
59
|
+
|
|
60
|
+
### B. RTK (Rust Token Killer / Repomix)
|
|
61
|
+
- In RTK configuration or execution flags:
|
|
62
|
+
```bash
|
|
63
|
+
# Ensure RTK's target upstream proxy is LLM Switcher
|
|
64
|
+
export UPSTREAM_URL="http://127.0.0.1:3456"
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
### C. Ponytail / Custom Proxies
|
|
68
|
+
- Always set the proxy's upstream host to:
|
|
69
|
+
`http://127.0.0.1:3456` (or `http://127.0.0.1:3456/v1` for OpenAI-compatible tools).
|
|
70
|
+
|
|
71
|
+
### D. Sub-Agents & Spawned Terminal Commands
|
|
72
|
+
When an agent spawns sub-processes that call LLM CLIs:
|
|
73
|
+
- **Never advise sourcing `env.sh` or `env.cmd`.** Those files carry no proxy variables any
|
|
74
|
+
more, so sourcing one changes nothing — and a stale variable an older version once wrote
|
|
75
|
+
would still point the tool at a port where nothing listens.
|
|
76
|
+
- Run `switch claude <profile>` or `switch codex <profile>` instead. The shims in
|
|
77
|
+
`~/.llm-switcher/bin` inject the proxy variables into the tool process itself and clear
|
|
78
|
+
anything stale first.
|
|
79
|
+
- `claude` and `codex` on `PATH` already are those shims: invoke them normally, and never
|
|
80
|
+
wrap them in another script.
|
|
81
|
+
|
|
82
|
+
## 4. Operational Rules for AI Agents
|
|
83
|
+
|
|
84
|
+
1. **NEVER edit `~/.claude/settings.json` directly.**
|
|
85
|
+
LLM Switcher uses zero-mutation launcher flags (`active.flag`). Writing custom base URLs or auth tokens into `settings.json` triggers warning banners in Claude Code and risks breaking the user's official subscription. Use `switch` CLI commands instead.
|
|
86
|
+
2. **Check Gateway Health Before Complex Operations:**
|
|
87
|
+
Run `switch status` or call the `switcher_audit` MCP tool to confirm:
|
|
88
|
+
- LLM Switcher is active on port `3456`.
|
|
89
|
+
- The active profile matches the intended CLI target (Claude Code, Codex, or OpenAI).
|
|
90
|
+
3. **Verify Routing When Errors Occur:**
|
|
91
|
+
If a tool fails with `HTTP 400`, `HTTP 502`, or connection errors:
|
|
92
|
+
- Run `switch doctor` to audit port collisions and environment variables.
|
|
93
|
+
- Inspect recent request logs via `http://127.0.0.1:3456/ui` (Tab 4: Live Inspector) to see if an intermediary tool mangled the payload.
|