llm-switcher 1.1.11 → 1.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +51 -0
- package/README.md +202 -257
- package/README.vi.md +200 -256
- package/blindfold/blindfold.mjs +200 -53
- package/blindfold/make-certs.sh +26 -7
- package/catalog.mjs +246 -0
- package/config.example.json +12 -34
- package/docs/TOKEN-OPTIMIZER-INTEROP.md +110 -110
- package/docs/codex-blindfold.md +28 -17
- package/docs/cross-platform.md +16 -7
- package/docs/diagrams/ir-healer-pipeline.mmd +16 -0
- package/docs/diagrams/ir-healer-pipeline.png +0 -0
- package/docs/diagrams/ir-healer-pipeline.svg +90 -0
- package/docs/diagrams/ir-translation-pipeline.html +14925 -0
- package/docs/diagrams/ir-translation-pipeline.sequence.json +31 -0
- package/docs/diagrams/ir-translation-pipeline.svg +5128 -0
- package/docs/diagrams/system-architecture.architecture.json +76 -0
- package/docs/diagrams/system-architecture.html +14978 -0
- package/docs/diagrams/system-architecture.svg +5147 -0
- package/docs/diagrams/system-topology.mmd +30 -0
- package/docs/diagrams/system-topology.png +0 -0
- package/docs/diagrams/system-topology.svg +125 -0
- package/docs/response-matrix.json +1130 -1130
- package/ensure-ca-bundle.mjs +28 -0
- package/formats.mjs +13 -155
- package/mcp.mjs +39 -11
- package/package.json +1 -1
- package/proxy.mjs +92 -27
- package/shim.mjs +200 -57
- package/skills/llm-switcher/SKILL.md +93 -88
- package/state.mjs +1100 -191
- package/switch +0 -0
- package/switch.cmd +2 -2
- package/switch.mjs +228 -53
- package/tests/blindfold-e2e.test.mjs +380 -0
- package/tests/blindfold-task5.test.mjs +429 -0
- package/tests/blindfold.task3.test.mjs +700 -0
- package/tests/blindfold.test.mjs +10 -5
- package/tests/catalog.test.mjs +147 -0
- package/tests/contract-lab.test.mjs +22 -7
- package/tests/formats.test.mjs +33 -46
- package/tests/gateway.e2e.test.mjs +136 -36
- package/tests/helpers.mjs +24 -24
- package/tests/lifecycle.test.mjs +16 -10
- package/tests/live-optimizer-interop.mjs +205 -205
- package/tests/mcp.test.mjs +78 -2
- package/tests/real-user-sim.test.mjs +464 -0
- package/tests/shim.test.mjs +159 -66
- package/tests/state.test.mjs +975 -193
- package/tests/switch.test.mjs +446 -2
- package/ui.html +61 -154
package/shim.mjs
CHANGED
|
@@ -26,33 +26,121 @@ import fs from 'node:fs';
|
|
|
26
26
|
import path from 'node:path';
|
|
27
27
|
import os from 'node:os';
|
|
28
28
|
import { execFileSync } from 'node:child_process';
|
|
29
|
+
import { fileURLToPath } from 'node:url';
|
|
29
30
|
import { paths, STATE_DIR } from './state.mjs';
|
|
30
31
|
|
|
31
32
|
export const SHIM_DIR = path.join(os.homedir(), '.llm-switcher', 'bin');
|
|
32
33
|
|
|
33
|
-
//
|
|
34
|
-
//
|
|
35
|
-
//
|
|
36
|
-
//
|
|
37
|
-
|
|
38
|
-
export const CODE_X_CATALOG_PATH = paths.codexCatalog.replace(/\\/g, '/');
|
|
34
|
+
// The shims hand the CA work to Node: reading files, comparing content and renaming atomically
|
|
35
|
+
// has no place in a shell script, and none at all in cmd.exe. It is a FILE rather than
|
|
36
|
+
// `node -e`, so neither cmd.exe quoting nor URL encoding can corrupt it: pathToFileURL turns a
|
|
37
|
+
// space in the checkout into %20, which cmd.exe then reads as the (undefined) argument %2.
|
|
38
|
+
const STATE_HELPER = path.join(path.dirname(fileURLToPath(import.meta.url)), 'ensure-ca-bundle.mjs');
|
|
39
39
|
|
|
40
40
|
// CLIs to wrap. `claude` is the most important case (--resume), codex included for completeness.
|
|
41
41
|
export const SHIMMED = ['claude', 'codex'];
|
|
42
42
|
|
|
43
|
-
|
|
43
|
+
// One tool, one env file. Sourcing the shared one is how one tool used to capture the other.
|
|
44
|
+
const envFileOf = (name) => (name === 'codex' ? 'env-codex.sh' : 'env-claude.sh');
|
|
45
|
+
|
|
46
|
+
const POSIX_TEMPLATE = (name) => {
|
|
47
|
+
const envFile = envFileOf(name);
|
|
48
|
+
const caBlock = name === 'codex' ? '' : `
|
|
49
|
+
# R2: one bundle holds the user's own CA and the switcher CA, so trusting the interceptor never
|
|
50
|
+
# costs a certificate the user already had. Decided here, at launch, because only now is it known
|
|
51
|
+
# whether the user brought one — and never while this tool is off.
|
|
52
|
+
if [ "$TOOL_ACTIVE" = "1" ]; then
|
|
53
|
+
if [ -z "$INHERITED_CA" ]; then
|
|
54
|
+
if [ -f "$SWITCHER_DIR/blindfold/certs/ca.pem" ]; then
|
|
55
|
+
export NODE_EXTRA_CA_CERTS="$SWITCHER_DIR/blindfold/certs/ca.pem"
|
|
56
|
+
else
|
|
57
|
+
echo "[llm-switcher] $SWITCHER_DIR/blindfold/certs/ca.pem not found; NODE_EXTRA_CA_CERTS kept as inherited." >&2
|
|
58
|
+
fi
|
|
59
|
+
else
|
|
60
|
+
CA_BUNDLE="$(node "$STATE_HELPER" "$INHERITED_CA" "$SWITCHER_DIR/blindfold/certs/ca.pem" "$SWITCHER_DIR")" || CA_BUNDLE=""
|
|
61
|
+
if [ -n "$CA_BUNDLE" ] && [ -f "$CA_BUNDLE" ]; then
|
|
62
|
+
export NODE_EXTRA_CA_CERTS="$CA_BUNDLE"
|
|
63
|
+
fi
|
|
64
|
+
fi
|
|
65
|
+
fi`;
|
|
66
|
+
return `#!/usr/bin/env bash
|
|
44
67
|
# Auto-generated by LLM Switcher — DO NOT EDIT.
|
|
45
|
-
#
|
|
46
|
-
#
|
|
68
|
+
# Hand THIS tool its own gateway env, scrub what an older switcher left behind, then run the real
|
|
69
|
+
# "${name}", from any shell — including one that never sourced anything (claude --resume).
|
|
47
70
|
SWITCHER_DIR="${STATE_DIR}"
|
|
71
|
+
STATE_HELPER="${STATE_HELPER}"
|
|
48
72
|
SHIM_DIR="\${BASH_SOURCE%/*}"
|
|
49
73
|
|
|
50
|
-
#
|
|
51
|
-
#
|
|
52
|
-
|
|
53
|
-
|
|
74
|
+
# Captured BEFORE the tool's env file is sourced: the CA decision must see the value this shell
|
|
75
|
+
# already had, not one this run is about to write.
|
|
76
|
+
INHERITED_CA="\${NODE_EXTRA_CA_CERTS:-}"
|
|
77
|
+
|
|
78
|
+
# Only this tool's own file, only while the gateway is on, and only files this account owns:
|
|
79
|
+
# another account could have created the directory or the file. Non-empty means the tool runs;
|
|
80
|
+
# an empty file means it is off, and then nothing is sourced at all.
|
|
81
|
+
if [ -O "$SWITCHER_DIR" ] && [ -O "$SWITCHER_DIR/${envFile}" ] && [ -f "$SWITCHER_DIR/active.flag" ] && [ -s "$SWITCHER_DIR/${envFile}" ]; then
|
|
82
|
+
. "$SWITCHER_DIR/${envFile}"
|
|
83
|
+
TOOL_ACTIVE=1
|
|
84
|
+
else
|
|
85
|
+
TOOL_ACTIVE=0
|
|
86
|
+
fi
|
|
87
|
+
|
|
88
|
+
# R8: a variable is removed ONLY when its value is one an older switcher wrote. The user's own
|
|
89
|
+
# proxy, base URL, model or LLM_SWITCHER_* setting is theirs and is kept.
|
|
90
|
+
GW_PORT="$( { [ -O "$SWITCHER_DIR/gateway.port" ] && tr -d '[:space:]' < "$SWITCHER_DIR/gateway.port"; } 2>/dev/null )"
|
|
91
|
+
case "$GW_PORT" in ''|*[!0-9]*) GW_PORT=3456 ;; esac
|
|
92
|
+
PORTS="$GW_PORT 3456"
|
|
93
|
+
[ -n "$LLM_SWITCHER_PORT" ] && PORTS="$PORTS $LLM_SWITCHER_PORT"
|
|
94
|
+
|
|
95
|
+
# Every loopback shape the switcher ever wrote: host, optional /v1, optional trailing slash.
|
|
96
|
+
SWITCHER_URLS=""
|
|
97
|
+
for _p in $PORTS; do
|
|
98
|
+
for _h in 127.0.0.1 localhost '[::1]'; do
|
|
99
|
+
SWITCHER_URLS="$SWITCHER_URLS $_h:$_p $_h:$_p/ $_h:$_p/v1 $_h:$_p/v1/"
|
|
100
|
+
done
|
|
101
|
+
done
|
|
102
|
+
|
|
103
|
+
_u="\${ANTHROPIC_BASE_URL:-}"
|
|
104
|
+
if [ -n "$_u" ]; then
|
|
105
|
+
case " $SWITCHER_URLS " in
|
|
106
|
+
*" \${_u#http://} "*) unset ANTHROPIC_BASE_URL ;;
|
|
107
|
+
esac
|
|
108
|
+
fi
|
|
109
|
+
_u="\${OPENAI_BASE_URL:-}"
|
|
110
|
+
if [ -n "$_u" ]; then
|
|
111
|
+
case " $SWITCHER_URLS " in
|
|
112
|
+
*" \${_u#http://} "*) unset OPENAI_BASE_URL ;;
|
|
113
|
+
esac
|
|
54
114
|
fi
|
|
55
115
|
|
|
116
|
+
# The suffix the old switcher forced on every tier. Nothing else matches: the user's own
|
|
117
|
+
# ANTHROPIC_MODEL=opus, or a real model id, is theirs.
|
|
118
|
+
case "\${ANTHROPIC_MODEL:-}" in
|
|
119
|
+
opus\\[1m\\]|sonnet\\[1m\\]|haiku\\[1m\\]|fable\\[1m\\]) unset ANTHROPIC_MODEL ;;
|
|
120
|
+
esac
|
|
121
|
+
tier_is_stale() {
|
|
122
|
+
[ -n "$1" ] || return 1
|
|
123
|
+
case "$(printf '%s' "$1" | tr '[:upper:]' '[:lower:]')" in
|
|
124
|
+
"$2"\\[1m\\]) return 0 ;;
|
|
125
|
+
*) return 1 ;;
|
|
126
|
+
esac
|
|
127
|
+
}
|
|
128
|
+
if tier_is_stale "$ANTHROPIC_DEFAULT_OPUS_MODEL" opus; then unset ANTHROPIC_DEFAULT_OPUS_MODEL; fi
|
|
129
|
+
if tier_is_stale "$ANTHROPIC_DEFAULT_SONNET_MODEL" sonnet; then unset ANTHROPIC_DEFAULT_SONNET_MODEL; fi
|
|
130
|
+
if tier_is_stale "$ANTHROPIC_DEFAULT_HAIKU_MODEL" haiku; then unset ANTHROPIC_DEFAULT_HAIKU_MODEL; fi
|
|
131
|
+
if tier_is_stale "$ANTHROPIC_DEFAULT_FABLE_MODEL" fable; then unset ANTHROPIC_DEFAULT_FABLE_MODEL; fi
|
|
132
|
+
|
|
133
|
+
[ "\${CLAUDE_CODE_DISABLE_UNKNOWN_MODEL_WINDOW_ENFORCEMENT:-}" = "1" ] && unset CLAUDE_CODE_DISABLE_UNKNOWN_MODEL_WINDOW_ENFORCEMENT
|
|
134
|
+
[ "\${CLAUDE_CODE_AUTO_COMPACT_WINDOW:-}" = "900000" ] && unset CLAUDE_CODE_AUTO_COMPACT_WINDOW
|
|
135
|
+
[ "\${OPENAI_MAX_CONTEXT_TOKENS:-}" = "1000000" ] && unset OPENAI_MAX_CONTEXT_TOKENS
|
|
136
|
+
|
|
137
|
+
# LLM_SWITCHER_CODEX_* was always a shim input, never the user's. Every OTHER LLM_SWITCHER_*
|
|
138
|
+
# name is a user setting and is kept.
|
|
139
|
+
for _v in $(compgen -v LLM_SWITCHER_CODEX_ 2>/dev/null); do
|
|
140
|
+
unset "$_v"
|
|
141
|
+
done
|
|
142
|
+
${caBlock}
|
|
143
|
+
|
|
56
144
|
# Find the real binary: drop the shim directory from PATH so it doesn't call itself.
|
|
57
145
|
CLEAN_PATH=""
|
|
58
146
|
IFS=':'
|
|
@@ -72,65 +160,120 @@ if [ -z "$REAL" ]; then
|
|
|
72
160
|
exit 127
|
|
73
161
|
fi
|
|
74
162
|
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
. "$SWITCHER_DIR/env-codex.sh"
|
|
78
|
-
fi
|
|
79
|
-
|
|
80
|
-
CODEX_SWITCHER_ARGS=()
|
|
81
|
-
if [ -n "\${LLM_SWITCHER_CODEX_BASE_URL:-}" ]; then
|
|
82
|
-
CODEX_SWITCHER_ARGS+=(--config "openai_base_url=\${LLM_SWITCHER_CODEX_BASE_URL}")
|
|
83
|
-
fi
|
|
84
|
-
if [ -O "${CODE_X_CATALOG_PATH}" ]; then
|
|
85
|
-
CODEX_SWITCHER_ARGS+=(--config "model_catalog_json=${CODE_X_CATALOG_PATH}")
|
|
86
|
-
fi
|
|
87
|
-
if [ -n "\${LLM_SWITCHER_CODEX_MAIN_MODEL:-}" ]; then
|
|
88
|
-
CODEX_SWITCHER_ARGS+=(--config "model=\\"\${LLM_SWITCHER_CODEX_MAIN_MODEL}\\"")
|
|
89
|
-
fi
|
|
90
|
-
if [ -n "\${LLM_SWITCHER_CODEX_REVIEW_MODEL:-}" ]; then
|
|
91
|
-
CODEX_SWITCHER_ARGS+=(--config "review_model=\\"\${LLM_SWITCHER_CODEX_REVIEW_MODEL}\\"")
|
|
92
|
-
fi
|
|
93
|
-
if [ -n "\${LLM_SWITCHER_CODEX_SUBAGENT_MODEL:-}" ]; then
|
|
94
|
-
CODEX_SWITCHER_ARGS+=(--config "agents.default_subagent_model=\\"\${LLM_SWITCHER_CODEX_SUBAGENT_MODEL}\\"")
|
|
95
|
-
fi
|
|
96
|
-
if [ -n "\${LLM_SWITCHER_CODEX_CONTEXT_WINDOW:-}" ]; then
|
|
97
|
-
CODEX_SWITCHER_ARGS+=(--config "model_context_window=\${LLM_SWITCHER_CODEX_CONTEXT_WINDOW}")
|
|
98
|
-
fi
|
|
99
|
-
if [ -n "\${LLM_SWITCHER_CODEX_AUTO_COMPACT_LIMIT:-}" ]; then
|
|
100
|
-
CODEX_SWITCHER_ARGS+=(--config "model_auto_compact_token_limit=\${LLM_SWITCHER_CODEX_AUTO_COMPACT_LIMIT}")
|
|
101
|
-
fi
|
|
102
|
-
|
|
103
|
-
exec "$REAL" "\${CODEX_SWITCHER_ARGS[@]}" "$@"` : `exec "$REAL" "$@"`}
|
|
163
|
+
# No override of any kind: the tool keeps its own configuration and its own model names (F3, R1).
|
|
164
|
+
exec "$REAL" "$@"
|
|
104
165
|
`;
|
|
166
|
+
};
|
|
105
167
|
|
|
106
|
-
const WINDOWS_TEMPLATE = (name) =>
|
|
168
|
+
const WINDOWS_TEMPLATE = (name) => {
|
|
169
|
+
const envFile = envFileOf(name).replace(/\.sh$/, '.cmd');
|
|
170
|
+
const caBlock = name === 'codex' ? '' : `
|
|
171
|
+
REM R2: one bundle holds the user's own CA and the switcher CA, so trusting the interceptor
|
|
172
|
+
REM never costs a certificate the user already had. Decided here, at launch, because only now
|
|
173
|
+
REM is it known whether the user brought one - and never while this tool is off.
|
|
174
|
+
if not "%TOOL_ACTIVE%"=="1" goto :CA_DONE
|
|
175
|
+
if defined INHERITED_CA goto :CA_BUNDLE
|
|
176
|
+
if exist "%SWITCHER_DIR%\\blindfold\\certs\\ca.pem" set "NODE_EXTRA_CA_CERTS=%SWITCHER_DIR%\\blindfold\\certs\\ca.pem"
|
|
177
|
+
if not exist "%SWITCHER_DIR%\\blindfold\\certs\\ca.pem" echo [llm-switcher] %SWITCHER_DIR%\\blindfold\\certs\\ca.pem not found; NODE_EXTRA_CA_CERTS kept as inherited. 1>&2
|
|
178
|
+
goto :CA_DONE
|
|
179
|
+
:CA_BUNDLE
|
|
180
|
+
set "CA_BUNDLE="
|
|
181
|
+
for /f "delims=" %%B in ('node "%STATE_HELPER%" "%INHERITED_CA%" "%SWITCHER_DIR%\\blindfold\\certs\\ca.pem" "%SWITCHER_DIR%" 2^>nul') do set "CA_BUNDLE=%%B"
|
|
182
|
+
if not defined CA_BUNDLE goto :CA_DONE
|
|
183
|
+
if not exist "%CA_BUNDLE%" goto :CA_DONE
|
|
184
|
+
set "NODE_EXTRA_CA_CERTS=%CA_BUNDLE%"
|
|
185
|
+
:CA_DONE`;
|
|
186
|
+
return `@echo off
|
|
107
187
|
REM Auto-generated by LLM Switcher - DO NOT EDIT.
|
|
188
|
+
setlocal
|
|
108
189
|
set "SWITCHER_DIR=${STATE_DIR}"
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
if
|
|
190
|
+
${name === 'codex' ? '' : `set "STATE_HELPER=${STATE_HELPER}"
|
|
191
|
+
|
|
192
|
+
REM Captured BEFORE the tool's env file runs: the CA decision must see the value this shell
|
|
193
|
+
REM already had, not one this run is about to write.
|
|
194
|
+
set "INHERITED_CA=%NODE_EXTRA_CA_CERTS%"
|
|
195
|
+
`}
|
|
196
|
+
|
|
197
|
+
REM Only this tool's own file, only while the gateway is on, and only a non-empty one: an empty
|
|
198
|
+
REM file means this tool is off. The shared env.cmd is never called (F5).
|
|
199
|
+
set "TOOL_ACTIVE=0"
|
|
200
|
+
if exist "%SWITCHER_DIR%\\active.flag" if exist "%SWITCHER_DIR%\\${envFile}" (
|
|
201
|
+
for %%A in ("%SWITCHER_DIR%\\${envFile}") do if %%~zA GTR 0 set "TOOL_ACTIVE=1"
|
|
202
|
+
)
|
|
203
|
+
if "%TOOL_ACTIVE%"=="1" call "%SWITCHER_DIR%\\${envFile}"
|
|
204
|
+
|
|
205
|
+
REM R8: a variable is removed ONLY when its value is one an older switcher wrote. The user's own
|
|
206
|
+
REM proxy, base URL, model or LLM_SWITCHER_* setting is theirs and is kept.
|
|
207
|
+
set "_GWPORT="
|
|
208
|
+
if exist "%SWITCHER_DIR%\\gateway.port" for /f "usebackq delims=" %%P in ("%SWITCHER_DIR%\\gateway.port") do set "_GWPORT=%%P"
|
|
209
|
+
if not defined _GWPORT set "_GWPORT=3456"
|
|
210
|
+
if defined ANTHROPIC_BASE_URL call :SCRUB_URL ANTHROPIC_BASE_URL "%ANTHROPIC_BASE_URL%"
|
|
211
|
+
if defined OPENAI_BASE_URL call :SCRUB_URL OPENAI_BASE_URL "%OPENAI_BASE_URL%"
|
|
212
|
+
if /i "%ANTHROPIC_MODEL%"=="opus[1m]" set "ANTHROPIC_MODEL="
|
|
213
|
+
if /i "%ANTHROPIC_MODEL%"=="sonnet[1m]" set "ANTHROPIC_MODEL="
|
|
214
|
+
if /i "%ANTHROPIC_MODEL%"=="haiku[1m]" set "ANTHROPIC_MODEL="
|
|
215
|
+
if /i "%ANTHROPIC_MODEL%"=="fable[1m]" set "ANTHROPIC_MODEL="
|
|
216
|
+
if defined ANTHROPIC_DEFAULT_OPUS_MODEL call :SCRUB_TIER ANTHROPIC_DEFAULT_OPUS_MODEL "%ANTHROPIC_DEFAULT_OPUS_MODEL%" opus
|
|
217
|
+
if defined ANTHROPIC_DEFAULT_SONNET_MODEL call :SCRUB_TIER ANTHROPIC_DEFAULT_SONNET_MODEL "%ANTHROPIC_DEFAULT_SONNET_MODEL%" sonnet
|
|
218
|
+
if defined ANTHROPIC_DEFAULT_HAIKU_MODEL call :SCRUB_TIER ANTHROPIC_DEFAULT_HAIKU_MODEL "%ANTHROPIC_DEFAULT_HAIKU_MODEL%" haiku
|
|
219
|
+
if defined ANTHROPIC_DEFAULT_FABLE_MODEL call :SCRUB_TIER ANTHROPIC_DEFAULT_FABLE_MODEL "%ANTHROPIC_DEFAULT_FABLE_MODEL%" fable
|
|
220
|
+
if "%CLAUDE_CODE_DISABLE_UNKNOWN_MODEL_WINDOW_ENFORCEMENT%"=="1" set "CLAUDE_CODE_DISABLE_UNKNOWN_MODEL_WINDOW_ENFORCEMENT="
|
|
221
|
+
if "%CLAUDE_CODE_AUTO_COMPACT_WINDOW%"=="900000" set "CLAUDE_CODE_AUTO_COMPACT_WINDOW="
|
|
222
|
+
if "%OPENAI_MAX_CONTEXT_TOKENS%"=="1000000" set "OPENAI_MAX_CONTEXT_TOKENS="
|
|
223
|
+
for /f "tokens=1 delims==" %%V in ('set LLM_SWITCHER_CODEX_ 2^>nul ^| findstr /b /c:"LLM_SWITCHER_CODEX_"') do set "%%V="
|
|
224
|
+
${caBlock}
|
|
225
|
+
|
|
226
|
+
REM Find the real binary: drop this shim's own directory so it never calls itself.
|
|
120
227
|
for /f "delims=" %%i in ('where ${name}.cmd 2^>nul ^| findstr /v /i "\\.llm-switcher\\\\bin"') do (
|
|
121
|
-
call "%%i"
|
|
228
|
+
call "%%i" %*
|
|
122
229
|
exit /b
|
|
123
230
|
)
|
|
124
231
|
for /f "delims=" %%i in ('where ${name}.exe 2^>nul ^| findstr /v /i "\\.llm-switcher\\\\bin"') do (
|
|
125
|
-
call "%%i"
|
|
232
|
+
call "%%i" %*
|
|
126
233
|
exit /b
|
|
127
234
|
)
|
|
128
235
|
echo [llm-switcher] cannot find the real '${name}' on PATH.>&2
|
|
129
236
|
exit /b 127
|
|
237
|
+
|
|
238
|
+
REM ---- R8 subroutines ----------------------------------------------------
|
|
239
|
+
|
|
240
|
+
REM %1 = variable name, %2 = its value. Unsets %1 only for a loopback gateway URL on a port this
|
|
241
|
+
REM switcher has used: the recorded one, the default 3456, or LLM_SWITCHER_PORT.
|
|
242
|
+
:SCRUB_URL
|
|
243
|
+
set "_T=%~2"
|
|
244
|
+
if /i "%_T:~0,7%"=="http://" set "_T=%_T:~7%"
|
|
245
|
+
set "_R="
|
|
246
|
+
if /i "%_T:~0,10%"=="127.0.0.1:" set "_R=%_T:~10%"
|
|
247
|
+
if /i "%_T:~0,10%"=="localhost:" set "_R=%_T:~10%"
|
|
248
|
+
if /i "%_T:~0,6%"=="[::1]:" set "_R=%_T:~6%"
|
|
249
|
+
if not defined _R goto :eof
|
|
250
|
+
if /i "%_R:~-4%"=="/v1/" set "_R=%_R:~0,-4%"
|
|
251
|
+
if /i "%_R:~-3%"=="/v1" set "_R=%_R:~0,-3%"
|
|
252
|
+
if "%_R:~-1%"=="/" set "_R=%_R:~0,-1%"
|
|
253
|
+
if not defined _R goto :eof
|
|
254
|
+
if not "%_R%"=="" for /f "delims=0123456789" %%X in ("%_R%") do goto :eof
|
|
255
|
+
if "%_R%"=="%_GWPORT%" goto :DROP_URL
|
|
256
|
+
if "%_R%"=="3456" goto :DROP_URL
|
|
257
|
+
if defined LLM_SWITCHER_PORT if "%_R%"=="%LLM_SWITCHER_PORT%" goto :DROP_URL
|
|
258
|
+
goto :eof
|
|
259
|
+
:DROP_URL
|
|
260
|
+
set "%~1="
|
|
261
|
+
goto :eof
|
|
262
|
+
|
|
263
|
+
REM %1 = variable name, %2 = its value, %3 = the lowercase tier it must still match. A value of
|
|
264
|
+
REM another tier (ANTHROPIC_DEFAULT_OPUS_MODEL=sonnet[1m]) is not the switcher's and is kept.
|
|
265
|
+
:SCRUB_TIER
|
|
266
|
+
if /i "%~2"=="%~3[1m]" set "%~1="
|
|
267
|
+
goto :eof
|
|
130
268
|
`;
|
|
269
|
+
};
|
|
131
270
|
|
|
132
271
|
export function renderShim(name, platform = process.platform) {
|
|
133
|
-
|
|
272
|
+
if (platform !== 'win32') return POSIX_TEMPLATE(name);
|
|
273
|
+
// cmd.exe finds a subroutine by scanning line starts, and on a file written with bare LF
|
|
274
|
+
// it drifts: a label near the end is reported as "cannot find the batch label". CRLF is
|
|
275
|
+
// what every .cmd on disk already is, and it is what cmd.exe expects.
|
|
276
|
+
return WINDOWS_TEMPLATE(name).replace(/\r?\n/g, '\r\n');
|
|
134
277
|
}
|
|
135
278
|
|
|
136
279
|
function shimPath(name) {
|
|
@@ -1,88 +1,93 @@
|
|
|
1
|
-
---
|
|
2
|
-
name: llm-switcher
|
|
3
|
-
description: Mandatory guidance for AI coding agents and token compression tools (Headroom, RTK, Ponytail, Repomix). Enforces routing all LLM traffic through the local LLM Switcher edge gateway (http://127.0.0.1:3456) instead of directly calling provider endpoints.
|
|
4
|
-
---
|
|
5
|
-
|
|
6
|
-
# LLM Switcher Edge Gateway Skill
|
|
7
|
-
|
|
8
|
-
This skill governs how AI coding agents (Claude Code, Codex, Cursor, Windsurf, Opencode) and prompt optimization/compression tools (Headroom, RTK, Ponytail, Repomix) must interact with LLM providers on this workstation.
|
|
9
|
-
|
|
10
|
-
## 1. The Core Directive
|
|
11
|
-
|
|
12
|
-
> **ALL OUTBOUND LLM REQUESTS MUST ROUTE THROUGH LLM SWITCHER (`http://127.0.0.1:3456`).**
|
|
13
|
-
>
|
|
14
|
-
> Never configure any tool, script, sub-agent, or token compression proxy to call external provider endpoints directly (such as `https://api.anthropic.com`, `https://api.openai.com`, or `https://generativelanguage.googleapis.com`).
|
|
15
|
-
>
|
|
16
|
-
> If a token compression tool (Headroom, RTK, Ponytail) is used, it MUST be positioned as an **inner processing layer**, and its **upstream destination MUST be configured to point to LLM Switcher** at `http://127.0.0.1:3456`.
|
|
17
|
-
|
|
18
|
-
```
|
|
19
|
-
[Client CLI / Agent]
|
|
20
|
-
│
|
|
21
|
-
▼
|
|
22
|
-
[Token Compressor: RTK / Headroom / Ponytail] <-- Inner Layer
|
|
23
|
-
│
|
|
24
|
-
▼ (MANDATORY: Upstream MUST point to http://127.0.0.1:3456)
|
|
25
|
-
[LLM Switcher Edge Gateway (:3456)] <-- Outermost Gatekeeper
|
|
26
|
-
│
|
|
27
|
-
▼ (Heals schemas, maps 1M context, extracts thinking, converts protocol)
|
|
28
|
-
[Internet / 9Router / Upstream LLM]
|
|
29
|
-
```
|
|
30
|
-
|
|
31
|
-
## 2. Why This is Mandatory
|
|
32
|
-
|
|
33
|
-
Direct outbound calls from token compressors will break the developer environment for several reasons:
|
|
34
|
-
|
|
35
|
-
1. **Schema Breakage & The Healer Engine:**
|
|
36
|
-
Token optimizers frequently prune conversation history to save tokens. In doing so, they often delete an `assistant` turn containing a `tool_use`, while leaving the subsequent `user` turn containing a `tool_result`.
|
|
37
|
-
- Anthropic and OpenAI APIs strictly reject this with `HTTP 400 (orphaned tool_use_id)`.
|
|
38
|
-
- **LLM Switcher** intercepts these malformed requests and runs its **Healer Engine**, automatically converting orphaned results into safe context text blocks and merging consecutive same-role turns so upstream providers accept the payload cleanly.
|
|
39
|
-
2. **Thinking / Reasoning Protection:**
|
|
40
|
-
Many token compressors naively strip `thinking` parameters or reasoning budgets to reduce token count.
|
|
41
|
-
- When routed through LLM Switcher, the gateway detects reasoning models (such as `ag/claude-opus-4-6-thinking` or `ag/gemini-3.8-flash`) and **automatically restores thinking parameters**, ensuring the model retains its full reasoning capabilities.
|
|
42
|
-
3. **
|
|
43
|
-
|
|
44
|
-
4. **
|
|
45
|
-
LLM Switcher
|
|
46
|
-
|
|
47
|
-
## 3. Configuring Token Compression Tools
|
|
48
|
-
|
|
49
|
-
When setting up or assisting the user with token compression tools, apply these exact configurations:
|
|
50
|
-
|
|
51
|
-
### A. Headroom (Prompt Compression Proxy)
|
|
52
|
-
When configuring Headroom's upstream endpoint:
|
|
53
|
-
- In Headroom config (`headroom.yaml` or environment variables):
|
|
54
|
-
```bash
|
|
55
|
-
ANTHROPIC_BASE_URL="http://127.0.0.1:3456"
|
|
56
|
-
OPENAI_BASE_URL="http://127.0.0.1:3456/v1"
|
|
57
|
-
```
|
|
58
|
-
- Point your CLI to Headroom's listening port (e.g. `http://127.0.0.1:8787`). Headroom compresses the prompt and forwards it to LLM Switcher on port `3456`.
|
|
59
|
-
|
|
60
|
-
### B. RTK (Rust Token Killer / Repomix)
|
|
61
|
-
- In RTK configuration or execution flags:
|
|
62
|
-
```bash
|
|
63
|
-
# Ensure RTK's target upstream proxy is LLM Switcher
|
|
64
|
-
export UPSTREAM_URL="http://127.0.0.1:3456"
|
|
65
|
-
```
|
|
66
|
-
|
|
67
|
-
### C. Ponytail / Custom Proxies
|
|
68
|
-
- Always set the proxy's upstream host to:
|
|
69
|
-
`http://127.0.0.1:3456` (or `http://127.0.0.1:3456/v1` for OpenAI-compatible tools).
|
|
70
|
-
|
|
71
|
-
### D. Sub-Agents & Spawned Terminal Commands
|
|
72
|
-
When an agent spawns sub-processes
|
|
73
|
-
- **
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
-
|
|
1
|
+
---
|
|
2
|
+
name: llm-switcher
|
|
3
|
+
description: Mandatory guidance for AI coding agents and token compression tools (Headroom, RTK, Ponytail, Repomix). Enforces routing all LLM traffic through the local LLM Switcher edge gateway (http://127.0.0.1:3456) instead of directly calling provider endpoints.
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# LLM Switcher Edge Gateway Skill
|
|
7
|
+
|
|
8
|
+
This skill governs how AI coding agents (Claude Code, Codex, Cursor, Windsurf, Opencode) and prompt optimization/compression tools (Headroom, RTK, Ponytail, Repomix) must interact with LLM providers on this workstation.
|
|
9
|
+
|
|
10
|
+
## 1. The Core Directive
|
|
11
|
+
|
|
12
|
+
> **ALL OUTBOUND LLM REQUESTS MUST ROUTE THROUGH LLM SWITCHER (`http://127.0.0.1:3456`).**
|
|
13
|
+
>
|
|
14
|
+
> Never configure any tool, script, sub-agent, or token compression proxy to call external provider endpoints directly (such as `https://api.anthropic.com`, `https://api.openai.com`, or `https://generativelanguage.googleapis.com`).
|
|
15
|
+
>
|
|
16
|
+
> If a token compression tool (Headroom, RTK, Ponytail) is used, it MUST be positioned as an **inner processing layer**, and its **upstream destination MUST be configured to point to LLM Switcher** at `http://127.0.0.1:3456`.
|
|
17
|
+
|
|
18
|
+
```
|
|
19
|
+
[Client CLI / Agent]
|
|
20
|
+
│
|
|
21
|
+
▼
|
|
22
|
+
[Token Compressor: RTK / Headroom / Ponytail] <-- Inner Layer
|
|
23
|
+
│
|
|
24
|
+
▼ (MANDATORY: Upstream MUST point to http://127.0.0.1:3456)
|
|
25
|
+
[LLM Switcher Edge Gateway (:3456)] <-- Outermost Gatekeeper
|
|
26
|
+
│
|
|
27
|
+
▼ (Heals schemas, maps 1M context, extracts thinking, converts protocol)
|
|
28
|
+
[Internet / intact (Recommended) / 9Router / Upstream LLM]
|
|
29
|
+
```
|
|
30
|
+
|
|
31
|
+
## 2. Why This is Mandatory
|
|
32
|
+
|
|
33
|
+
Direct outbound calls from token compressors will break the developer environment for several reasons:
|
|
34
|
+
|
|
35
|
+
1. **Schema Breakage & The Healer Engine:**
|
|
36
|
+
Token optimizers frequently prune conversation history to save tokens. In doing so, they often delete an `assistant` turn containing a `tool_use`, while leaving the subsequent `user` turn containing a `tool_result`.
|
|
37
|
+
- Anthropic and OpenAI APIs strictly reject this with `HTTP 400 (orphaned tool_use_id)`.
|
|
38
|
+
- **LLM Switcher** intercepts these malformed requests and runs its **Healer Engine**, automatically converting orphaned results into safe context text blocks and merging consecutive same-role turns so upstream providers accept the payload cleanly.
|
|
39
|
+
2. **Thinking / Reasoning Protection:**
|
|
40
|
+
Many token compressors naively strip `thinking` parameters or reasoning budgets to reduce token count.
|
|
41
|
+
- When routed through LLM Switcher, the gateway detects reasoning models (such as `ag/claude-opus-4-6-thinking` or `ag/gemini-3.8-flash`) and **automatically restores thinking parameters**, ensuring the model retains its full reasoning capabilities.
|
|
42
|
+
3. **Official Context Windows & Dynamic Model Discovery:**
|
|
43
|
+
Context windows follow the model you select without mutating permanent settings. The gateway auto-discovers official models from Anthropic and OpenAI, refreshing mappings on tool version upgrades (`switch models`).
|
|
44
|
+
4. **Server Gateway Synergy (intact / 9Router):**
|
|
45
|
+
LLM Switcher pairs directly with server-side gateways like **intact** (strongly recommended for credential proxying, account pooling, and contract lab self-healing) or 9Router, handling retries and quotas far better than any local tool.
|
|
46
|
+
|
|
47
|
+
## 3. Configuring Token Compression Tools
|
|
48
|
+
|
|
49
|
+
When setting up or assisting the user with token compression tools, apply these exact configurations:
|
|
50
|
+
|
|
51
|
+
### A. Headroom (Prompt Compression Proxy)
|
|
52
|
+
When configuring Headroom's upstream endpoint:
|
|
53
|
+
- In Headroom config (`headroom.yaml` or environment variables):
|
|
54
|
+
```bash
|
|
55
|
+
ANTHROPIC_BASE_URL="http://127.0.0.1:3456"
|
|
56
|
+
OPENAI_BASE_URL="http://127.0.0.1:3456/v1"
|
|
57
|
+
```
|
|
58
|
+
- Point your CLI to Headroom's listening port (e.g. `http://127.0.0.1:8787`). Headroom compresses the prompt and forwards it to LLM Switcher on port `3456`.
|
|
59
|
+
|
|
60
|
+
### B. RTK (Rust Token Killer / Repomix)
|
|
61
|
+
- In RTK configuration or execution flags:
|
|
62
|
+
```bash
|
|
63
|
+
# Ensure RTK's target upstream proxy is LLM Switcher
|
|
64
|
+
export UPSTREAM_URL="http://127.0.0.1:3456"
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
### C. Ponytail / Custom Proxies
|
|
68
|
+
- Always set the proxy's upstream host to:
|
|
69
|
+
`http://127.0.0.1:3456` (or `http://127.0.0.1:3456/v1` for OpenAI-compatible tools).
|
|
70
|
+
|
|
71
|
+
### D. Sub-Agents & Spawned Terminal Commands
|
|
72
|
+
When an agent spawns sub-processes that call LLM CLIs:
|
|
73
|
+
- **Never advise sourcing `env.sh` or `env.cmd`.** Those files carry no proxy variables any
|
|
74
|
+
more, so sourcing one changes nothing — and a stale variable an older version once wrote
|
|
75
|
+
would still point the tool at a port where nothing listens.
|
|
76
|
+
- Run `switch claude <profile>` or `switch codex <profile>` instead. The shims in
|
|
77
|
+
`~/.llm-switcher/bin` inject the proxy variables into the tool process itself and clear
|
|
78
|
+
anything stale first.
|
|
79
|
+
- `claude` and `codex` on `PATH` already are those shims: invoke them normally, and never
|
|
80
|
+
wrap them in another script.
|
|
81
|
+
|
|
82
|
+
## 4. Operational Rules for AI Agents
|
|
83
|
+
|
|
84
|
+
1. **NEVER edit `~/.claude/settings.json` directly.**
|
|
85
|
+
LLM Switcher uses zero-mutation launcher flags (`active.flag`). Writing custom base URLs or auth tokens into `settings.json` triggers warning banners in Claude Code and risks breaking the user's official subscription. Use `switch` CLI commands instead.
|
|
86
|
+
2. **Check Gateway Health Before Complex Operations:**
|
|
87
|
+
Run `switch status` or call the `switcher_audit` MCP tool to confirm:
|
|
88
|
+
- LLM Switcher is active on port `3456`.
|
|
89
|
+
- The active profile matches the intended CLI target (Claude Code, Codex, or OpenAI).
|
|
90
|
+
3. **Verify Routing When Errors Occur:**
|
|
91
|
+
If a tool fails with `HTTP 400`, `HTTP 502`, or connection errors:
|
|
92
|
+
- Run `switch doctor` to audit port collisions and environment variables.
|
|
93
|
+
- Inspect recent request logs via `http://127.0.0.1:3456/ui` (Tab 4: Live Inspector) to see if an intermediary tool mangled the payload.
|