llm-switcher 1.1.11 → 1.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. package/CHANGELOG.md +51 -0
  2. package/README.md +202 -257
  3. package/README.vi.md +200 -256
  4. package/blindfold/blindfold.mjs +200 -53
  5. package/blindfold/make-certs.sh +26 -7
  6. package/catalog.mjs +246 -0
  7. package/config.example.json +12 -34
  8. package/docs/TOKEN-OPTIMIZER-INTEROP.md +110 -110
  9. package/docs/codex-blindfold.md +28 -17
  10. package/docs/cross-platform.md +16 -7
  11. package/docs/diagrams/ir-healer-pipeline.mmd +16 -0
  12. package/docs/diagrams/ir-healer-pipeline.png +0 -0
  13. package/docs/diagrams/ir-healer-pipeline.svg +90 -0
  14. package/docs/diagrams/ir-translation-pipeline.html +14925 -0
  15. package/docs/diagrams/ir-translation-pipeline.sequence.json +31 -0
  16. package/docs/diagrams/ir-translation-pipeline.svg +5128 -0
  17. package/docs/diagrams/system-architecture.architecture.json +76 -0
  18. package/docs/diagrams/system-architecture.html +14978 -0
  19. package/docs/diagrams/system-architecture.svg +5147 -0
  20. package/docs/diagrams/system-topology.mmd +30 -0
  21. package/docs/diagrams/system-topology.png +0 -0
  22. package/docs/diagrams/system-topology.svg +125 -0
  23. package/docs/response-matrix.json +1130 -1130
  24. package/ensure-ca-bundle.mjs +28 -0
  25. package/formats.mjs +13 -155
  26. package/mcp.mjs +39 -11
  27. package/package.json +1 -1
  28. package/proxy.mjs +92 -27
  29. package/shim.mjs +200 -57
  30. package/skills/llm-switcher/SKILL.md +93 -88
  31. package/state.mjs +1100 -191
  32. package/switch +0 -0
  33. package/switch.cmd +2 -2
  34. package/switch.mjs +228 -53
  35. package/tests/blindfold-e2e.test.mjs +380 -0
  36. package/tests/blindfold-task5.test.mjs +429 -0
  37. package/tests/blindfold.task3.test.mjs +700 -0
  38. package/tests/blindfold.test.mjs +10 -5
  39. package/tests/catalog.test.mjs +147 -0
  40. package/tests/contract-lab.test.mjs +22 -7
  41. package/tests/formats.test.mjs +33 -46
  42. package/tests/gateway.e2e.test.mjs +136 -36
  43. package/tests/helpers.mjs +24 -24
  44. package/tests/lifecycle.test.mjs +16 -10
  45. package/tests/live-optimizer-interop.mjs +205 -205
  46. package/tests/mcp.test.mjs +78 -2
  47. package/tests/real-user-sim.test.mjs +464 -0
  48. package/tests/shim.test.mjs +159 -66
  49. package/tests/state.test.mjs +975 -193
  50. package/tests/switch.test.mjs +446 -2
  51. package/ui.html +61 -154
package/shim.mjs CHANGED
@@ -26,33 +26,121 @@ import fs from 'node:fs';
26
26
  import path from 'node:path';
27
27
  import os from 'node:os';
28
28
  import { execFileSync } from 'node:child_process';
29
+ import { fileURLToPath } from 'node:url';
29
30
  import { paths, STATE_DIR } from './state.mjs';
30
31
 
31
32
  export const SHIM_DIR = path.join(os.homedir(), '.llm-switcher', 'bin');
32
33
 
33
- // Local model catalog: generated by applyLaunchState from the active profile's
34
- // publicModels, so it holds official OpenAI slugs only. It gives Codex the metadata
35
- // for the names it requests (otherwise the CLI falls back to full-context mode) and
36
- // it is what the /model picker renders — the picker never calls /v1/models.
37
- // Forward slashes for the TOML value. The Windows `if exist` test uses the native path instead.
38
- export const CODE_X_CATALOG_PATH = paths.codexCatalog.replace(/\\/g, '/');
34
+ // The shims hand the CA work to Node: reading files, comparing content and renaming atomically
35
+ // has no place in a shell script, and none at all in cmd.exe. It is a FILE rather than
36
+ // `node -e`, so neither cmd.exe quoting nor URL encoding can corrupt it: pathToFileURL turns a
37
+ // space in the checkout into %20, which cmd.exe then reads as the (undefined) argument %2.
38
+ const STATE_HELPER = path.join(path.dirname(fileURLToPath(import.meta.url)), 'ensure-ca-bundle.mjs');
39
39
 
40
40
  // CLIs to wrap. `claude` is the most important case (--resume), codex included for completeness.
41
41
  export const SHIMMED = ['claude', 'codex'];
42
42
 
43
- const POSIX_TEMPLATE = (name) => `#!/usr/bin/env bash
43
+ // One tool, one env file. Sourcing the shared one is how one tool used to capture the other.
44
+ const envFileOf = (name) => (name === 'codex' ? 'env-codex.sh' : 'env-claude.sh');
45
+
46
+ const POSIX_TEMPLATE = (name) => {
47
+ const envFile = envFileOf(name);
48
+ const caBlock = name === 'codex' ? '' : `
49
+ # R2: one bundle holds the user's own CA and the switcher CA, so trusting the interceptor never
50
+ # costs a certificate the user already had. Decided here, at launch, because only now is it known
51
+ # whether the user brought one — and never while this tool is off.
52
+ if [ "$TOOL_ACTIVE" = "1" ]; then
53
+ if [ -z "$INHERITED_CA" ]; then
54
+ if [ -f "$SWITCHER_DIR/blindfold/certs/ca.pem" ]; then
55
+ export NODE_EXTRA_CA_CERTS="$SWITCHER_DIR/blindfold/certs/ca.pem"
56
+ else
57
+ echo "[llm-switcher] $SWITCHER_DIR/blindfold/certs/ca.pem not found; NODE_EXTRA_CA_CERTS kept as inherited." >&2
58
+ fi
59
+ else
60
+ CA_BUNDLE="$(node "$STATE_HELPER" "$INHERITED_CA" "$SWITCHER_DIR/blindfold/certs/ca.pem" "$SWITCHER_DIR")" || CA_BUNDLE=""
61
+ if [ -n "$CA_BUNDLE" ] && [ -f "$CA_BUNDLE" ]; then
62
+ export NODE_EXTRA_CA_CERTS="$CA_BUNDLE"
63
+ fi
64
+ fi
65
+ fi`;
66
+ return `#!/usr/bin/env bash
44
67
  # Auto-generated by LLM Switcher — DO NOT EDIT.
45
- # Load the gateway env then run the real "${name}", even when the shell hasn't sourced env.sh
46
- # (e.g.: claude --resume reopening an old session in a clean terminal).
68
+ # Hand THIS tool its own gateway env, scrub what an older switcher left behind, then run the real
69
+ # "${name}", from any shell — including one that never sourced anything (claude --resume).
47
70
  SWITCHER_DIR="${STATE_DIR}"
71
+ STATE_HELPER="${STATE_HELPER}"
48
72
  SHIM_DIR="\${BASH_SOURCE%/*}"
49
73
 
50
- # Only load env when the gateway is on; when off, let the CLI run as-is. The shim runs what env.sh says,
51
- # so the directory and the file must belong to this account: another one could have created them.
52
- if [ -O "$SWITCHER_DIR" ] && [ -O "$SWITCHER_DIR/env.sh" ] && [ -f "$SWITCHER_DIR/active.flag" ]; then
53
- . "$SWITCHER_DIR/env.sh"
74
+ # Captured BEFORE the tool's env file is sourced: the CA decision must see the value this shell
75
+ # already had, not one this run is about to write.
76
+ INHERITED_CA="\${NODE_EXTRA_CA_CERTS:-}"
77
+
78
+ # Only this tool's own file, only while the gateway is on, and only files this account owns:
79
+ # another account could have created the directory or the file. Non-empty means the tool runs;
80
+ # an empty file means it is off, and then nothing is sourced at all.
81
+ if [ -O "$SWITCHER_DIR" ] && [ -O "$SWITCHER_DIR/${envFile}" ] && [ -f "$SWITCHER_DIR/active.flag" ] && [ -s "$SWITCHER_DIR/${envFile}" ]; then
82
+ . "$SWITCHER_DIR/${envFile}"
83
+ TOOL_ACTIVE=1
84
+ else
85
+ TOOL_ACTIVE=0
86
+ fi
87
+
88
+ # R8: a variable is removed ONLY when its value is one an older switcher wrote. The user's own
89
+ # proxy, base URL, model or LLM_SWITCHER_* setting is theirs and is kept.
90
+ GW_PORT="$( { [ -O "$SWITCHER_DIR/gateway.port" ] && tr -d '[:space:]' < "$SWITCHER_DIR/gateway.port"; } 2>/dev/null )"
91
+ case "$GW_PORT" in ''|*[!0-9]*) GW_PORT=3456 ;; esac
92
+ PORTS="$GW_PORT 3456"
93
+ [ -n "$LLM_SWITCHER_PORT" ] && PORTS="$PORTS $LLM_SWITCHER_PORT"
94
+
95
+ # Every loopback shape the switcher ever wrote: host, optional /v1, optional trailing slash.
96
+ SWITCHER_URLS=""
97
+ for _p in $PORTS; do
98
+ for _h in 127.0.0.1 localhost '[::1]'; do
99
+ SWITCHER_URLS="$SWITCHER_URLS $_h:$_p $_h:$_p/ $_h:$_p/v1 $_h:$_p/v1/"
100
+ done
101
+ done
102
+
103
+ _u="\${ANTHROPIC_BASE_URL:-}"
104
+ if [ -n "$_u" ]; then
105
+ case " $SWITCHER_URLS " in
106
+ *" \${_u#http://} "*) unset ANTHROPIC_BASE_URL ;;
107
+ esac
108
+ fi
109
+ _u="\${OPENAI_BASE_URL:-}"
110
+ if [ -n "$_u" ]; then
111
+ case " $SWITCHER_URLS " in
112
+ *" \${_u#http://} "*) unset OPENAI_BASE_URL ;;
113
+ esac
54
114
  fi
55
115
 
116
+ # The suffix the old switcher forced on every tier. Nothing else matches: the user's own
117
+ # ANTHROPIC_MODEL=opus, or a real model id, is theirs.
118
+ case "\${ANTHROPIC_MODEL:-}" in
119
+ opus\\[1m\\]|sonnet\\[1m\\]|haiku\\[1m\\]|fable\\[1m\\]) unset ANTHROPIC_MODEL ;;
120
+ esac
121
+ tier_is_stale() {
122
+ [ -n "$1" ] || return 1
123
+ case "$(printf '%s' "$1" | tr '[:upper:]' '[:lower:]')" in
124
+ "$2"\\[1m\\]) return 0 ;;
125
+ *) return 1 ;;
126
+ esac
127
+ }
128
+ if tier_is_stale "$ANTHROPIC_DEFAULT_OPUS_MODEL" opus; then unset ANTHROPIC_DEFAULT_OPUS_MODEL; fi
129
+ if tier_is_stale "$ANTHROPIC_DEFAULT_SONNET_MODEL" sonnet; then unset ANTHROPIC_DEFAULT_SONNET_MODEL; fi
130
+ if tier_is_stale "$ANTHROPIC_DEFAULT_HAIKU_MODEL" haiku; then unset ANTHROPIC_DEFAULT_HAIKU_MODEL; fi
131
+ if tier_is_stale "$ANTHROPIC_DEFAULT_FABLE_MODEL" fable; then unset ANTHROPIC_DEFAULT_FABLE_MODEL; fi
132
+
133
+ [ "\${CLAUDE_CODE_DISABLE_UNKNOWN_MODEL_WINDOW_ENFORCEMENT:-}" = "1" ] && unset CLAUDE_CODE_DISABLE_UNKNOWN_MODEL_WINDOW_ENFORCEMENT
134
+ [ "\${CLAUDE_CODE_AUTO_COMPACT_WINDOW:-}" = "900000" ] && unset CLAUDE_CODE_AUTO_COMPACT_WINDOW
135
+ [ "\${OPENAI_MAX_CONTEXT_TOKENS:-}" = "1000000" ] && unset OPENAI_MAX_CONTEXT_TOKENS
136
+
137
+ # LLM_SWITCHER_CODEX_* was always a shim input, never the user's. Every OTHER LLM_SWITCHER_*
138
+ # name is a user setting and is kept.
139
+ for _v in $(compgen -v LLM_SWITCHER_CODEX_ 2>/dev/null); do
140
+ unset "$_v"
141
+ done
142
+ ${caBlock}
143
+
56
144
  # Find the real binary: drop the shim directory from PATH so it doesn't call itself.
57
145
  CLEAN_PATH=""
58
146
  IFS=':'
@@ -72,65 +160,120 @@ if [ -z "$REAL" ]; then
72
160
  exit 127
73
161
  fi
74
162
 
75
- ${name === 'codex' ? `# Codex-only variables (blindfold proxy + CA). The claude shim must not load these.
76
- if [ -O "$SWITCHER_DIR" ] && [ -O "$SWITCHER_DIR/env-codex.sh" ] && [ -f "$SWITCHER_DIR/active.flag" ]; then
77
- . "$SWITCHER_DIR/env-codex.sh"
78
- fi
79
-
80
- CODEX_SWITCHER_ARGS=()
81
- if [ -n "\${LLM_SWITCHER_CODEX_BASE_URL:-}" ]; then
82
- CODEX_SWITCHER_ARGS+=(--config "openai_base_url=\${LLM_SWITCHER_CODEX_BASE_URL}")
83
- fi
84
- if [ -O "${CODE_X_CATALOG_PATH}" ]; then
85
- CODEX_SWITCHER_ARGS+=(--config "model_catalog_json=${CODE_X_CATALOG_PATH}")
86
- fi
87
- if [ -n "\${LLM_SWITCHER_CODEX_MAIN_MODEL:-}" ]; then
88
- CODEX_SWITCHER_ARGS+=(--config "model=\\"\${LLM_SWITCHER_CODEX_MAIN_MODEL}\\"")
89
- fi
90
- if [ -n "\${LLM_SWITCHER_CODEX_REVIEW_MODEL:-}" ]; then
91
- CODEX_SWITCHER_ARGS+=(--config "review_model=\\"\${LLM_SWITCHER_CODEX_REVIEW_MODEL}\\"")
92
- fi
93
- if [ -n "\${LLM_SWITCHER_CODEX_SUBAGENT_MODEL:-}" ]; then
94
- CODEX_SWITCHER_ARGS+=(--config "agents.default_subagent_model=\\"\${LLM_SWITCHER_CODEX_SUBAGENT_MODEL}\\"")
95
- fi
96
- if [ -n "\${LLM_SWITCHER_CODEX_CONTEXT_WINDOW:-}" ]; then
97
- CODEX_SWITCHER_ARGS+=(--config "model_context_window=\${LLM_SWITCHER_CODEX_CONTEXT_WINDOW}")
98
- fi
99
- if [ -n "\${LLM_SWITCHER_CODEX_AUTO_COMPACT_LIMIT:-}" ]; then
100
- CODEX_SWITCHER_ARGS+=(--config "model_auto_compact_token_limit=\${LLM_SWITCHER_CODEX_AUTO_COMPACT_LIMIT}")
101
- fi
102
-
103
- exec "$REAL" "\${CODEX_SWITCHER_ARGS[@]}" "$@"` : `exec "$REAL" "$@"`}
163
+ # No override of any kind: the tool keeps its own configuration and its own model names (F3, R1).
164
+ exec "$REAL" "$@"
104
165
  `;
166
+ };
105
167
 
106
- const WINDOWS_TEMPLATE = (name) => `@echo off
168
+ const WINDOWS_TEMPLATE = (name) => {
169
+ const envFile = envFileOf(name).replace(/\.sh$/, '.cmd');
170
+ const caBlock = name === 'codex' ? '' : `
171
+ REM R2: one bundle holds the user's own CA and the switcher CA, so trusting the interceptor
172
+ REM never costs a certificate the user already had. Decided here, at launch, because only now
173
+ REM is it known whether the user brought one - and never while this tool is off.
174
+ if not "%TOOL_ACTIVE%"=="1" goto :CA_DONE
175
+ if defined INHERITED_CA goto :CA_BUNDLE
176
+ if exist "%SWITCHER_DIR%\\blindfold\\certs\\ca.pem" set "NODE_EXTRA_CA_CERTS=%SWITCHER_DIR%\\blindfold\\certs\\ca.pem"
177
+ if not exist "%SWITCHER_DIR%\\blindfold\\certs\\ca.pem" echo [llm-switcher] %SWITCHER_DIR%\\blindfold\\certs\\ca.pem not found; NODE_EXTRA_CA_CERTS kept as inherited. 1>&2
178
+ goto :CA_DONE
179
+ :CA_BUNDLE
180
+ set "CA_BUNDLE="
181
+ for /f "delims=" %%B in ('node "%STATE_HELPER%" "%INHERITED_CA%" "%SWITCHER_DIR%\\blindfold\\certs\\ca.pem" "%SWITCHER_DIR%" 2^>nul') do set "CA_BUNDLE=%%B"
182
+ if not defined CA_BUNDLE goto :CA_DONE
183
+ if not exist "%CA_BUNDLE%" goto :CA_DONE
184
+ set "NODE_EXTRA_CA_CERTS=%CA_BUNDLE%"
185
+ :CA_DONE`;
186
+ return `@echo off
107
187
  REM Auto-generated by LLM Switcher - DO NOT EDIT.
188
+ setlocal
108
189
  set "SWITCHER_DIR=${STATE_DIR}"
109
- if exist "%SWITCHER_DIR%\\active.flag" if exist "%SWITCHER_DIR%\\env.cmd" call "%SWITCHER_DIR%\\env.cmd"
110
- ${name === 'codex' ? `REM Codex-only variables (blindfold proxy + CA). The claude shim must not load these.
111
- if exist "%SWITCHER_DIR%\\active.flag" if exist "%SWITCHER_DIR%\\env-codex.cmd" call "%SWITCHER_DIR%\\env-codex.cmd"
112
- set "CODEX_SWITCHER_ARGS="
113
- if defined LLM_SWITCHER_CODEX_BASE_URL set "CODEX_SWITCHER_ARGS=%CODEX_SWITCHER_ARGS% --config openai_base_url=%LLM_SWITCHER_CODEX_BASE_URL%"
114
- if exist "%SWITCHER_DIR%\\model-catalog.json" set "CODEX_SWITCHER_ARGS=%CODEX_SWITCHER_ARGS% --config model_catalog_json=${CODE_X_CATALOG_PATH}"
115
- if defined LLM_SWITCHER_CODEX_MAIN_MODEL set "CODEX_SWITCHER_ARGS=%CODEX_SWITCHER_ARGS% --config model=%LLM_SWITCHER_CODEX_MAIN_MODEL%"
116
- if defined LLM_SWITCHER_CODEX_REVIEW_MODEL set "CODEX_SWITCHER_ARGS=%CODEX_SWITCHER_ARGS% --config review_model=%LLM_SWITCHER_CODEX_REVIEW_MODEL%"
117
- if defined LLM_SWITCHER_CODEX_SUBAGENT_MODEL set "CODEX_SWITCHER_ARGS=%CODEX_SWITCHER_ARGS% --config agents.default_subagent_model=%LLM_SWITCHER_CODEX_SUBAGENT_MODEL%"
118
- if defined LLM_SWITCHER_CODEX_CONTEXT_WINDOW set "CODEX_SWITCHER_ARGS=%CODEX_SWITCHER_ARGS% --config model_context_window=%LLM_SWITCHER_CODEX_CONTEXT_WINDOW%"
119
- if defined LLM_SWITCHER_CODEX_AUTO_COMPACT_LIMIT set "CODEX_SWITCHER_ARGS=%CODEX_SWITCHER_ARGS% --config model_auto_compact_token_limit=%LLM_SWITCHER_CODEX_AUTO_COMPACT_LIMIT%"` : ''}
190
+ ${name === 'codex' ? '' : `set "STATE_HELPER=${STATE_HELPER}"
191
+
192
+ REM Captured BEFORE the tool's env file runs: the CA decision must see the value this shell
193
+ REM already had, not one this run is about to write.
194
+ set "INHERITED_CA=%NODE_EXTRA_CA_CERTS%"
195
+ `}
196
+
197
+ REM Only this tool's own file, only while the gateway is on, and only a non-empty one: an empty
198
+ REM file means this tool is off. The shared env.cmd is never called (F5).
199
+ set "TOOL_ACTIVE=0"
200
+ if exist "%SWITCHER_DIR%\\active.flag" if exist "%SWITCHER_DIR%\\${envFile}" (
201
+ for %%A in ("%SWITCHER_DIR%\\${envFile}") do if %%~zA GTR 0 set "TOOL_ACTIVE=1"
202
+ )
203
+ if "%TOOL_ACTIVE%"=="1" call "%SWITCHER_DIR%\\${envFile}"
204
+
205
+ REM R8: a variable is removed ONLY when its value is one an older switcher wrote. The user's own
206
+ REM proxy, base URL, model or LLM_SWITCHER_* setting is theirs and is kept.
207
+ set "_GWPORT="
208
+ if exist "%SWITCHER_DIR%\\gateway.port" for /f "usebackq delims=" %%P in ("%SWITCHER_DIR%\\gateway.port") do set "_GWPORT=%%P"
209
+ if not defined _GWPORT set "_GWPORT=3456"
210
+ if defined ANTHROPIC_BASE_URL call :SCRUB_URL ANTHROPIC_BASE_URL "%ANTHROPIC_BASE_URL%"
211
+ if defined OPENAI_BASE_URL call :SCRUB_URL OPENAI_BASE_URL "%OPENAI_BASE_URL%"
212
+ if /i "%ANTHROPIC_MODEL%"=="opus[1m]" set "ANTHROPIC_MODEL="
213
+ if /i "%ANTHROPIC_MODEL%"=="sonnet[1m]" set "ANTHROPIC_MODEL="
214
+ if /i "%ANTHROPIC_MODEL%"=="haiku[1m]" set "ANTHROPIC_MODEL="
215
+ if /i "%ANTHROPIC_MODEL%"=="fable[1m]" set "ANTHROPIC_MODEL="
216
+ if defined ANTHROPIC_DEFAULT_OPUS_MODEL call :SCRUB_TIER ANTHROPIC_DEFAULT_OPUS_MODEL "%ANTHROPIC_DEFAULT_OPUS_MODEL%" opus
217
+ if defined ANTHROPIC_DEFAULT_SONNET_MODEL call :SCRUB_TIER ANTHROPIC_DEFAULT_SONNET_MODEL "%ANTHROPIC_DEFAULT_SONNET_MODEL%" sonnet
218
+ if defined ANTHROPIC_DEFAULT_HAIKU_MODEL call :SCRUB_TIER ANTHROPIC_DEFAULT_HAIKU_MODEL "%ANTHROPIC_DEFAULT_HAIKU_MODEL%" haiku
219
+ if defined ANTHROPIC_DEFAULT_FABLE_MODEL call :SCRUB_TIER ANTHROPIC_DEFAULT_FABLE_MODEL "%ANTHROPIC_DEFAULT_FABLE_MODEL%" fable
220
+ if "%CLAUDE_CODE_DISABLE_UNKNOWN_MODEL_WINDOW_ENFORCEMENT%"=="1" set "CLAUDE_CODE_DISABLE_UNKNOWN_MODEL_WINDOW_ENFORCEMENT="
221
+ if "%CLAUDE_CODE_AUTO_COMPACT_WINDOW%"=="900000" set "CLAUDE_CODE_AUTO_COMPACT_WINDOW="
222
+ if "%OPENAI_MAX_CONTEXT_TOKENS%"=="1000000" set "OPENAI_MAX_CONTEXT_TOKENS="
223
+ for /f "tokens=1 delims==" %%V in ('set LLM_SWITCHER_CODEX_ 2^>nul ^| findstr /b /c:"LLM_SWITCHER_CODEX_"') do set "%%V="
224
+ ${caBlock}
225
+
226
+ REM Find the real binary: drop this shim's own directory so it never calls itself.
120
227
  for /f "delims=" %%i in ('where ${name}.cmd 2^>nul ^| findstr /v /i "\\.llm-switcher\\\\bin"') do (
121
- call "%%i" ${name === 'codex' ? '%CODEX_SWITCHER_ARGS% ' : ''}%*
228
+ call "%%i" %*
122
229
  exit /b
123
230
  )
124
231
  for /f "delims=" %%i in ('where ${name}.exe 2^>nul ^| findstr /v /i "\\.llm-switcher\\\\bin"') do (
125
- call "%%i" ${name === 'codex' ? '%CODEX_SWITCHER_ARGS% ' : ''}%*
232
+ call "%%i" %*
126
233
  exit /b
127
234
  )
128
235
  echo [llm-switcher] cannot find the real '${name}' on PATH.>&2
129
236
  exit /b 127
237
+
238
+ REM ---- R8 subroutines ----------------------------------------------------
239
+
240
+ REM %1 = variable name, %2 = its value. Unsets %1 only for a loopback gateway URL on a port this
241
+ REM switcher has used: the recorded one, the default 3456, or LLM_SWITCHER_PORT.
242
+ :SCRUB_URL
243
+ set "_T=%~2"
244
+ if /i "%_T:~0,7%"=="http://" set "_T=%_T:~7%"
245
+ set "_R="
246
+ if /i "%_T:~0,10%"=="127.0.0.1:" set "_R=%_T:~10%"
247
+ if /i "%_T:~0,10%"=="localhost:" set "_R=%_T:~10%"
248
+ if /i "%_T:~0,6%"=="[::1]:" set "_R=%_T:~6%"
249
+ if not defined _R goto :eof
250
+ if /i "%_R:~-4%"=="/v1/" set "_R=%_R:~0,-4%"
251
+ if /i "%_R:~-3%"=="/v1" set "_R=%_R:~0,-3%"
252
+ if "%_R:~-1%"=="/" set "_R=%_R:~0,-1%"
253
+ if not defined _R goto :eof
254
+ if not "%_R%"=="" for /f "delims=0123456789" %%X in ("%_R%") do goto :eof
255
+ if "%_R%"=="%_GWPORT%" goto :DROP_URL
256
+ if "%_R%"=="3456" goto :DROP_URL
257
+ if defined LLM_SWITCHER_PORT if "%_R%"=="%LLM_SWITCHER_PORT%" goto :DROP_URL
258
+ goto :eof
259
+ :DROP_URL
260
+ set "%~1="
261
+ goto :eof
262
+
263
+ REM %1 = variable name, %2 = its value, %3 = the lowercase tier it must still match. A value of
264
+ REM another tier (ANTHROPIC_DEFAULT_OPUS_MODEL=sonnet[1m]) is not the switcher's and is kept.
265
+ :SCRUB_TIER
266
+ if /i "%~2"=="%~3[1m]" set "%~1="
267
+ goto :eof
130
268
  `;
269
+ };
131
270
 
132
271
  export function renderShim(name, platform = process.platform) {
133
- return platform === 'win32' ? WINDOWS_TEMPLATE(name) : POSIX_TEMPLATE(name);
272
+ if (platform !== 'win32') return POSIX_TEMPLATE(name);
273
+ // cmd.exe finds a subroutine by scanning line starts, and on a file written with bare LF
274
+ // it drifts: a label near the end is reported as "cannot find the batch label". CRLF is
275
+ // what every .cmd on disk already is, and it is what cmd.exe expects.
276
+ return WINDOWS_TEMPLATE(name).replace(/\r?\n/g, '\r\n');
134
277
  }
135
278
 
136
279
  function shimPath(name) {
@@ -1,88 +1,93 @@
1
- ---
2
- name: llm-switcher
3
- description: Mandatory guidance for AI coding agents and token compression tools (Headroom, RTK, Ponytail, Repomix). Enforces routing all LLM traffic through the local LLM Switcher edge gateway (http://127.0.0.1:3456) instead of directly calling provider endpoints.
4
- ---
5
-
6
- # LLM Switcher Edge Gateway Skill
7
-
8
- This skill governs how AI coding agents (Claude Code, Codex, Cursor, Windsurf, Opencode) and prompt optimization/compression tools (Headroom, RTK, Ponytail, Repomix) must interact with LLM providers on this workstation.
9
-
10
- ## 1. The Core Directive
11
-
12
- > **ALL OUTBOUND LLM REQUESTS MUST ROUTE THROUGH LLM SWITCHER (`http://127.0.0.1:3456`).**
13
- >
14
- > Never configure any tool, script, sub-agent, or token compression proxy to call external provider endpoints directly (such as `https://api.anthropic.com`, `https://api.openai.com`, or `https://generativelanguage.googleapis.com`).
15
- >
16
- > If a token compression tool (Headroom, RTK, Ponytail) is used, it MUST be positioned as an **inner processing layer**, and its **upstream destination MUST be configured to point to LLM Switcher** at `http://127.0.0.1:3456`.
17
-
18
- ```
19
- [Client CLI / Agent]
20
- │
21
- ▼
22
- [Token Compressor: RTK / Headroom / Ponytail] <-- Inner Layer
23
- │
24
- ▼ (MANDATORY: Upstream MUST point to http://127.0.0.1:3456)
25
- [LLM Switcher Edge Gateway (:3456)] <-- Outermost Gatekeeper
26
- │
27
- ▼ (Heals schemas, maps 1M context, extracts thinking, converts protocol)
28
- [Internet / 9Router / Upstream LLM]
29
- ```
30
-
31
- ## 2. Why This is Mandatory
32
-
33
- Direct outbound calls from token compressors will break the developer environment for several reasons:
34
-
35
- 1. **Schema Breakage & The Healer Engine:**
36
- Token optimizers frequently prune conversation history to save tokens. In doing so, they often delete an `assistant` turn containing a `tool_use`, while leaving the subsequent `user` turn containing a `tool_result`.
37
- - Anthropic and OpenAI APIs strictly reject this with `HTTP 400 (orphaned tool_use_id)`.
38
- - **LLM Switcher** intercepts these malformed requests and runs its **Healer Engine**, automatically converting orphaned results into safe context text blocks and merging consecutive same-role turns so upstream providers accept the payload cleanly.
39
- 2. **Thinking / Reasoning Protection:**
40
- Many token compressors naively strip `thinking` parameters or reasoning budgets to reduce token count.
41
- - When routed through LLM Switcher, the gateway detects reasoning models (such as `ag/claude-opus-4-6-thinking` or `ag/gemini-3.8-flash`) and **automatically restores thinking parameters**, ensuring the model retains its full reasoning capabilities.
42
- 3. **1M Context Window Unlocking:**
43
- LLM Switcher dynamically injects `ANTHROPIC_DEFAULT_<TIER>_MODEL=<tier>[1m]` for every tier the profile marks 1M and calculates auto-compact thresholds (`900,000` tokens) without mutating permanent settings.
44
- 4. **Centralized Gateway Connection (9Router):**
45
- LLM Switcher connects directly to server-side gateways like **9Router**, which manage multi-account pooling, quota tracking, load balancing, and rate-limit retries far better than any local tool.
46
-
47
- ## 3. Configuring Token Compression Tools
48
-
49
- When setting up or assisting the user with token compression tools, apply these exact configurations:
50
-
51
- ### A. Headroom (Prompt Compression Proxy)
52
- When configuring Headroom's upstream endpoint:
53
- - In Headroom config (`headroom.yaml` or environment variables):
54
- ```bash
55
- ANTHROPIC_BASE_URL="http://127.0.0.1:3456"
56
- OPENAI_BASE_URL="http://127.0.0.1:3456/v1"
57
- ```
58
- - Point your CLI to Headroom's listening port (e.g. `http://127.0.0.1:8787`). Headroom compresses the prompt and forwards it to LLM Switcher on port `3456`.
59
-
60
- ### B. RTK (Rust Token Killer / Repomix)
61
- - In RTK configuration or execution flags:
62
- ```bash
63
- # Ensure RTK's target upstream proxy is LLM Switcher
64
- export UPSTREAM_URL="http://127.0.0.1:3456"
65
- ```
66
-
67
- ### C. Ponytail / Custom Proxies
68
- - Always set the proxy's upstream host to:
69
- `http://127.0.0.1:3456` (or `http://127.0.0.1:3456/v1` for OpenAI-compatible tools).
70
-
71
- ### D. Sub-Agents & Spawned Terminal Commands
72
- When an agent spawns sub-processes via `bash` or terminal execution that call LLM CLIs:
73
- - **On Windows:** Always run `call "path\to\llm-switcher\env.cmd"` first.
74
- - **On macOS / Linux:** Always run `source "path/to/llm-switcher/env.sh"` first.
75
- - This prevents sub-agents from leaking calls directly to expensive official APIs.
76
-
77
- ## 4. Operational Rules for AI Agents
78
-
79
- 1. **NEVER edit `~/.claude/settings.json` directly.**
80
- LLM Switcher uses zero-mutation launcher flags (`active.flag`, `1m.flag`). Writing custom base URLs or auth tokens into `settings.json` triggers warning banners in Claude Code and risks breaking the user's official subscription. Use `switch` CLI commands instead.
81
- 2. **Check Gateway Health Before Complex Operations:**
82
- Run `switch status` or call the `switcher_audit` MCP tool to confirm:
83
- - LLM Switcher is active on port `3456`.
84
- - The active profile matches the intended CLI target (Claude Code, Codex, or OpenAI).
85
- 3. **Verify Routing When Errors Occur:**
86
- If a tool fails with `HTTP 400`, `HTTP 502`, or connection errors:
87
- - Run `switch doctor` to audit port collisions and environment variables.
88
- - Inspect recent request logs via `http://127.0.0.1:3456/ui` (Tab 4: Live Inspector) to see if an intermediary tool mangled the payload.
1
+ ---
2
+ name: llm-switcher
3
+ description: Mandatory guidance for AI coding agents and token compression tools (Headroom, RTK, Ponytail, Repomix). Enforces routing all LLM traffic through the local LLM Switcher edge gateway (http://127.0.0.1:3456) instead of directly calling provider endpoints.
4
+ ---
5
+
6
+ # LLM Switcher Edge Gateway Skill
7
+
8
+ This skill governs how AI coding agents (Claude Code, Codex, Cursor, Windsurf, Opencode) and prompt optimization/compression tools (Headroom, RTK, Ponytail, Repomix) must interact with LLM providers on this workstation.
9
+
10
+ ## 1. The Core Directive
11
+
12
+ > **ALL OUTBOUND LLM REQUESTS MUST ROUTE THROUGH LLM SWITCHER (`http://127.0.0.1:3456`).**
13
+ >
14
+ > Never configure any tool, script, sub-agent, or token compression proxy to call external provider endpoints directly (such as `https://api.anthropic.com`, `https://api.openai.com`, or `https://generativelanguage.googleapis.com`).
15
+ >
16
+ > If a token compression tool (Headroom, RTK, Ponytail) is used, it MUST be positioned as an **inner processing layer**, and its **upstream destination MUST be configured to point to LLM Switcher** at `http://127.0.0.1:3456`.
17
+
18
+ ```
19
+ [Client CLI / Agent]
20
+ │
21
+ ▼
22
+ [Token Compressor: RTK / Headroom / Ponytail] <-- Inner Layer
23
+ │
24
+ ▼ (MANDATORY: Upstream MUST point to http://127.0.0.1:3456)
25
+ [LLM Switcher Edge Gateway (:3456)] <-- Outermost Gatekeeper
26
+ │
27
+ ▼ (Heals schemas, maps 1M context, extracts thinking, converts protocol)
28
+ [Internet / intact (Recommended) / 9Router / Upstream LLM]
29
+ ```
30
+
31
+ ## 2. Why This is Mandatory
32
+
33
+ Direct outbound calls from token compressors will break the developer environment for several reasons:
34
+
35
+ 1. **Schema Breakage & The Healer Engine:**
36
+ Token optimizers frequently prune conversation history to save tokens. In doing so, they often delete an `assistant` turn containing a `tool_use`, while leaving the subsequent `user` turn containing a `tool_result`.
37
+ - Anthropic and OpenAI APIs strictly reject this with `HTTP 400 (orphaned tool_use_id)`.
38
+ - **LLM Switcher** intercepts these malformed requests and runs its **Healer Engine**, automatically converting orphaned results into safe context text blocks and merging consecutive same-role turns so upstream providers accept the payload cleanly.
39
+ 2. **Thinking / Reasoning Protection:**
40
+ Many token compressors naively strip `thinking` parameters or reasoning budgets to reduce token count.
41
+ - When routed through LLM Switcher, the gateway detects reasoning models (such as `ag/claude-opus-4-6-thinking` or `ag/gemini-3.8-flash`) and **automatically restores thinking parameters**, ensuring the model retains its full reasoning capabilities.
42
+ 3. **Official Context Windows & Dynamic Model Discovery:**
43
+ Context windows follow the model you select without mutating permanent settings. The gateway auto-discovers official models from Anthropic and OpenAI, refreshing mappings on tool version upgrades (`switch models`).
44
+ 4. **Server Gateway Synergy (intact / 9Router):**
45
+ LLM Switcher pairs directly with server-side gateways like **intact** (strongly recommended for credential proxying, account pooling, and contract lab self-healing) or 9Router, handling retries and quotas far better than any local tool.
46
+
47
+ ## 3. Configuring Token Compression Tools
48
+
49
+ When setting up or assisting the user with token compression tools, apply these exact configurations:
50
+
51
+ ### A. Headroom (Prompt Compression Proxy)
52
+ When configuring Headroom's upstream endpoint:
53
+ - In Headroom config (`headroom.yaml` or environment variables):
54
+ ```bash
55
+ ANTHROPIC_BASE_URL="http://127.0.0.1:3456"
56
+ OPENAI_BASE_URL="http://127.0.0.1:3456/v1"
57
+ ```
58
+ - Point your CLI to Headroom's listening port (e.g. `http://127.0.0.1:8787`). Headroom compresses the prompt and forwards it to LLM Switcher on port `3456`.
59
+
60
+ ### B. RTK (Rust Token Killer / Repomix)
61
+ - In RTK configuration or execution flags:
62
+ ```bash
63
+ # Ensure RTK's target upstream proxy is LLM Switcher
64
+ export UPSTREAM_URL="http://127.0.0.1:3456"
65
+ ```
66
+
67
+ ### C. Ponytail / Custom Proxies
68
+ - Always set the proxy's upstream host to:
69
+ `http://127.0.0.1:3456` (or `http://127.0.0.1:3456/v1` for OpenAI-compatible tools).
70
+
71
+ ### D. Sub-Agents & Spawned Terminal Commands
72
+ When an agent spawns sub-processes that call LLM CLIs:
73
+ - **Never advise sourcing `env.sh` or `env.cmd`.** Those files carry no proxy variables any
74
+ more, so sourcing one changes nothing — and a stale variable an older version once wrote
75
+ would still point the tool at a port where nothing listens.
76
+ - Run `switch claude <profile>` or `switch codex <profile>` instead. The shims in
77
+ `~/.llm-switcher/bin` inject the proxy variables into the tool process itself and clear
78
+ anything stale first.
79
+ - `claude` and `codex` on `PATH` already are those shims: invoke them normally, and never
80
+ wrap them in another script.
81
+
82
+ ## 4. Operational Rules for AI Agents
83
+
84
+ 1. **NEVER edit `~/.claude/settings.json` directly.**
85
+ LLM Switcher uses zero-mutation launcher flags (`active.flag`). Writing custom base URLs or auth tokens into `settings.json` triggers warning banners in Claude Code and risks breaking the user's official subscription. Use `switch` CLI commands instead.
86
+ 2. **Check Gateway Health Before Complex Operations:**
87
+ Run `switch status` or call the `switcher_audit` MCP tool to confirm:
88
+ - LLM Switcher is active on port `3456`.
89
+ - The active profile matches the intended CLI target (Claude Code, Codex, or OpenAI).
90
+ 3. **Verify Routing When Errors Occur:**
91
+ If a tool fails with `HTTP 400`, `HTTP 502`, or connection errors:
92
+ - Run `switch doctor` to audit port collisions and environment variables.
93
+ - Inspect recent request logs via `http://127.0.0.1:3456/ui` (Tab 4: Live Inspector) to see if an intermediary tool mangled the payload.