llm-switcher 1.2.8 → 1.2.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +5 -1
- package/README.md +13 -0
- package/README.vi.md +13 -0
- package/docs/TOKEN-OPTIMIZER-INTEROP.md +110 -110
- package/docs/response-matrix.json +1130 -1130
- package/package.json +1 -1
- package/proxy.mjs +76 -7
- package/skills/llm-switcher/SKILL.md +93 -93
- package/switch +0 -0
- package/switch.cmd +2 -2
- package/tests/contract-lab.test.mjs +5 -0
- package/tests/gateway.e2e.test.mjs +50 -1
- package/tests/helpers.mjs +24 -24
- package/tests/live-optimizer-interop.mjs +205 -205
- package/ui.html +1 -1
- package/.impeccable/hook.cache.json +0 -1
package/package.json
CHANGED
package/proxy.mjs
CHANGED
|
@@ -500,7 +500,54 @@ function previewOf(ir) {
|
|
|
500
500
|
// ----------------------------------------------------
|
|
501
501
|
const HOP_BY_HOP = new Set(['content-length', 'content-encoding', 'transfer-encoding', 'connection', 'keep-alive']);
|
|
502
502
|
|
|
503
|
-
|
|
503
|
+
// Bifrost: intact names, per model, the User-Agent prefix of the provider's own client (bifrost_ua).
|
|
504
|
+
// When the caller is that client, the request crosses as it is: only the key changes.
|
|
505
|
+
const bifrostUACache = new Map();
|
|
506
|
+
async function bifrostUAFor(profile, model) {
|
|
507
|
+
const base = String(profile.baseURL || '').replace(/\/+$/, '');
|
|
508
|
+
if (!base || !model) return '';
|
|
509
|
+
const key = `${base}\n${model}`;
|
|
510
|
+
const hit = bifrostUACache.get(key);
|
|
511
|
+
if (hit && hit.until > Date.now()) return hit.ua;
|
|
512
|
+
let ua = '';
|
|
513
|
+
let ttl = 10 * 60 * 1000;
|
|
514
|
+
try {
|
|
515
|
+
const r = await fetch(`${base}/models/${model.split('/').map(encodeURIComponent).join('/')}`, {
|
|
516
|
+
headers: { authorization: `Bearer ${profile.apiKey || ''}` }, signal: AbortSignal.timeout(3000)
|
|
517
|
+
});
|
|
518
|
+
if (r.ok) ua = String((await r.json())?.bifrost_ua || '');
|
|
519
|
+
} catch {
|
|
520
|
+
ttl = 30 * 1000; // an unreachable list must not pin the decision for long
|
|
521
|
+
}
|
|
522
|
+
bifrostUACache.set(key, { ua, until: Date.now() + ttl });
|
|
523
|
+
return ua;
|
|
524
|
+
}
|
|
525
|
+
|
|
526
|
+
async function crossesBifrost(req, profile, model) {
|
|
527
|
+
const clientUA = String(req.headers['user-agent'] || '');
|
|
528
|
+
if (!clientUA) return false;
|
|
529
|
+
const ua = await bifrostUAFor(profile, model);
|
|
530
|
+
return Boolean(ua) && clientUA.startsWith(ua);
|
|
531
|
+
}
|
|
532
|
+
|
|
533
|
+
// Every client header except its own credentials, the switcher's control headers and hop-by-hop.
|
|
534
|
+
function bifrostHeaders(req, profile) {
|
|
535
|
+
const headers = {};
|
|
536
|
+
for (const [k, v] of Object.entries(req.headers)) {
|
|
537
|
+
const lk = k.toLowerCase();
|
|
538
|
+
if (BLOCKED_PASSTHROUGH.has(lk) || HOP_BY_HOP.has(lk) || lk === 'host' || lk === 'authorization' || lk === 'accept-encoding') continue;
|
|
539
|
+
headers[lk] = v;
|
|
540
|
+
}
|
|
541
|
+
headers['x-api-key'] = profile.apiKey || '';
|
|
542
|
+
return headers;
|
|
543
|
+
}
|
|
544
|
+
|
|
545
|
+
function withClientQuery(url, req) {
|
|
546
|
+
const q = String(req.url || '').indexOf('?');
|
|
547
|
+
return q < 0 ? url : url + (url.includes('?') ? '&' : '?') + req.url.slice(q + 1);
|
|
548
|
+
}
|
|
549
|
+
|
|
550
|
+
async function forwardAnthropicDirect(res, payload, bodyBuffer, url, headers, mappedModel, signal, profile, bifrost = false) {
|
|
504
551
|
debugLog('Direct forward to native Anthropic endpoint:', url);
|
|
505
552
|
|
|
506
553
|
// Only re-serialize when a fix is actually needed; otherwise forward the client's original bytes.
|
|
@@ -510,13 +557,13 @@ async function forwardAnthropicDirect(res, payload, bodyBuffer, url, headers, ma
|
|
|
510
557
|
json = { ...json, model: mappedModel };
|
|
511
558
|
modified = true;
|
|
512
559
|
}
|
|
513
|
-
const healed = healAnthropicPayload(json);
|
|
560
|
+
const healed = bifrost ? { changed: false, notes: [] } : healAnthropicPayload(json);
|
|
514
561
|
if (healed.changed) {
|
|
515
562
|
json = healed.payload;
|
|
516
563
|
modified = true;
|
|
517
564
|
debugLog('Healer (direct):', healed.notes.join('; '));
|
|
518
565
|
}
|
|
519
|
-
if (profile?.thinkingMode === 'off' && json.thinking) {
|
|
566
|
+
if (!bifrost && profile?.thinkingMode === 'off' && json.thinking) {
|
|
520
567
|
json = { ...json };
|
|
521
568
|
delete json.thinking;
|
|
522
569
|
modified = true;
|
|
@@ -752,6 +799,25 @@ async function handleConvert(clientFormat, req, res, bodyBuffer, opts = {}) {
|
|
|
752
799
|
let answered = false; // set only after a complete 2xx answer; the half upload depends on it
|
|
753
800
|
|
|
754
801
|
try {
|
|
802
|
+
// Bifrost: the provider's own client reaching its own account through intact. Bytes and headers
|
|
803
|
+
// go as sent, only the key changes: no healer, no thinkingMode, no conversion.
|
|
804
|
+
if (clientFormat === 'anthropic' && await crossesBifrost(req, profile, mappedModel)) {
|
|
805
|
+
const { url } = upstreamEndpoint(profile, 'anthropic', mappedModel, ir.stream, req);
|
|
806
|
+
const headers = bifrostHeaders(req, profile);
|
|
807
|
+
if (traceId) headers['x-intact-trace'] = traceId;
|
|
808
|
+
try {
|
|
809
|
+
const r = await forwardAnthropicDirect(res, payload, bodyBuffer, withClientQuery(url, req), headers, mappedModel, ac.signal, profile, true);
|
|
810
|
+
answered = !r.error && r.status >= 200 && r.status < 300;
|
|
811
|
+
log({ outFormat: 'anthropic', status: r.status, tokens: r.tokens, responsePreview: '(bifrost)', error: r.error || undefined });
|
|
812
|
+
} catch (err) {
|
|
813
|
+
if (ac.signal.aborted) return log({ status: 499, error: 'client disconnected' });
|
|
814
|
+
console.error(`[${profileKey}] Bifrost forward error:`, err.message);
|
|
815
|
+
sendClientError(res, clientFormat, 502, `Bifrost forward error: ${err.message}`);
|
|
816
|
+
log({ status: 502, error: err.message });
|
|
817
|
+
}
|
|
818
|
+
return;
|
|
819
|
+
}
|
|
820
|
+
|
|
755
821
|
// Fast path: anthropic in/out goes straight through, preserving original bytes (including thinking signatures).
|
|
756
822
|
// Note: this branch skips the Healer Engine because it bypasses the IR.
|
|
757
823
|
if (clientFormat === 'anthropic' && outFormat === 'anthropic') {
|
|
@@ -889,11 +955,14 @@ async function handleCountTokens(req, res, buf) {
|
|
|
889
955
|
const { profile } = getActiveProfile('anthropic', req);
|
|
890
956
|
if (profile) {
|
|
891
957
|
const mappedModel = mapModel(payload.model || '', profile, 'anthropic');
|
|
892
|
-
|
|
893
|
-
|
|
894
|
-
const
|
|
958
|
+
const bifrost = await crossesBifrost(req, profile, mappedModel);
|
|
959
|
+
if (bifrost || resolveOutFormat(profile, mappedModel) === 'anthropic') {
|
|
960
|
+
const ep = upstreamEndpoint(profile, 'anthropic', mappedModel, false, req);
|
|
961
|
+
const headers = bifrost ? bifrostHeaders(req, profile) : ep.headers;
|
|
962
|
+
let countUrl = profile.endpoints?.countTokens || ep.url.replace(/\/messages$/, '/messages/count_tokens');
|
|
963
|
+
if (bifrost) countUrl = withClientQuery(countUrl, req);
|
|
895
964
|
try {
|
|
896
|
-
const body = healAnthropicPayload({ ...payload, model: mappedModel }).payload;
|
|
965
|
+
const body = bifrost ? { ...payload, model: mappedModel } : healAnthropicPayload({ ...payload, model: mappedModel }).payload;
|
|
897
966
|
const r = await fetch(countUrl, { method: 'POST', headers, body: JSON.stringify(body), signal: AbortSignal.timeout(15000) });
|
|
898
967
|
if (r.ok) {
|
|
899
968
|
const j = await r.json();
|
|
@@ -1,93 +1,93 @@
|
|
|
1
|
-
---
|
|
2
|
-
name: llm-switcher
|
|
3
|
-
description: Mandatory guidance for AI coding agents and token compression tools (Headroom, RTK, Ponytail, Repomix). Enforces routing all LLM traffic through the local LLM Switcher edge gateway (http://127.0.0.1:3456) instead of directly calling provider endpoints.
|
|
4
|
-
---
|
|
5
|
-
|
|
6
|
-
# LLM Switcher Edge Gateway Skill
|
|
7
|
-
|
|
8
|
-
This skill governs how AI coding agents (Claude Code, Codex, Cursor, Windsurf, Opencode) and prompt optimization/compression tools (Headroom, RTK, Ponytail, Repomix) must interact with LLM providers on this workstation.
|
|
9
|
-
|
|
10
|
-
## 1. The Core Directive
|
|
11
|
-
|
|
12
|
-
> **ALL OUTBOUND LLM REQUESTS MUST ROUTE THROUGH LLM SWITCHER (`http://127.0.0.1:3456`).**
|
|
13
|
-
>
|
|
14
|
-
> Never configure any tool, script, sub-agent, or token compression proxy to call external provider endpoints directly (such as `https://api.anthropic.com`, `https://api.openai.com`, or `https://generativelanguage.googleapis.com`).
|
|
15
|
-
>
|
|
16
|
-
> If a token compression tool (Headroom, RTK, Ponytail) is used, it MUST be positioned as an **inner processing layer**, and its **upstream destination MUST be configured to point to LLM Switcher** at `http://127.0.0.1:3456`.
|
|
17
|
-
|
|
18
|
-
```
|
|
19
|
-
[Client CLI / Agent]
|
|
20
|
-
│
|
|
21
|
-
▼
|
|
22
|
-
[Token Compressor: RTK / Headroom / Ponytail] <-- Inner Layer
|
|
23
|
-
│
|
|
24
|
-
▼ (MANDATORY: Upstream MUST point to http://127.0.0.1:3456)
|
|
25
|
-
[LLM Switcher Edge Gateway (:3456)] <-- Outermost Gatekeeper
|
|
26
|
-
│
|
|
27
|
-
▼ (Heals schemas, maps 1M context, extracts thinking, converts protocol)
|
|
28
|
-
[Internet / intact (Recommended) / 9Router / Upstream LLM]
|
|
29
|
-
```
|
|
30
|
-
|
|
31
|
-
## 2. Why This is Mandatory
|
|
32
|
-
|
|
33
|
-
Direct outbound calls from token compressors will break the developer environment for several reasons:
|
|
34
|
-
|
|
35
|
-
1. **Schema Breakage & The Healer Engine:**
|
|
36
|
-
Token optimizers frequently prune conversation history to save tokens. In doing so, they often delete an `assistant` turn containing a `tool_use`, while leaving the subsequent `user` turn containing a `tool_result`.
|
|
37
|
-
- Anthropic and OpenAI APIs strictly reject this with `HTTP 400 (orphaned tool_use_id)`.
|
|
38
|
-
- **LLM Switcher** intercepts these malformed requests and runs its **Healer Engine**, automatically converting orphaned results into safe context text blocks and merging consecutive same-role turns so upstream providers accept the payload cleanly.
|
|
39
|
-
2. **Thinking / Reasoning Protection:**
|
|
40
|
-
Many token compressors naively strip `thinking` parameters or reasoning budgets to reduce token count.
|
|
41
|
-
- When routed through LLM Switcher, the gateway detects reasoning models (such as `ag/claude-opus-4-6-thinking` or `ag/gemini-3.8-flash`) and **automatically restores thinking parameters**, ensuring the model retains its full reasoning capabilities.
|
|
42
|
-
3. **Official Context Windows & Dynamic Model Discovery:**
|
|
43
|
-
Context windows follow the model you select without mutating permanent settings. The gateway auto-discovers official models from Anthropic and OpenAI, refreshing mappings on tool version upgrades (`switch models`).
|
|
44
|
-
4. **Server Gateway Synergy (intact / 9Router):**
|
|
45
|
-
LLM Switcher pairs directly with server-side gateways like **intact** (strongly recommended for credential proxying, account pooling, and contract lab self-healing) or 9Router, handling retries and quotas far better than any local tool.
|
|
46
|
-
|
|
47
|
-
## 3. Configuring Token Compression Tools
|
|
48
|
-
|
|
49
|
-
When setting up or assisting the user with token compression tools, apply these exact configurations:
|
|
50
|
-
|
|
51
|
-
### A. Headroom (Prompt Compression Proxy)
|
|
52
|
-
When configuring Headroom's upstream endpoint:
|
|
53
|
-
- In Headroom config (`headroom.yaml` or environment variables):
|
|
54
|
-
```bash
|
|
55
|
-
ANTHROPIC_BASE_URL="http://127.0.0.1:3456"
|
|
56
|
-
OPENAI_BASE_URL="http://127.0.0.1:3456/v1"
|
|
57
|
-
```
|
|
58
|
-
- Point your CLI to Headroom's listening port (e.g. `http://127.0.0.1:8787`). Headroom compresses the prompt and forwards it to LLM Switcher on port `3456`.
|
|
59
|
-
|
|
60
|
-
### B. RTK (Rust Token Killer / Repomix)
|
|
61
|
-
- In RTK configuration or execution flags:
|
|
62
|
-
```bash
|
|
63
|
-
# Ensure RTK's target upstream proxy is LLM Switcher
|
|
64
|
-
export UPSTREAM_URL="http://127.0.0.1:3456"
|
|
65
|
-
```
|
|
66
|
-
|
|
67
|
-
### C. Ponytail / Custom Proxies
|
|
68
|
-
- Always set the proxy's upstream host to:
|
|
69
|
-
`http://127.0.0.1:3456` (or `http://127.0.0.1:3456/v1` for OpenAI-compatible tools).
|
|
70
|
-
|
|
71
|
-
### D. Sub-Agents & Spawned Terminal Commands
|
|
72
|
-
When an agent spawns sub-processes that call LLM CLIs:
|
|
73
|
-
- **Never advise sourcing `env.sh` or `env.cmd`.** Those files carry no proxy variables any
|
|
74
|
-
more, so sourcing one changes nothing — and a stale variable an older version once wrote
|
|
75
|
-
would still point the tool at a port where nothing listens.
|
|
76
|
-
- Run `switch claude <profile>` or `switch codex <profile>` instead. The shims in
|
|
77
|
-
`~/.llm-switcher/bin` inject the proxy variables into the tool process itself and clear
|
|
78
|
-
anything stale first.
|
|
79
|
-
- `claude` and `codex` on `PATH` already are those shims: invoke them normally, and never
|
|
80
|
-
wrap them in another script.
|
|
81
|
-
|
|
82
|
-
## 4. Operational Rules for AI Agents
|
|
83
|
-
|
|
84
|
-
1. **NEVER edit `~/.claude/settings.json` directly.**
|
|
85
|
-
LLM Switcher uses zero-mutation launcher flags (`active.flag`). Writing custom base URLs or auth tokens into `settings.json` triggers warning banners in Claude Code and risks breaking the user's official subscription. Use `switch` CLI commands instead.
|
|
86
|
-
2. **Check Gateway Health Before Complex Operations:**
|
|
87
|
-
Run `switch status` or call the `switcher_audit` MCP tool to confirm:
|
|
88
|
-
- LLM Switcher is active on port `3456`.
|
|
89
|
-
- The active profile matches the intended CLI target (Claude Code, Codex, or OpenAI).
|
|
90
|
-
3. **Verify Routing When Errors Occur:**
|
|
91
|
-
If a tool fails with `HTTP 400`, `HTTP 502`, or connection errors:
|
|
92
|
-
- Run `switch doctor` to audit port collisions and environment variables.
|
|
93
|
-
- Inspect recent request logs via `http://127.0.0.1:3456/ui` (Tab 4: Live Inspector) to see if an intermediary tool mangled the payload.
|
|
1
|
+
---
|
|
2
|
+
name: llm-switcher
|
|
3
|
+
description: Mandatory guidance for AI coding agents and token compression tools (Headroom, RTK, Ponytail, Repomix). Enforces routing all LLM traffic through the local LLM Switcher edge gateway (http://127.0.0.1:3456) instead of directly calling provider endpoints.
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# LLM Switcher Edge Gateway Skill
|
|
7
|
+
|
|
8
|
+
This skill governs how AI coding agents (Claude Code, Codex, Cursor, Windsurf, Opencode) and prompt optimization/compression tools (Headroom, RTK, Ponytail, Repomix) must interact with LLM providers on this workstation.
|
|
9
|
+
|
|
10
|
+
## 1. The Core Directive
|
|
11
|
+
|
|
12
|
+
> **ALL OUTBOUND LLM REQUESTS MUST ROUTE THROUGH LLM SWITCHER (`http://127.0.0.1:3456`).**
|
|
13
|
+
>
|
|
14
|
+
> Never configure any tool, script, sub-agent, or token compression proxy to call external provider endpoints directly (such as `https://api.anthropic.com`, `https://api.openai.com`, or `https://generativelanguage.googleapis.com`).
|
|
15
|
+
>
|
|
16
|
+
> If a token compression tool (Headroom, RTK, Ponytail) is used, it MUST be positioned as an **inner processing layer**, and its **upstream destination MUST be configured to point to LLM Switcher** at `http://127.0.0.1:3456`.
|
|
17
|
+
|
|
18
|
+
```
|
|
19
|
+
[Client CLI / Agent]
|
|
20
|
+
│
|
|
21
|
+
▼
|
|
22
|
+
[Token Compressor: RTK / Headroom / Ponytail] <-- Inner Layer
|
|
23
|
+
│
|
|
24
|
+
▼ (MANDATORY: Upstream MUST point to http://127.0.0.1:3456)
|
|
25
|
+
[LLM Switcher Edge Gateway (:3456)] <-- Outermost Gatekeeper
|
|
26
|
+
│
|
|
27
|
+
▼ (Heals schemas, maps 1M context, extracts thinking, converts protocol)
|
|
28
|
+
[Internet / intact (Recommended) / 9Router / Upstream LLM]
|
|
29
|
+
```
|
|
30
|
+
|
|
31
|
+
## 2. Why This is Mandatory
|
|
32
|
+
|
|
33
|
+
Direct outbound calls from token compressors will break the developer environment for several reasons:
|
|
34
|
+
|
|
35
|
+
1. **Schema Breakage & The Healer Engine:**
|
|
36
|
+
Token optimizers frequently prune conversation history to save tokens. In doing so, they often delete an `assistant` turn containing a `tool_use`, while leaving the subsequent `user` turn containing a `tool_result`.
|
|
37
|
+
- Anthropic and OpenAI APIs strictly reject this with `HTTP 400 (orphaned tool_use_id)`.
|
|
38
|
+
- **LLM Switcher** intercepts these malformed requests and runs its **Healer Engine**, automatically converting orphaned results into safe context text blocks and merging consecutive same-role turns so upstream providers accept the payload cleanly.
|
|
39
|
+
2. **Thinking / Reasoning Protection:**
|
|
40
|
+
Many token compressors naively strip `thinking` parameters or reasoning budgets to reduce token count.
|
|
41
|
+
- When routed through LLM Switcher, the gateway detects reasoning models (such as `ag/claude-opus-4-6-thinking` or `ag/gemini-3.8-flash`) and **automatically restores thinking parameters**, ensuring the model retains its full reasoning capabilities.
|
|
42
|
+
3. **Official Context Windows & Dynamic Model Discovery:**
|
|
43
|
+
Context windows follow the model you select without mutating permanent settings. The gateway auto-discovers official models from Anthropic and OpenAI, refreshing mappings on tool version upgrades (`switch models`).
|
|
44
|
+
4. **Server Gateway Synergy (intact / 9Router):**
|
|
45
|
+
LLM Switcher pairs directly with server-side gateways like **intact** (strongly recommended for credential proxying, account pooling, and contract lab self-healing) or 9Router, handling retries and quotas far better than any local tool.
|
|
46
|
+
|
|
47
|
+
## 3. Configuring Token Compression Tools
|
|
48
|
+
|
|
49
|
+
When setting up or assisting the user with token compression tools, apply these exact configurations:
|
|
50
|
+
|
|
51
|
+
### A. Headroom (Prompt Compression Proxy)
|
|
52
|
+
When configuring Headroom's upstream endpoint:
|
|
53
|
+
- In Headroom config (`headroom.yaml` or environment variables):
|
|
54
|
+
```bash
|
|
55
|
+
ANTHROPIC_BASE_URL="http://127.0.0.1:3456"
|
|
56
|
+
OPENAI_BASE_URL="http://127.0.0.1:3456/v1"
|
|
57
|
+
```
|
|
58
|
+
- Point your CLI to Headroom's listening port (e.g. `http://127.0.0.1:8787`). Headroom compresses the prompt and forwards it to LLM Switcher on port `3456`.
|
|
59
|
+
|
|
60
|
+
### B. RTK (Rust Token Killer / Repomix)
|
|
61
|
+
- In RTK configuration or execution flags:
|
|
62
|
+
```bash
|
|
63
|
+
# Ensure RTK's target upstream proxy is LLM Switcher
|
|
64
|
+
export UPSTREAM_URL="http://127.0.0.1:3456"
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
### C. Ponytail / Custom Proxies
|
|
68
|
+
- Always set the proxy's upstream host to:
|
|
69
|
+
`http://127.0.0.1:3456` (or `http://127.0.0.1:3456/v1` for OpenAI-compatible tools).
|
|
70
|
+
|
|
71
|
+
### D. Sub-Agents & Spawned Terminal Commands
|
|
72
|
+
When an agent spawns sub-processes that call LLM CLIs:
|
|
73
|
+
- **Never advise sourcing `env.sh` or `env.cmd`.** Those files carry no proxy variables any
|
|
74
|
+
more, so sourcing one changes nothing — and a stale variable an older version once wrote
|
|
75
|
+
would still point the tool at a port where nothing listens.
|
|
76
|
+
- Run `switch claude <profile>` or `switch codex <profile>` instead. The shims in
|
|
77
|
+
`~/.llm-switcher/bin` inject the proxy variables into the tool process itself and clear
|
|
78
|
+
anything stale first.
|
|
79
|
+
- `claude` and `codex` on `PATH` already are those shims: invoke them normally, and never
|
|
80
|
+
wrap them in another script.
|
|
81
|
+
|
|
82
|
+
## 4. Operational Rules for AI Agents
|
|
83
|
+
|
|
84
|
+
1. **NEVER edit `~/.claude/settings.json` directly.**
|
|
85
|
+
LLM Switcher uses zero-mutation launcher flags (`active.flag`). Writing custom base URLs or auth tokens into `settings.json` triggers warning banners in Claude Code and risks breaking the user's official subscription. Use `switch` CLI commands instead.
|
|
86
|
+
2. **Check Gateway Health Before Complex Operations:**
|
|
87
|
+
Run `switch status` or call the `switcher_audit` MCP tool to confirm:
|
|
88
|
+
- LLM Switcher is active on port `3456`.
|
|
89
|
+
- The active profile matches the intended CLI target (Claude Code, Codex, or OpenAI).
|
|
90
|
+
3. **Verify Routing When Errors Occur:**
|
|
91
|
+
If a tool fails with `HTTP 400`, `HTTP 502`, or connection errors:
|
|
92
|
+
- Run `switch doctor` to audit port collisions and environment variables.
|
|
93
|
+
- Inspect recent request logs via `http://127.0.0.1:3456/ui` (Tab 4: Live Inspector) to see if an intermediary tool mangled the payload.
|
package/switch
CHANGED
|
File without changes
|
package/switch.cmd
CHANGED
|
@@ -1,2 +1,2 @@
|
|
|
1
|
-
@echo off
|
|
2
|
-
node "%~dp0switch.mjs" %*
|
|
1
|
+
@echo off
|
|
2
|
+
node "%~dp0switch.mjs" %*
|
|
@@ -260,6 +260,11 @@ function startUpstream() {
|
|
|
260
260
|
let body = '';
|
|
261
261
|
req.on('data', c => { body += c; });
|
|
262
262
|
req.on('end', () => {
|
|
263
|
+
// The Bifrost lookup reads one model entry; it is not a probe request.
|
|
264
|
+
if (req.method === 'GET' && req.url.includes('/models/')) {
|
|
265
|
+
res.writeHead(404, { 'Content-Type': 'application/json' });
|
|
266
|
+
return res.end('{}');
|
|
267
|
+
}
|
|
263
268
|
const json = body ? JSON.parse(body) : {};
|
|
264
269
|
received.push({ url: req.url, headers: req.headers, body: json });
|
|
265
270
|
if (JSON.stringify(json.messages || '').includes('ERR_400')) {
|
|
@@ -68,9 +68,23 @@ function startUpstream() {
|
|
|
68
68
|
req.on('data', c => { body += c; });
|
|
69
69
|
req.on('end', () => {
|
|
70
70
|
const json = body ? JSON.parse(body) : {};
|
|
71
|
-
received.push({ url: req.url, headers: req.headers, body: json });
|
|
71
|
+
received.push({ url: req.url, headers: req.headers, body: json, raw: body });
|
|
72
72
|
const lastUser = JSON.stringify(json.messages?.at(-1) ?? json.contents?.at(-1) ?? '');
|
|
73
73
|
|
|
74
|
+
// A stand-in intact: claude/* is a Claude Code account, the rest is not.
|
|
75
|
+
if (req.method === 'GET' && req.url.startsWith('/intact/v1/models/')) {
|
|
76
|
+
const id = decodeURIComponent(req.url.slice('/intact/v1/models/'.length));
|
|
77
|
+
res.writeHead(200, { 'Content-Type': 'application/json' });
|
|
78
|
+
return res.end(JSON.stringify({ id, owned_by: id.split('/')[0], ...(id.startsWith('claude/') ? { bifrost_ua: 'claude-cli/' } : {}) }));
|
|
79
|
+
}
|
|
80
|
+
if (req.url.startsWith('/intact/v1/messages')) {
|
|
81
|
+
res.writeHead(200, { 'Content-Type': 'application/json' });
|
|
82
|
+
return res.end(JSON.stringify({ id: 'msg_n', type: 'message', role: 'assistant', model: json.model, content: [{ type: 'text', text: 'bifrost ok' }], stop_reason: 'end_turn', usage: { input_tokens: 3, output_tokens: 2 } }));
|
|
83
|
+
}
|
|
84
|
+
if (req.url.startsWith('/intact/v1/chat/completions')) {
|
|
85
|
+
res.writeHead(200, { 'Content-Type': 'application/json' });
|
|
86
|
+
return res.end(JSON.stringify({ id: 'c1', object: 'chat.completion', model: json.model, choices: [{ index: 0, message: { role: 'assistant', content: 'converted ok' }, finish_reason: 'stop' }], usage: { prompt_tokens: 1, completion_tokens: 1 } }));
|
|
87
|
+
}
|
|
74
88
|
if (req.url.startsWith('/chat/v1/chat/completions')) {
|
|
75
89
|
if (lastUser.includes('RATE_LIMIT')) {
|
|
76
90
|
res.writeHead(429, { 'Content-Type': 'application/json', 'retry-after': '7' });
|
|
@@ -202,6 +216,7 @@ before(async () => {
|
|
|
202
216
|
pub: { name: 'Mock Public', mode: 'convert', inFormat: 'responses', baseURL: `${base}/chat/v1`, apiKey: 'sk-secret-pub', publicModels: ['gpt-5.6-sol', 'gpt-5.2'], defaultModels: { main: 'ag/mock-flash' }, model1M: { main: true } },
|
|
203
217
|
roles: { name: 'Mock Public Roles', mode: 'convert', inFormat: 'responses', baseURL: `${base}/chat/v1`, apiKey: 'sk-secret-roles', publicModels: ['gpt-5.6-sol', 'gpt-5.6-terra', 'gpt-5.6-luna'], defaultModels: { main: 'ag/mock-flash', review: 'ag/mock-review', subagent: 'ag/mock-low' } },
|
|
204
218
|
native: { name: 'Mock Strict OpenAI', mode: 'convert', inFormat: 'auto', outFormat: 'openai-chat', thinkingMode: 'native', baseURL: `${base}/chat/v1`, apiKey: 'sk-secret-native', defaultModels: models },
|
|
219
|
+
intactcc: { name: 'Mock intact', mode: 'convert', tool: 'claude', outFormat: 'openai-chat', thinkingMode: 'off', baseURL: `${base}/intact/v1`, apiKey: 'sk-intact-user', defaultModels: { opus: 'claude/claude-opus-5', sonnet: 'gem/flash', haiku: 'claude/claude-haiku-4-5', fable: 'gem/flash' } },
|
|
205
220
|
ant: { name: 'Mock Anthropic', mode: 'direct', inFormat: 'auto', outFormat: 'anthropic', baseURL: `${base}/ant`, apiKey: 'sk-secret-ant', defaultModels: models }
|
|
206
221
|
}
|
|
207
222
|
};
|
|
@@ -577,6 +592,40 @@ test('Direct Anthropic passthrough strips hop-by-hop headers and blocks client c
|
|
|
577
592
|
assert.equal(up.body.model, 'up-opus');
|
|
578
593
|
});
|
|
579
594
|
|
|
595
|
+
test('Bifrost: Claude Code to a Claude Code account on intact changes only the key', async () => {
|
|
596
|
+
// Orphaned tool_result and a thinking block: the healer and thinkingMode=off would both change these.
|
|
597
|
+
const body = { model: 'claude-opus-4-6', max_tokens: 10, thinking: { type: 'adaptive' }, metadata: { user_id: 'u1' },
|
|
598
|
+
messages: [{ role: 'user', content: [{ type: 'tool_result', tool_use_id: 'gone', content: 'x' }] }] };
|
|
599
|
+
const res = await post('/v1/messages?beta=true', body, {
|
|
600
|
+
'x-llm-profile': 'intactcc', 'user-agent': 'claude-cli/2.1.300 (external, cli)', 'x-api-key': 'client-own-key',
|
|
601
|
+
authorization: 'Bearer client-oauth', 'anthropic-beta': 'oauth-2025-04-20,new-beta', 'anthropic-dangerous-direct-browser-access': 'true', 'x-app': 'cli'
|
|
602
|
+
});
|
|
603
|
+
assert.equal(res.status, 200);
|
|
604
|
+
assert.equal((await res.json()).content[0].text, 'bifrost ok');
|
|
605
|
+
const up = received.at(-1);
|
|
606
|
+
assert.equal(up.url, '/intact/v1/messages?beta=true');
|
|
607
|
+
assert.equal(up.headers['x-api-key'], 'sk-intact-user');
|
|
608
|
+
assert.equal(up.headers.authorization, undefined);
|
|
609
|
+
assert.equal(up.headers['user-agent'], 'claude-cli/2.1.300 (external, cli)');
|
|
610
|
+
assert.equal(up.headers['anthropic-beta'], 'oauth-2025-04-20,new-beta');
|
|
611
|
+
assert.equal(up.headers['anthropic-dangerous-direct-browser-access'], 'true');
|
|
612
|
+
assert.equal(up.headers['x-app'], 'cli');
|
|
613
|
+
assert.equal(up.raw, JSON.stringify({ ...body, model: 'claude/claude-opus-5' }));
|
|
614
|
+
});
|
|
615
|
+
|
|
616
|
+
test('Bifrost stays off for a model that is not a Claude Code account', async () => {
|
|
617
|
+
const res = await post('/v1/messages', { model: 'claude-sonnet-5', max_tokens: 10, messages: [{ role: 'user', content: 'hi' }] },
|
|
618
|
+
{ 'x-llm-profile': 'intactcc', 'user-agent': 'claude-cli/2.1.300 (external, cli)' });
|
|
619
|
+
assert.equal(res.status, 200);
|
|
620
|
+
assert.equal(received.at(-1).url, '/intact/v1/chat/completions');
|
|
621
|
+
});
|
|
622
|
+
|
|
623
|
+
test('Bifrost stays off for a client that is not Claude Code', async () => {
|
|
624
|
+
await post('/v1/messages', { model: 'claude-opus-4-6', max_tokens: 10, messages: [{ role: 'user', content: 'hi' }] },
|
|
625
|
+
{ 'x-llm-profile': 'intactcc', 'user-agent': 'some-sdk/1.0' });
|
|
626
|
+
assert.equal(received.at(-1).url, '/intact/v1/chat/completions');
|
|
627
|
+
});
|
|
628
|
+
|
|
580
629
|
test('Healer: orphaned tool_result and missing tool_result produce a valid chat history', async () => {
|
|
581
630
|
await post('/v1/messages', {
|
|
582
631
|
model: 'claude-opus-4-6', max_tokens: 50, messages: [
|
package/tests/helpers.mjs
CHANGED
|
@@ -1,24 +1,24 @@
|
|
|
1
|
-
import assert from 'node:assert/strict';
|
|
2
|
-
|
|
3
|
-
// Validate an Anthropic event sequence: indexes increase in start order, deltas only target an open block.
|
|
4
|
-
export function assertValidAnthropicEvents(events) {
|
|
5
|
-
const open = new Set();
|
|
6
|
-
const seen = new Set();
|
|
7
|
-
let expectedNext = 0;
|
|
8
|
-
for (const { event, data } of events) {
|
|
9
|
-
if (event === 'content_block_start') {
|
|
10
|
-
assert.equal(data.index, expectedNext, `block index must be sequential (got ${data.index}, want ${expectedNext})`);
|
|
11
|
-
assert.ok(!seen.has(data.index), `index ${data.index} reused`);
|
|
12
|
-
seen.add(data.index);
|
|
13
|
-
open.add(data.index);
|
|
14
|
-
expectedNext++;
|
|
15
|
-
} else if (event === 'content_block_delta') {
|
|
16
|
-
assert.ok(open.has(data.index), `delta for block ${data.index} which is not open`);
|
|
17
|
-
} else if (event === 'content_block_stop') {
|
|
18
|
-
assert.ok(open.has(data.index), `stop for block ${data.index} which is not open`);
|
|
19
|
-
open.delete(data.index);
|
|
20
|
-
} else if (event === 'message_stop') {
|
|
21
|
-
assert.equal(open.size, 0, 'all blocks must be closed before message_stop');
|
|
22
|
-
}
|
|
23
|
-
}
|
|
24
|
-
}
|
|
1
|
+
import assert from 'node:assert/strict';
|
|
2
|
+
|
|
3
|
+
// Validate an Anthropic event sequence: indexes increase in start order, deltas only target an open block.
|
|
4
|
+
export function assertValidAnthropicEvents(events) {
|
|
5
|
+
const open = new Set();
|
|
6
|
+
const seen = new Set();
|
|
7
|
+
let expectedNext = 0;
|
|
8
|
+
for (const { event, data } of events) {
|
|
9
|
+
if (event === 'content_block_start') {
|
|
10
|
+
assert.equal(data.index, expectedNext, `block index must be sequential (got ${data.index}, want ${expectedNext})`);
|
|
11
|
+
assert.ok(!seen.has(data.index), `index ${data.index} reused`);
|
|
12
|
+
seen.add(data.index);
|
|
13
|
+
open.add(data.index);
|
|
14
|
+
expectedNext++;
|
|
15
|
+
} else if (event === 'content_block_delta') {
|
|
16
|
+
assert.ok(open.has(data.index), `delta for block ${data.index} which is not open`);
|
|
17
|
+
} else if (event === 'content_block_stop') {
|
|
18
|
+
assert.ok(open.has(data.index), `stop for block ${data.index} which is not open`);
|
|
19
|
+
open.delete(data.index);
|
|
20
|
+
} else if (event === 'message_stop') {
|
|
21
|
+
assert.equal(open.size, 0, 'all blocks must be closed before message_stop');
|
|
22
|
+
}
|
|
23
|
+
}
|
|
24
|
+
}
|