llm-switcher 1.2.8 → 1.2.10

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "llm-switcher",
3
- "version": "1.2.8",
3
+ "version": "1.2.10",
4
4
  "description": "Zero-dependency multi-protocol edge gateway & provider switcher for Claude Code and Codex, with OpenAI-compatible, Anthropic and Vertex upstreams",
5
5
  "keywords": [
6
6
  "llm",
package/proxy.mjs CHANGED
@@ -500,7 +500,54 @@ function previewOf(ir) {
500
500
  // ----------------------------------------------------
501
501
  const HOP_BY_HOP = new Set(['content-length', 'content-encoding', 'transfer-encoding', 'connection', 'keep-alive']);
502
502
 
503
- async function forwardAnthropicDirect(res, payload, bodyBuffer, url, headers, mappedModel, signal, profile) {
503
+ // Bifrost: intact names, per model, the User-Agent prefix of the provider's own client (bifrost_ua).
504
+ // When the caller is that client, the request crosses as it is: only the key changes.
505
+ const bifrostUACache = new Map();
506
+ async function bifrostUAFor(profile, model) {
507
+ const base = String(profile.baseURL || '').replace(/\/+$/, '');
508
+ if (!base || !model) return '';
509
+ const key = `${base}\n${model}`;
510
+ const hit = bifrostUACache.get(key);
511
+ if (hit && hit.until > Date.now()) return hit.ua;
512
+ let ua = '';
513
+ let ttl = 10 * 60 * 1000;
514
+ try {
515
+ const r = await fetch(`${base}/models/${model.split('/').map(encodeURIComponent).join('/')}`, {
516
+ headers: { authorization: `Bearer ${profile.apiKey || ''}` }, signal: AbortSignal.timeout(3000)
517
+ });
518
+ if (r.ok) ua = String((await r.json())?.bifrost_ua || '');
519
+ } catch {
520
+ ttl = 30 * 1000; // an unreachable list must not pin the decision for long
521
+ }
522
+ bifrostUACache.set(key, { ua, until: Date.now() + ttl });
523
+ return ua;
524
+ }
525
+
526
+ async function crossesBifrost(req, profile, model) {
527
+ const clientUA = String(req.headers['user-agent'] || '');
528
+ if (!clientUA) return false;
529
+ const ua = await bifrostUAFor(profile, model);
530
+ return Boolean(ua) && clientUA.startsWith(ua);
531
+ }
532
+
533
+ // Every client header except its own credentials, the switcher's control headers and hop-by-hop.
534
+ function bifrostHeaders(req, profile) {
535
+ const headers = {};
536
+ for (const [k, v] of Object.entries(req.headers)) {
537
+ const lk = k.toLowerCase();
538
+ if (BLOCKED_PASSTHROUGH.has(lk) || HOP_BY_HOP.has(lk) || lk === 'host' || lk === 'authorization' || lk === 'accept-encoding') continue;
539
+ headers[lk] = v;
540
+ }
541
+ headers['x-api-key'] = profile.apiKey || '';
542
+ return headers;
543
+ }
544
+
545
+ function withClientQuery(url, req) {
546
+ const q = String(req.url || '').indexOf('?');
547
+ return q < 0 ? url : url + (url.includes('?') ? '&' : '?') + req.url.slice(q + 1);
548
+ }
549
+
550
+ async function forwardAnthropicDirect(res, payload, bodyBuffer, url, headers, mappedModel, signal, profile, bifrost = false) {
504
551
  debugLog('Direct forward to native Anthropic endpoint:', url);
505
552
 
506
553
  // Only re-serialize when a fix is actually needed; otherwise forward the client's original bytes.
@@ -510,13 +557,13 @@ async function forwardAnthropicDirect(res, payload, bodyBuffer, url, headers, ma
510
557
  json = { ...json, model: mappedModel };
511
558
  modified = true;
512
559
  }
513
- const healed = healAnthropicPayload(json);
560
+ const healed = bifrost ? { changed: false, notes: [] } : healAnthropicPayload(json);
514
561
  if (healed.changed) {
515
562
  json = healed.payload;
516
563
  modified = true;
517
564
  debugLog('Healer (direct):', healed.notes.join('; '));
518
565
  }
519
- if (profile?.thinkingMode === 'off' && json.thinking) {
566
+ if (!bifrost && profile?.thinkingMode === 'off' && json.thinking) {
520
567
  json = { ...json };
521
568
  delete json.thinking;
522
569
  modified = true;
@@ -728,10 +775,12 @@ async function handleConvert(clientFormat, req, res, bodyBuffer, opts = {}) {
728
775
  const requestPreview = previewOf(ir);
729
776
  const requestedModel = ir.model || payload.model || '';
730
777
  const mappedModel = mapModel(requestedModel, profile, clientFormat);
778
+ const bifrost = clientFormat === 'anthropic' && await crossesBifrost(req, profile, mappedModel);
731
779
  const outFormat = resolveOutFormat(profile, mappedModel);
732
- console.log(`[llm-switcher] ${clientFormat} -> ${outFormat} "${requestedModel}" -> "${mappedModel}" [${profile.name || profileKey}]`);
780
+ const route = bifrost ? 'bifrost' : outFormat;
781
+ console.log(`[llm-switcher] ${clientFormat} -> ${route} "${requestedModel}" -> "${mappedModel}" [${profile.name || profileKey}]`);
733
782
 
734
- const logBase = { clientFormat, outFormat, profile: profileKey, model: mappedModel, stream: ir.stream, requestPreview };
783
+ const logBase = { clientFormat, outFormat: route, profile: profileKey, model: mappedModel, stream: ir.stream, requestPreview };
735
784
  const log = (extra) => logInspection({ ...logBase, duration: Date.now() - reqStartTime, tokens: { prompt: 0, completion: 0 }, ...extra });
736
785
 
737
786
  // Contract lab: a sampled request carries a trace id to intact, and the bytes this gateway
@@ -752,6 +801,25 @@ async function handleConvert(clientFormat, req, res, bodyBuffer, opts = {}) {
752
801
  let answered = false; // set only after a complete 2xx answer; the half upload depends on it
753
802
 
754
803
  try {
804
+ // Bifrost: the provider's own client reaching its own account through intact. Bytes and headers
805
+ // go as sent, only the key changes: no healer, no thinkingMode, no conversion.
806
+ if (bifrost) {
807
+ const { url } = upstreamEndpoint(profile, 'anthropic', mappedModel, ir.stream, req);
808
+ const headers = bifrostHeaders(req, profile);
809
+ if (traceId) headers['x-intact-trace'] = traceId;
810
+ try {
811
+ const r = await forwardAnthropicDirect(res, payload, bodyBuffer, withClientQuery(url, req), headers, mappedModel, ac.signal, profile, true);
812
+ answered = !r.error && r.status >= 200 && r.status < 300;
813
+ log({ status: r.status, tokens: r.tokens, responsePreview: '(bifrost)', error: r.error || undefined });
814
+ } catch (err) {
815
+ if (ac.signal.aborted) return log({ status: 499, error: 'client disconnected' });
816
+ console.error(`[${profileKey}] Bifrost forward error:`, err.message);
817
+ sendClientError(res, clientFormat, 502, `Bifrost forward error: ${err.message}`);
818
+ log({ status: 502, error: err.message });
819
+ }
820
+ return;
821
+ }
822
+
755
823
  // Fast path: anthropic in/out goes straight through, preserving original bytes (including thinking signatures).
756
824
  // Note: this branch skips the Healer Engine because it bypasses the IR.
757
825
  if (clientFormat === 'anthropic' && outFormat === 'anthropic') {
@@ -870,7 +938,7 @@ async function handleConvert(clientFormat, req, res, bodyBuffer, opts = {}) {
870
938
  toolResponse: halfTap?.text() || '',
871
939
  toolVersion: toolVersionFromUA(req.headers['user-agent']),
872
940
  inFormat: clientFormat,
873
- outFormat,
941
+ outFormat: bifrost ? 'anthropic' : outFormat, // Bifrost converts nothing
874
942
  openerKey: profile.apiKey || ''
875
943
  });
876
944
  }
@@ -889,11 +957,14 @@ async function handleCountTokens(req, res, buf) {
889
957
  const { profile } = getActiveProfile('anthropic', req);
890
958
  if (profile) {
891
959
  const mappedModel = mapModel(payload.model || '', profile, 'anthropic');
892
- if (resolveOutFormat(profile, mappedModel) === 'anthropic') {
893
- const { url, headers } = upstreamEndpoint(profile, 'anthropic', mappedModel, false, req);
894
- const countUrl = profile.endpoints?.countTokens || url.replace(/\/messages$/, '/messages/count_tokens');
960
+ const bifrost = await crossesBifrost(req, profile, mappedModel);
961
+ if (bifrost || resolveOutFormat(profile, mappedModel) === 'anthropic') {
962
+ const ep = upstreamEndpoint(profile, 'anthropic', mappedModel, false, req);
963
+ const headers = bifrost ? bifrostHeaders(req, profile) : ep.headers;
964
+ let countUrl = profile.endpoints?.countTokens || ep.url.replace(/\/messages$/, '/messages/count_tokens');
965
+ if (bifrost) countUrl = withClientQuery(countUrl, req);
895
966
  try {
896
- const body = healAnthropicPayload({ ...payload, model: mappedModel }).payload;
967
+ const body = bifrost ? { ...payload, model: mappedModel } : healAnthropicPayload({ ...payload, model: mappedModel }).payload;
897
968
  const r = await fetch(countUrl, { method: 'POST', headers, body: JSON.stringify(body), signal: AbortSignal.timeout(15000) });
898
969
  if (r.ok) {
899
970
  const j = await r.json();
@@ -1,93 +1,93 @@
1
- ---
2
- name: llm-switcher
3
- description: Mandatory guidance for AI coding agents and token compression tools (Headroom, RTK, Ponytail, Repomix). Enforces routing all LLM traffic through the local LLM Switcher edge gateway (http://127.0.0.1:3456) instead of directly calling provider endpoints.
4
- ---
5
-
6
- # LLM Switcher Edge Gateway Skill
7
-
8
- This skill governs how AI coding agents (Claude Code, Codex, Cursor, Windsurf, Opencode) and prompt optimization/compression tools (Headroom, RTK, Ponytail, Repomix) must interact with LLM providers on this workstation.
9
-
10
- ## 1. The Core Directive
11
-
12
- > **ALL OUTBOUND LLM REQUESTS MUST ROUTE THROUGH LLM SWITCHER (`http://127.0.0.1:3456`).**
13
- >
14
- > Never configure any tool, script, sub-agent, or token compression proxy to call external provider endpoints directly (such as `https://api.anthropic.com`, `https://api.openai.com`, or `https://generativelanguage.googleapis.com`).
15
- >
16
- > If a token compression tool (Headroom, RTK, Ponytail) is used, it MUST be positioned as an **inner processing layer**, and its **upstream destination MUST be configured to point to LLM Switcher** at `http://127.0.0.1:3456`.
17
-
18
- ```
19
- [Client CLI / Agent]
20
- │
21
- ▼
22
- [Token Compressor: RTK / Headroom / Ponytail] <-- Inner Layer
23
- │
24
- ▼ (MANDATORY: Upstream MUST point to http://127.0.0.1:3456)
25
- [LLM Switcher Edge Gateway (:3456)] <-- Outermost Gatekeeper
26
- │
27
- ▼ (Heals schemas, maps 1M context, extracts thinking, converts protocol)
28
- [Internet / intact (Recommended) / 9Router / Upstream LLM]
29
- ```
30
-
31
- ## 2. Why This is Mandatory
32
-
33
- Direct outbound calls from token compressors will break the developer environment for several reasons:
34
-
35
- 1. **Schema Breakage & The Healer Engine:**
36
- Token optimizers frequently prune conversation history to save tokens. In doing so, they often delete an `assistant` turn containing a `tool_use`, while leaving the subsequent `user` turn containing a `tool_result`.
37
- - Anthropic and OpenAI APIs strictly reject this with `HTTP 400 (orphaned tool_use_id)`.
38
- - **LLM Switcher** intercepts these malformed requests and runs its **Healer Engine**, automatically converting orphaned results into safe context text blocks and merging consecutive same-role turns so upstream providers accept the payload cleanly.
39
- 2. **Thinking / Reasoning Protection:**
40
- Many token compressors naively strip `thinking` parameters or reasoning budgets to reduce token count.
41
- - When routed through LLM Switcher, the gateway detects reasoning models (such as `ag/claude-opus-4-6-thinking` or `ag/gemini-3.8-flash`) and **automatically restores thinking parameters**, ensuring the model retains its full reasoning capabilities.
42
- 3. **Official Context Windows & Dynamic Model Discovery:**
43
- Context windows follow the model you select without mutating permanent settings. The gateway auto-discovers official models from Anthropic and OpenAI, refreshing mappings on tool version upgrades (`switch models`).
44
- 4. **Server Gateway Synergy (intact / 9Router):**
45
- LLM Switcher pairs directly with server-side gateways like **intact** (strongly recommended for credential proxying, account pooling, and contract lab self-healing) or 9Router, handling retries and quotas far better than any local tool.
46
-
47
- ## 3. Configuring Token Compression Tools
48
-
49
- When setting up or assisting the user with token compression tools, apply these exact configurations:
50
-
51
- ### A. Headroom (Prompt Compression Proxy)
52
- When configuring Headroom's upstream endpoint:
53
- - In Headroom config (`headroom.yaml` or environment variables):
54
- ```bash
55
- ANTHROPIC_BASE_URL="http://127.0.0.1:3456"
56
- OPENAI_BASE_URL="http://127.0.0.1:3456/v1"
57
- ```
58
- - Point your CLI to Headroom's listening port (e.g. `http://127.0.0.1:8787`). Headroom compresses the prompt and forwards it to LLM Switcher on port `3456`.
59
-
60
- ### B. RTK (Rust Token Killer / Repomix)
61
- - In RTK configuration or execution flags:
62
- ```bash
63
- # Ensure RTK's target upstream proxy is LLM Switcher
64
- export UPSTREAM_URL="http://127.0.0.1:3456"
65
- ```
66
-
67
- ### C. Ponytail / Custom Proxies
68
- - Always set the proxy's upstream host to:
69
- `http://127.0.0.1:3456` (or `http://127.0.0.1:3456/v1` for OpenAI-compatible tools).
70
-
71
- ### D. Sub-Agents & Spawned Terminal Commands
72
- When an agent spawns sub-processes that call LLM CLIs:
73
- - **Never advise sourcing `env.sh` or `env.cmd`.** Those files carry no proxy variables any
74
- more, so sourcing one changes nothing — and a stale variable an older version once wrote
75
- would still point the tool at a port where nothing listens.
76
- - Run `switch claude <profile>` or `switch codex <profile>` instead. The shims in
77
- `~/.llm-switcher/bin` inject the proxy variables into the tool process itself and clear
78
- anything stale first.
79
- - `claude` and `codex` on `PATH` already are those shims: invoke them normally, and never
80
- wrap them in another script.
81
-
82
- ## 4. Operational Rules for AI Agents
83
-
84
- 1. **NEVER edit `~/.claude/settings.json` directly.**
85
- LLM Switcher uses zero-mutation launcher flags (`active.flag`). Writing custom base URLs or auth tokens into `settings.json` triggers warning banners in Claude Code and risks breaking the user's official subscription. Use `switch` CLI commands instead.
86
- 2. **Check Gateway Health Before Complex Operations:**
87
- Run `switch status` or call the `switcher_audit` MCP tool to confirm:
88
- - LLM Switcher is active on port `3456`.
89
- - The active profile matches the intended CLI target (Claude Code, Codex, or OpenAI).
90
- 3. **Verify Routing When Errors Occur:**
91
- If a tool fails with `HTTP 400`, `HTTP 502`, or connection errors:
92
- - Run `switch doctor` to audit port collisions and environment variables.
93
- - Inspect recent request logs via `http://127.0.0.1:3456/ui` (Tab 4: Live Inspector) to see if an intermediary tool mangled the payload.
1
+ ---
2
+ name: llm-switcher
3
+ description: Mandatory guidance for AI coding agents and token compression tools (Headroom, RTK, Ponytail, Repomix). Enforces routing all LLM traffic through the local LLM Switcher edge gateway (http://127.0.0.1:3456) instead of directly calling provider endpoints.
4
+ ---
5
+
6
+ # LLM Switcher Edge Gateway Skill
7
+
8
+ This skill governs how AI coding agents (Claude Code, Codex, Cursor, Windsurf, Opencode) and prompt optimization/compression tools (Headroom, RTK, Ponytail, Repomix) must interact with LLM providers on this workstation.
9
+
10
+ ## 1. The Core Directive
11
+
12
+ > **ALL OUTBOUND LLM REQUESTS MUST ROUTE THROUGH LLM SWITCHER (`http://127.0.0.1:3456`).**
13
+ >
14
+ > Never configure any tool, script, sub-agent, or token compression proxy to call external provider endpoints directly (such as `https://api.anthropic.com`, `https://api.openai.com`, or `https://generativelanguage.googleapis.com`).
15
+ >
16
+ > If a token compression tool (Headroom, RTK, Ponytail) is used, it MUST be positioned as an **inner processing layer**, and its **upstream destination MUST be configured to point to LLM Switcher** at `http://127.0.0.1:3456`.
17
+
18
+ ```
19
+ [Client CLI / Agent]
20
+ │
21
+ ▼
22
+ [Token Compressor: RTK / Headroom / Ponytail] <-- Inner Layer
23
+ │
24
+ ▼ (MANDATORY: Upstream MUST point to http://127.0.0.1:3456)
25
+ [LLM Switcher Edge Gateway (:3456)] <-- Outermost Gatekeeper
26
+ │
27
+ ▼ (Heals schemas, maps 1M context, extracts thinking, converts protocol)
28
+ [Internet / intact (Recommended) / 9Router / Upstream LLM]
29
+ ```
30
+
31
+ ## 2. Why This is Mandatory
32
+
33
+ Direct outbound calls from token compressors will break the developer environment for several reasons:
34
+
35
+ 1. **Schema Breakage & The Healer Engine:**
36
+ Token optimizers frequently prune conversation history to save tokens. In doing so, they often delete an `assistant` turn containing a `tool_use`, while leaving the subsequent `user` turn containing a `tool_result`.
37
+ - Anthropic and OpenAI APIs strictly reject this with `HTTP 400 (orphaned tool_use_id)`.
38
+ - **LLM Switcher** intercepts these malformed requests and runs its **Healer Engine**, automatically converting orphaned results into safe context text blocks and merging consecutive same-role turns so upstream providers accept the payload cleanly.
39
+ 2. **Thinking / Reasoning Protection:**
40
+ Many token compressors naively strip `thinking` parameters or reasoning budgets to reduce token count.
41
+ - When routed through LLM Switcher, the gateway detects reasoning models (such as `ag/claude-opus-4-6-thinking` or `ag/gemini-3.8-flash`) and **automatically restores thinking parameters**, ensuring the model retains its full reasoning capabilities.
42
+ 3. **Official Context Windows & Dynamic Model Discovery:**
43
+ Context windows follow the model you select without mutating permanent settings. The gateway auto-discovers official models from Anthropic and OpenAI, refreshing mappings on tool version upgrades (`switch models`).
44
+ 4. **Server Gateway Synergy (intact / 9Router):**
45
+ LLM Switcher pairs directly with server-side gateways like **intact** (strongly recommended for credential proxying, account pooling, and contract lab self-healing) or 9Router, handling retries and quotas far better than any local tool.
46
+
47
+ ## 3. Configuring Token Compression Tools
48
+
49
+ When setting up or assisting the user with token compression tools, apply these exact configurations:
50
+
51
+ ### A. Headroom (Prompt Compression Proxy)
52
+ When configuring Headroom's upstream endpoint:
53
+ - In Headroom config (`headroom.yaml` or environment variables):
54
+ ```bash
55
+ ANTHROPIC_BASE_URL="http://127.0.0.1:3456"
56
+ OPENAI_BASE_URL="http://127.0.0.1:3456/v1"
57
+ ```
58
+ - Point your CLI to Headroom's listening port (e.g. `http://127.0.0.1:8787`). Headroom compresses the prompt and forwards it to LLM Switcher on port `3456`.
59
+
60
+ ### B. RTK (Rust Token Killer / Repomix)
61
+ - In RTK configuration or execution flags:
62
+ ```bash
63
+ # Ensure RTK's target upstream proxy is LLM Switcher
64
+ export UPSTREAM_URL="http://127.0.0.1:3456"
65
+ ```
66
+
67
+ ### C. Ponytail / Custom Proxies
68
+ - Always set the proxy's upstream host to:
69
+ `http://127.0.0.1:3456` (or `http://127.0.0.1:3456/v1` for OpenAI-compatible tools).
70
+
71
+ ### D. Sub-Agents & Spawned Terminal Commands
72
+ When an agent spawns sub-processes that call LLM CLIs:
73
+ - **Never advise sourcing `env.sh` or `env.cmd`.** Those files carry no proxy variables any
74
+ more, so sourcing one changes nothing — and a stale variable an older version once wrote
75
+ would still point the tool at a port where nothing listens.
76
+ - Run `switch claude <profile>` or `switch codex <profile>` instead. The shims in
77
+ `~/.llm-switcher/bin` inject the proxy variables into the tool process itself and clear
78
+ anything stale first.
79
+ - `claude` and `codex` on `PATH` already are those shims: invoke them normally, and never
80
+ wrap them in another script.
81
+
82
+ ## 4. Operational Rules for AI Agents
83
+
84
+ 1. **NEVER edit `~/.claude/settings.json` directly.**
85
+ LLM Switcher uses zero-mutation launcher flags (`active.flag`). Writing custom base URLs or auth tokens into `settings.json` triggers warning banners in Claude Code and risks breaking the user's official subscription. Use `switch` CLI commands instead.
86
+ 2. **Check Gateway Health Before Complex Operations:**
87
+ Run `switch status` or call the `switcher_audit` MCP tool to confirm:
88
+ - LLM Switcher is active on port `3456`.
89
+ - The active profile matches the intended CLI target (Claude Code, Codex, or OpenAI).
90
+ 3. **Verify Routing When Errors Occur:**
91
+ If a tool fails with `HTTP 400`, `HTTP 502`, or connection errors:
92
+ - Run `switch doctor` to audit port collisions and environment variables.
93
+ - Inspect recent request logs via `http://127.0.0.1:3456/ui` (Tab 4: Live Inspector) to see if an intermediary tool mangled the payload.
package/switch CHANGED
File without changes
package/switch.cmd CHANGED
@@ -1,2 +1,2 @@
1
- @echo off
2
- node "%~dp0switch.mjs" %*
1
+ @echo off
2
+ node "%~dp0switch.mjs" %*
@@ -260,6 +260,11 @@ function startUpstream() {
260
260
  let body = '';
261
261
  req.on('data', c => { body += c; });
262
262
  req.on('end', () => {
263
+ // The Bifrost lookup reads one model entry; it is not a probe request.
264
+ if (req.method === 'GET' && req.url.includes('/models/')) {
265
+ res.writeHead(404, { 'Content-Type': 'application/json' });
266
+ return res.end('{}');
267
+ }
263
268
  const json = body ? JSON.parse(body) : {};
264
269
  received.push({ url: req.url, headers: req.headers, body: json });
265
270
  if (JSON.stringify(json.messages || '').includes('ERR_400')) {
@@ -68,9 +68,23 @@ function startUpstream() {
68
68
  req.on('data', c => { body += c; });
69
69
  req.on('end', () => {
70
70
  const json = body ? JSON.parse(body) : {};
71
- received.push({ url: req.url, headers: req.headers, body: json });
71
+ received.push({ url: req.url, headers: req.headers, body: json, raw: body });
72
72
  const lastUser = JSON.stringify(json.messages?.at(-1) ?? json.contents?.at(-1) ?? '');
73
73
 
74
+ // A stand-in intact: claude/* is a Claude Code account, the rest is not.
75
+ if (req.method === 'GET' && req.url.startsWith('/intact/v1/models/')) {
76
+ const id = decodeURIComponent(req.url.slice('/intact/v1/models/'.length));
77
+ res.writeHead(200, { 'Content-Type': 'application/json' });
78
+ return res.end(JSON.stringify({ id, owned_by: id.split('/')[0], ...(id.startsWith('claude/') ? { bifrost_ua: 'claude-cli/' } : {}) }));
79
+ }
80
+ if (req.url.startsWith('/intact/v1/messages')) {
81
+ res.writeHead(200, { 'Content-Type': 'application/json' });
82
+ return res.end(JSON.stringify({ id: 'msg_n', type: 'message', role: 'assistant', model: json.model, content: [{ type: 'text', text: 'bifrost ok' }], stop_reason: 'end_turn', usage: { input_tokens: 3, output_tokens: 2 } }));
83
+ }
84
+ if (req.url.startsWith('/intact/v1/chat/completions')) {
85
+ res.writeHead(200, { 'Content-Type': 'application/json' });
86
+ return res.end(JSON.stringify({ id: 'c1', object: 'chat.completion', model: json.model, choices: [{ index: 0, message: { role: 'assistant', content: 'converted ok' }, finish_reason: 'stop' }], usage: { prompt_tokens: 1, completion_tokens: 1 } }));
87
+ }
74
88
  if (req.url.startsWith('/chat/v1/chat/completions')) {
75
89
  if (lastUser.includes('RATE_LIMIT')) {
76
90
  res.writeHead(429, { 'Content-Type': 'application/json', 'retry-after': '7' });
@@ -202,6 +216,7 @@ before(async () => {
202
216
  pub: { name: 'Mock Public', mode: 'convert', inFormat: 'responses', baseURL: `${base}/chat/v1`, apiKey: 'sk-secret-pub', publicModels: ['gpt-5.6-sol', 'gpt-5.2'], defaultModels: { main: 'ag/mock-flash' }, model1M: { main: true } },
203
217
  roles: { name: 'Mock Public Roles', mode: 'convert', inFormat: 'responses', baseURL: `${base}/chat/v1`, apiKey: 'sk-secret-roles', publicModels: ['gpt-5.6-sol', 'gpt-5.6-terra', 'gpt-5.6-luna'], defaultModels: { main: 'ag/mock-flash', review: 'ag/mock-review', subagent: 'ag/mock-low' } },
204
218
  native: { name: 'Mock Strict OpenAI', mode: 'convert', inFormat: 'auto', outFormat: 'openai-chat', thinkingMode: 'native', baseURL: `${base}/chat/v1`, apiKey: 'sk-secret-native', defaultModels: models },
219
+ intactcc: { name: 'Mock intact', mode: 'convert', tool: 'claude', outFormat: 'openai-chat', thinkingMode: 'off', baseURL: `${base}/intact/v1`, apiKey: 'sk-intact-user', defaultModels: { opus: 'claude/claude-opus-5', sonnet: 'gem/flash', haiku: 'claude/claude-haiku-4-5', fable: 'gem/flash' } },
205
220
  ant: { name: 'Mock Anthropic', mode: 'direct', inFormat: 'auto', outFormat: 'anthropic', baseURL: `${base}/ant`, apiKey: 'sk-secret-ant', defaultModels: models }
206
221
  }
207
222
  };
@@ -577,6 +592,43 @@ test('Direct Anthropic passthrough strips hop-by-hop headers and blocks client c
577
592
  assert.equal(up.body.model, 'up-opus');
578
593
  });
579
594
 
595
+ test('Bifrost: Claude Code to a Claude Code account on intact changes only the key', async () => {
596
+ // Orphaned tool_result and a thinking block: the healer and thinkingMode=off would both change these.
597
+ const body = { model: 'claude-opus-4-6', max_tokens: 10, thinking: { type: 'adaptive' }, metadata: { user_id: 'u1' },
598
+ messages: [{ role: 'user', content: [{ type: 'tool_result', tool_use_id: 'gone', content: 'x' }] }] };
599
+ const res = await post('/v1/messages?beta=true', body, {
600
+ 'x-llm-profile': 'intactcc', 'user-agent': 'claude-cli/2.1.300 (external, cli)', 'x-api-key': 'client-own-key',
601
+ authorization: 'Bearer client-oauth', 'anthropic-beta': 'oauth-2025-04-20,new-beta', 'anthropic-dangerous-direct-browser-access': 'true', 'x-app': 'cli'
602
+ });
603
+ assert.equal(res.status, 200);
604
+ assert.equal((await res.json()).content[0].text, 'bifrost ok');
605
+ const up = received.at(-1);
606
+ assert.equal(up.url, '/intact/v1/messages?beta=true');
607
+ assert.equal(up.headers['x-api-key'], 'sk-intact-user');
608
+ assert.equal(up.headers.authorization, undefined);
609
+ assert.equal(up.headers['user-agent'], 'claude-cli/2.1.300 (external, cli)');
610
+ assert.equal(up.headers['anthropic-beta'], 'oauth-2025-04-20,new-beta');
611
+ assert.equal(up.headers['anthropic-dangerous-direct-browser-access'], 'true');
612
+ assert.equal(up.headers['x-app'], 'cli');
613
+ assert.equal(up.raw, JSON.stringify({ ...body, model: 'claude/claude-opus-5' }));
614
+ // The inspector and the log name the route that ran, not the profile's outFormat.
615
+ const logs = (await (await fetch(url('/api/logs'), { headers: withToken('/api/logs', {}) })).json()).logs;
616
+ assert.equal(logs[0].outFormat, 'bifrost');
617
+ });
618
+
619
+ test('Bifrost stays off for a model that is not a Claude Code account', async () => {
620
+ const res = await post('/v1/messages', { model: 'claude-sonnet-5', max_tokens: 10, messages: [{ role: 'user', content: 'hi' }] },
621
+ { 'x-llm-profile': 'intactcc', 'user-agent': 'claude-cli/2.1.300 (external, cli)' });
622
+ assert.equal(res.status, 200);
623
+ assert.equal(received.at(-1).url, '/intact/v1/chat/completions');
624
+ });
625
+
626
+ test('Bifrost stays off for a client that is not Claude Code', async () => {
627
+ await post('/v1/messages', { model: 'claude-opus-4-6', max_tokens: 10, messages: [{ role: 'user', content: 'hi' }] },
628
+ { 'x-llm-profile': 'intactcc', 'user-agent': 'some-sdk/1.0' });
629
+ assert.equal(received.at(-1).url, '/intact/v1/chat/completions');
630
+ });
631
+
580
632
  test('Healer: orphaned tool_result and missing tool_result produce a valid chat history', async () => {
581
633
  await post('/v1/messages', {
582
634
  model: 'claude-opus-4-6', max_tokens: 50, messages: [
package/tests/helpers.mjs CHANGED
@@ -1,24 +1,24 @@
1
- import assert from 'node:assert/strict';
2
-
3
- // Validate an Anthropic event sequence: indexes increase in start order, deltas only target an open block.
4
- export function assertValidAnthropicEvents(events) {
5
- const open = new Set();
6
- const seen = new Set();
7
- let expectedNext = 0;
8
- for (const { event, data } of events) {
9
- if (event === 'content_block_start') {
10
- assert.equal(data.index, expectedNext, `block index must be sequential (got ${data.index}, want ${expectedNext})`);
11
- assert.ok(!seen.has(data.index), `index ${data.index} reused`);
12
- seen.add(data.index);
13
- open.add(data.index);
14
- expectedNext++;
15
- } else if (event === 'content_block_delta') {
16
- assert.ok(open.has(data.index), `delta for block ${data.index} which is not open`);
17
- } else if (event === 'content_block_stop') {
18
- assert.ok(open.has(data.index), `stop for block ${data.index} which is not open`);
19
- open.delete(data.index);
20
- } else if (event === 'message_stop') {
21
- assert.equal(open.size, 0, 'all blocks must be closed before message_stop');
22
- }
23
- }
24
- }
1
+ import assert from 'node:assert/strict';
2
+
3
+ // Validate an Anthropic event sequence: indexes increase in start order, deltas only target an open block.
4
+ export function assertValidAnthropicEvents(events) {
5
+ const open = new Set();
6
+ const seen = new Set();
7
+ let expectedNext = 0;
8
+ for (const { event, data } of events) {
9
+ if (event === 'content_block_start') {
10
+ assert.equal(data.index, expectedNext, `block index must be sequential (got ${data.index}, want ${expectedNext})`);
11
+ assert.ok(!seen.has(data.index), `index ${data.index} reused`);
12
+ seen.add(data.index);
13
+ open.add(data.index);
14
+ expectedNext++;
15
+ } else if (event === 'content_block_delta') {
16
+ assert.ok(open.has(data.index), `delta for block ${data.index} which is not open`);
17
+ } else if (event === 'content_block_stop') {
18
+ assert.ok(open.has(data.index), `stop for block ${data.index} which is not open`);
19
+ open.delete(data.index);
20
+ } else if (event === 'message_stop') {
21
+ assert.equal(open.size, 0, 'all blocks must be closed before message_stop');
22
+ }
23
+ }
24
+ }