llm-switcher 1.1.11 → 1.2.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (61) hide show
  1. package/CHANGELOG.md +63 -0
  2. package/README.md +202 -257
  3. package/README.vi.md +200 -256
  4. package/blindfold/blindfold.mjs +200 -53
  5. package/blindfold/make-certs.sh +26 -7
  6. package/catalog.mjs +246 -0
  7. package/classifier.mjs +238 -0
  8. package/config.example.json +12 -34
  9. package/docs/codex-blindfold.md +28 -17
  10. package/docs/cross-platform.md +16 -7
  11. package/docs/diagrams/ir-healer-pipeline.mmd +16 -0
  12. package/docs/diagrams/ir-healer-pipeline.png +0 -0
  13. package/docs/diagrams/ir-healer-pipeline.svg +90 -0
  14. package/docs/diagrams/ir-translation-pipeline.html +14925 -0
  15. package/docs/diagrams/ir-translation-pipeline.sequence.json +31 -0
  16. package/docs/diagrams/ir-translation-pipeline.svg +5128 -0
  17. package/docs/diagrams/system-architecture.architecture.json +76 -0
  18. package/docs/diagrams/system-architecture.html +14978 -0
  19. package/docs/diagrams/system-architecture.svg +5147 -0
  20. package/docs/diagrams/system-topology.mmd +30 -0
  21. package/docs/diagrams/system-topology.png +0 -0
  22. package/docs/diagrams/system-topology.svg +125 -0
  23. package/ensure-ca-bundle.mjs +28 -0
  24. package/formats.mjs +43 -156
  25. package/icons/antigravity.png +0 -0
  26. package/icons/claude.png +0 -0
  27. package/icons/codex.png +0 -0
  28. package/icons/deepseek.png +0 -0
  29. package/icons/gemini.png +0 -0
  30. package/icons/github.png +0 -0
  31. package/icons/groq.png +0 -0
  32. package/icons/intact.svg +1 -0
  33. package/icons/ollama.png +0 -0
  34. package/icons/openai.png +0 -0
  35. package/icons/openrouter.png +0 -0
  36. package/icons/qwen.png +0 -0
  37. package/icons/vertex.png +0 -0
  38. package/mcp.mjs +39 -11
  39. package/package.json +1 -1
  40. package/proxy.mjs +114 -37
  41. package/shim.mjs +200 -57
  42. package/skills/llm-switcher/SKILL.md +15 -10
  43. package/state.mjs +1100 -191
  44. package/switch.cmd +2 -2
  45. package/switch.mjs +228 -53
  46. package/tests/blindfold-e2e.test.mjs +380 -0
  47. package/tests/blindfold-task5.test.mjs +429 -0
  48. package/tests/blindfold.task3.test.mjs +700 -0
  49. package/tests/blindfold.test.mjs +10 -5
  50. package/tests/catalog.test.mjs +147 -0
  51. package/tests/classifier.test.mjs +210 -0
  52. package/tests/contract-lab.test.mjs +22 -7
  53. package/tests/formats.test.mjs +63 -46
  54. package/tests/gateway.e2e.test.mjs +136 -36
  55. package/tests/lifecycle.test.mjs +16 -10
  56. package/tests/mcp.test.mjs +78 -2
  57. package/tests/real-user-sim.test.mjs +464 -0
  58. package/tests/shim.test.mjs +159 -66
  59. package/tests/state.test.mjs +975 -193
  60. package/tests/switch.test.mjs +446 -2
  61. package/ui.html +1710 -1726
@@ -50,11 +50,16 @@ test('a real Codex API path is routed, and its query string survives', () => {
50
50
  assert.equal(toGatewayPath('/backend-api/codex/responses'), `${GATEWAY_PREFIX}/responses`);
51
51
  });
52
52
 
53
- // A CONNECT to any other host must be tunneled, not intercepted: the process then
54
- // only copies bytes and never holds that host's plaintext.
55
- test('only the target host is intercepted; every other public host is tunneled', () => {
56
- assert.equal(isInterceptedHost('chatgpt.com'), true);
57
- for (const other of ['api.openai.com', 'auth.openai.com', 'example.com', 'chatgpt.com.evil.test']) {
53
+ // A CONNECT to any host outside the table must be tunneled, not intercepted: the process then
54
+ // only copies bytes and never holds that host's plaintext. The table of R3 names exactly three
55
+ // hosts, and it is an exact lookup — a name that merely ends in one of them is not in it.
56
+ test('the host table names exactly three hosts; every other public host is tunneled', () => {
57
+ for (const host of ['api.anthropic.com', 'api.openai.com', 'chatgpt.com']) {
58
+ assert.equal(isInterceptedHost(host), true, `must intercept ${host}`);
59
+ assert.equal(isInterceptedHost(host.toUpperCase()), true, 'DNS case never matters');
60
+ }
61
+ for (const other of ['auth.openai.com', 'api.chatgpt.com', 'openai.com', 'example.com',
62
+ 'chatgpt.com.evil.test', 'anthropic.com']) {
58
63
  assert.equal(isInterceptedHost(other), false, `must not intercept ${other}`);
59
64
  }
60
65
  });
@@ -0,0 +1,147 @@
1
+ // Unit tests for catalog.mjs (dynamic model discovery, caching and auto-role classification)
2
+ import test from 'node:test';
3
+ import assert from 'node:assert/strict';
4
+ import fs from 'node:fs';
5
+ import os from 'node:os';
6
+ import path from 'node:path';
7
+ import http from 'node:http';
8
+ import {
9
+ classifyClaudeTier,
10
+ classifyCodexRole,
11
+ loadCatalogCache,
12
+ saveCatalogCache,
13
+ fetchToolModels,
14
+ refreshCatalog,
15
+ detectToolVersion,
16
+ checkVersionAndRefresh,
17
+ BASELINE_MODELS
18
+ } from '../catalog.mjs';
19
+
20
+ test('classifyClaudeTier categorizes all Claude models into correct tiers', () => {
21
+ assert.equal(classifyClaudeTier('claude-opus-5-5'), 'opus');
22
+ assert.equal(classifyClaudeTier('claude-opus-4-6'), 'opus');
23
+ assert.equal(classifyClaudeTier('claude-3-opus-20240229'), 'opus');
24
+
25
+ assert.equal(classifyClaudeTier('claude-sonnet-4'), 'sonnet');
26
+ assert.equal(classifyClaudeTier('claude-3-7-sonnet-20250219'), 'sonnet');
27
+ assert.equal(classifyClaudeTier('claude-3-5-sonnet-20241022'), 'sonnet');
28
+
29
+ assert.equal(classifyClaudeTier('claude-haiku-4'), 'haiku');
30
+ assert.equal(classifyClaudeTier('claude-3-5-haiku-20241022'), 'haiku');
31
+
32
+ assert.equal(classifyClaudeTier('claude-fable-4'), 'fable');
33
+ assert.equal(classifyClaudeTier('custom-model'), 'sonnet', 'defaults unknown Claude models to sonnet tier');
34
+ });
35
+
36
+ test('classifyCodexRole categorizes all Codex models into correct roles', () => {
37
+ assert.equal(classifyCodexRole('gpt-6-sol'), 'main');
38
+ assert.equal(classifyCodexRole('gpt-5.6-sol'), 'main');
39
+ assert.equal(classifyCodexRole('gpt-5.2'), 'main');
40
+ assert.equal(classifyCodexRole('o3'), 'main');
41
+
42
+ assert.equal(classifyCodexRole('gpt-6-terra'), 'review');
43
+ assert.equal(classifyCodexRole('gpt-5.6-terra'), 'review');
44
+ assert.equal(classifyCodexRole('codex-review-model'), 'review');
45
+
46
+ assert.equal(classifyCodexRole('gpt-6-luna'), 'subagent');
47
+ assert.equal(classifyCodexRole('gpt-5.6-luna'), 'subagent');
48
+ assert.equal(classifyCodexRole('codex-subagent-worker'), 'subagent');
49
+ });
50
+
51
+ test('loadCatalogCache returns baseline models when cache file does not exist', () => {
52
+ const tmp = fs.mkdtempSync(path.join(os.tmpdir(), 'llmsw-cat-test-'));
53
+ try {
54
+ const catalog = loadCatalogCache(tmp);
55
+ assert.ok(catalog.claude.models.length > 0);
56
+ assert.ok(catalog.codex.models.length > 0);
57
+ assert.ok(catalog.claude.models.some(m => m.id === 'claude-opus-5-5'));
58
+ assert.ok(catalog.codex.models.some(m => m.id === 'gpt-6-sol'));
59
+ } finally {
60
+ fs.rmSync(tmp, { recursive: true, force: true });
61
+ }
62
+ });
63
+
64
+ test('saveCatalogCache writes cache atomically and loadCatalogCache reads it back', () => {
65
+ const tmp = fs.mkdtempSync(path.join(os.tmpdir(), 'llmsw-cat-test-'));
66
+ try {
67
+ const sample = {
68
+ updatedAt: 123456789,
69
+ claude: { models: [{ id: 'claude-future-opus', tier: 'opus' }] },
70
+ codex: { models: [{ id: 'gpt-7-sol', role: 'main' }] }
71
+ };
72
+ const saved = saveCatalogCache(tmp, sample);
73
+ assert.equal(saved, true);
74
+ const loaded = loadCatalogCache(tmp);
75
+ assert.equal(loaded.updatedAt, 123456789);
76
+ assert.equal(loaded.claude.models[0].id, 'claude-future-opus');
77
+ assert.equal(loaded.codex.models[0].id, 'gpt-7-sol');
78
+ } finally {
79
+ fs.rmSync(tmp, { recursive: true, force: true });
80
+ }
81
+ });
82
+
83
+ test('fetchToolModels queries endpoint and parses model list', async () => {
84
+ const server = http.createServer((req, res) => {
85
+ if (req.url === '/v1/models') {
86
+ res.writeHead(200, { 'Content-Type': 'application/json' });
87
+ res.end(JSON.stringify({
88
+ data: [
89
+ { id: 'gpt-6-sol', object: 'model' },
90
+ { id: 'gpt-6-terra', object: 'model' },
91
+ { id: 'gpt-6-luna', object: 'model' }
92
+ ]
93
+ }));
94
+ return;
95
+ }
96
+ res.writeHead(404);
97
+ res.end('{}');
98
+ });
99
+
100
+ const port = await new Promise(r => server.listen(0, '127.0.0.1', () => r(server.address().port)));
101
+ try {
102
+ const res = await fetchToolModels('codex', { url: `http://127.0.0.1:${port}/v1/models` });
103
+ assert.equal(res.ok, true);
104
+ assert.equal(res.models.length, 3);
105
+ assert.equal(res.models.find(m => m.id === 'gpt-6-sol').role, 'main');
106
+ assert.equal(res.models.find(m => m.id === 'gpt-6-terra').role, 'review');
107
+ assert.equal(res.models.find(m => m.id === 'gpt-6-luna').role, 'subagent');
108
+ } finally {
109
+ server.close();
110
+ }
111
+ });
112
+
113
+ test('fetchToolModels handles network errors gracefully without throwing', async () => {
114
+ const res = await fetchToolModels('claude', { url: 'http://127.0.0.1:1/nonexistent', timeout: 50 });
115
+ assert.equal(res.ok, false);
116
+ assert.ok(res.error, 'reports error reason without crashing');
117
+ });
118
+
119
+ test('detectToolVersion extracts semantic version from various User-Agent strings', () => {
120
+ assert.equal(detectToolVersion({ 'user-agent': 'claude-cli/2.1.280 (external, cli)' }), '2.1.280');
121
+ assert.equal(detectToolVersion({ 'user-agent': 'codex-cli/0.157.1 (Windows NT 10.0; Win64; x64)' }), '0.157.1');
122
+ assert.equal(detectToolVersion({ 'user-agent': 'claude-code/2.2.0 darwin' }), '2.2.0');
123
+ assert.equal(detectToolVersion({ 'user-agent': 'curl/7.68.0' }), '');
124
+ assert.equal(detectToolVersion({}), '');
125
+ });
126
+
127
+ test('checkVersionAndRefresh records new version and triggers update only on version bumps', async () => {
128
+ const tmp = fs.mkdtempSync(path.join(os.tmpdir(), 'llmsw-bump-test-'));
129
+ try {
130
+ // Initial request from version 1.1.0
131
+ checkVersionAndRefresh('claude', { 'user-agent': 'claude-cli/1.1.0' }, tmp);
132
+ const cat1 = loadCatalogCache(tmp);
133
+ assert.equal(cat1.claude.lastSeenVersion, '1.1.0');
134
+
135
+ // Repeated request from the same version 1.1.0 -> no change, no duplicate trigger
136
+ checkVersionAndRefresh('claude', { 'user-agent': 'claude-cli/1.1.0' }, tmp);
137
+ const cat2 = loadCatalogCache(tmp);
138
+ assert.equal(cat2.claude.lastSeenVersion, '1.1.0');
139
+
140
+ // Tool updates to version 1.2.0 -> version bump detected and saved
141
+ checkVersionAndRefresh('claude', { 'user-agent': 'claude-cli/1.2.0' }, tmp);
142
+ const cat3 = loadCatalogCache(tmp);
143
+ assert.equal(cat3.claude.lastSeenVersion, '1.2.0');
144
+ } finally {
145
+ fs.rmSync(tmp, { recursive: true, force: true });
146
+ }
147
+ });
@@ -0,0 +1,210 @@
1
+ import { describe, it } from 'node:test';
2
+ import assert from 'node:assert/strict';
3
+ import http from 'node:http';
4
+ import {
5
+ heuristicClassify,
6
+ findJevKey,
7
+ classifyPrompt,
8
+ checkSemanticEquivalence,
9
+ DEFAULT_JEV_URL,
10
+ OPENROUTER_JEV_URL,
11
+ } from '../classifier.mjs';
12
+
13
+ function createServer(handler) {
14
+ const server = http.createServer(handler);
15
+ return new Promise((resolve, reject) => {
16
+ server.listen(0, '127.0.0.1', () => {
17
+ const addr = server.address();
18
+ resolve({
19
+ url: `http://127.0.0.1:${addr.port}`,
20
+ close: () => new Promise((res) => server.close(res)),
21
+ });
22
+ });
23
+ server.on('error', reject);
24
+ });
25
+ }
26
+
27
+ describe('classifier module', () => {
28
+ describe('heuristicClassify', () => {
29
+ it('classifies complex tasks to opus', () => {
30
+ assert.equal(heuristicClassify('Design the system architecture for high throughput'), 'opus');
31
+ assert.equal(heuristicClassify('Find race condition in concurrent worker pool'), 'opus');
32
+ assert.equal(heuristicClassify('Conduct a security audit of our auth boundary'), 'opus');
33
+ });
34
+
35
+ it('classifies simple short tasks to haiku', () => {
36
+ assert.equal(heuristicClassify('Fix typo in comment'), 'haiku');
37
+ assert.equal(heuristicClassify('Format json output to string'), 'haiku');
38
+ assert.equal(heuristicClassify('Extract date from this line'), 'haiku');
39
+ });
40
+
41
+ it('defaults general programming tasks to sonnet', () => {
42
+ assert.equal(heuristicClassify('Refactor this helper function to accept options object and write tests'), 'sonnet');
43
+ });
44
+ });
45
+
46
+ describe('findJevKey', () => {
47
+ it('resolves TYPESAFE_API_KEY first', () => {
48
+ const res = findJevKey({ TYPESAFE_API_KEY: 'ts_key', JEV_API_KEY: 'jev_key' });
49
+ assert.deepEqual(res, { key: 'ts_key', source: 'typesafe' });
50
+ });
51
+
52
+ it('resolves JEV_API_KEY when TYPESAFE_API_KEY is absent', () => {
53
+ const res = findJevKey({ JEV_API_KEY: 'jev_key' });
54
+ assert.deepEqual(res, { key: 'jev_key', source: 'openrouter' });
55
+ });
56
+
57
+ it('returns null on empty env', () => {
58
+ assert.equal(findJevKey({}), null);
59
+ });
60
+ });
61
+
62
+ describe('classifyPrompt', () => {
63
+ it('returns heuristic tier with reason no-key when no key is present', async () => {
64
+ const res = await classifyPrompt({
65
+ prompt: 'Design architecture for database',
66
+ env: {},
67
+ });
68
+ assert.equal(res.tier, 'opus');
69
+ assert.equal(res.source, 'heuristic');
70
+ assert.equal(res.reason, 'no-key');
71
+ });
72
+
73
+ it('calls Jev endpoint and parses choice response', async () => {
74
+ let receivedAuth = null;
75
+ let receivedBody = null;
76
+
77
+ const server = await createServer(async (req, res) => {
78
+ receivedAuth = req.headers.authorization;
79
+ const chunks = [];
80
+ for await (const chunk of req) chunks.push(chunk);
81
+ receivedBody = JSON.parse(Buffer.concat(chunks).toString('utf8'));
82
+
83
+ res.writeHead(200, { 'Content-Type': 'application/json' });
84
+ res.end(
85
+ JSON.stringify({
86
+ model: 'jev-latest',
87
+ answers: {
88
+ recommended_tier: {
89
+ choice: 'haiku',
90
+ confidence: 0.94,
91
+ distribution: { haiku: 0.94, sonnet: 0.05, opus: 0.01 },
92
+ },
93
+ },
94
+ }),
95
+ );
96
+ });
97
+
98
+ try {
99
+ const res = await classifyPrompt({
100
+ prompt: 'Translate hello to French',
101
+ apiKey: 'test-ts-key',
102
+ url: server.url,
103
+ });
104
+
105
+ assert.equal(res.tier, 'haiku');
106
+ assert.equal(res.confidence, 0.94);
107
+ assert.equal(res.source, 'jev');
108
+ assert.equal(res.model, 'jev-latest');
109
+ assert.equal(receivedAuth, 'Bearer test-ts-key');
110
+ assert.equal(receivedBody.state, 'Translate hello to French');
111
+ assert.ok('recommended_tier' in receivedBody.questions);
112
+ } finally {
113
+ await server.close();
114
+ }
115
+ });
116
+
117
+ it('falls back gracefully to heuristic on HTTP error', async () => {
118
+ const server = await createServer((req, res) => {
119
+ res.writeHead(500, { 'Content-Type': 'application/json' });
120
+ res.end(JSON.stringify({ error: 'internal error' }));
121
+ });
122
+
123
+ try {
124
+ const res = await classifyPrompt({
125
+ prompt: 'Fix typo',
126
+ apiKey: 'key',
127
+ url: server.url,
128
+ });
129
+
130
+ assert.equal(res.tier, 'haiku');
131
+ assert.equal(res.source, 'heuristic');
132
+ assert.equal(res.reason, 'http-500');
133
+ } finally {
134
+ await server.close();
135
+ }
136
+ });
137
+
138
+ it('falls back gracefully to heuristic on timeout', async () => {
139
+ const server = await createServer((req, res) => {
140
+ // Hang indefinitely
141
+ });
142
+
143
+ try {
144
+ const res = await classifyPrompt({
145
+ prompt: 'Fix typo',
146
+ apiKey: 'key',
147
+ url: server.url,
148
+ timeoutMs: 50,
149
+ });
150
+
151
+ assert.equal(res.tier, 'haiku');
152
+ assert.equal(res.source, 'heuristic');
153
+ assert.equal(res.reason, 'timeout');
154
+ } finally {
155
+ await server.close();
156
+ }
157
+ });
158
+ });
159
+
160
+ describe('checkSemanticEquivalence', () => {
161
+ it('returns exact-match immediately when strings match', async () => {
162
+ const res = await checkSemanticEquivalence({
163
+ candidate: 'What is Node.js?',
164
+ target: 'What is Node.js?',
165
+ });
166
+ assert.equal(res.equivalent, true);
167
+ assert.equal(res.confidence, 1.0);
168
+ assert.equal(res.reason, 'exact-match');
169
+ });
170
+
171
+ it('bypasses immediately with false when no key is configured', async () => {
172
+ const res = await checkSemanticEquivalence({
173
+ candidate: 'Say hello',
174
+ target: 'Greet me',
175
+ env: {},
176
+ });
177
+ assert.equal(res.equivalent, false);
178
+ assert.equal(res.reason, 'no-key-fallback-bypass');
179
+ });
180
+
181
+ it('evaluates Noul probability from Jev and confirms equivalence >= 0.85', async () => {
182
+ const server = await createServer(async (req, res) => {
183
+ res.writeHead(200, { 'Content-Type': 'application/json' });
184
+ res.end(
185
+ JSON.stringify({
186
+ model: 'jev-latest',
187
+ answers: {
188
+ is_equivalent: 0.92,
189
+ },
190
+ }),
191
+ );
192
+ });
193
+
194
+ try {
195
+ const res = await checkSemanticEquivalence({
196
+ candidate: 'How do I read a file in Node?',
197
+ target: 'Read file contents in NodeJS',
198
+ apiKey: 'test-key',
199
+ url: server.url,
200
+ });
201
+
202
+ assert.equal(res.equivalent, true);
203
+ assert.equal(res.confidence, 0.92);
204
+ assert.equal(res.reason, 'jev-confirmed');
205
+ } finally {
206
+ await server.close();
207
+ }
208
+ });
209
+ });
210
+ });
@@ -10,7 +10,7 @@ import fs from 'node:fs';
10
10
  import os from 'node:os';
11
11
  import path from 'node:path';
12
12
  import { spawn, execFileSync } from 'node:child_process';
13
- import { fileURLToPath } from 'node:url';
13
+ import { fileURLToPath, pathToFileURL } from 'node:url';
14
14
  import { createFrameReader } from '../blindfold/wsframe.mjs';
15
15
  import {
16
16
  createContractLab, createHalfTap, newTraceId, switcherVersion, toolVersionFromUA, capJson,
@@ -85,7 +85,7 @@ test('contractLab is absent by default, and a saved block survives a rewrite', (
85
85
  fs.writeFileSync(cfgPath, JSON.stringify(cfg));
86
86
  const out = JSON.parse(spawnNodeSync(`
87
87
  import fs from 'node:fs';
88
- import { loadConfig, saveConfig, contractLabSettings, redactConfig } from '${ROOT}/state.mjs';
88
+ import { loadConfig, saveConfig, contractLabSettings, redactConfig } from '${pathToFileURL(path.join(ROOT, 'state.mjs')).href}';
89
89
  saveConfig({ ...loadConfig(), debug: true });
90
90
  const again = JSON.parse(fs.readFileSync(process.env.LLM_SWITCHER_CONFIG, 'utf8'));
91
91
  console.log(JSON.stringify({ settings: contractLabSettings(again), masked: redactConfig(again).contractLab }));
@@ -312,11 +312,13 @@ function startIntact() {
312
312
 
313
313
  async function startProxy(name, contractLab) {
314
314
  const port = await freePort();
315
+ const bfPort = await freePort();
315
316
  const dir = path.join(tmpDir, name);
316
317
  fs.mkdirSync(dir, { recursive: true });
317
318
  const models = { opus: 'up-opus', sonnet: 'up-sonnet', haiku: 'up-haiku', fable: 'up-fable' };
318
319
  const cfg = {
319
320
  port,
321
+ blindfold: { port: bfPort },
320
322
  activeProfiles: { anthropic: 'ant', responses: 'ant', 'openai-chat': 'ant', vertex: 'ant' },
321
323
  profiles: {
322
324
  ant: { name: 'Mock Anthropic', mode: 'direct', inFormat: 'auto', outFormat: 'anthropic', baseURL: `http://127.0.0.1:${upstreamPort}/ant`, apiKey: 'sk-secret-ant', defaultModels: models }
@@ -365,7 +367,16 @@ before(async () => {
365
367
  });
366
368
 
367
369
  after(() => {
368
- for (const p of Object.values(proxies)) p.child?.kill();
370
+ for (const [name, p] of Object.entries(proxies)) {
371
+ try {
372
+ const bfState = JSON.parse(fs.readFileSync(path.join(tmpDir, name, 'blindfold.json'), 'utf8'));
373
+ if (bfState?.pid) {
374
+ if (process.platform === 'win32') execFileSync('taskkill', ['/F', '/PID', String(bfState.pid)], { stdio: 'ignore' });
375
+ else process.kill(bfState.pid, 'SIGTERM');
376
+ }
377
+ } catch {}
378
+ p.child?.kill();
379
+ }
369
380
  upstream?.close();
370
381
  intact?.close();
371
382
  if (tmpDir) fs.rmSync(tmpDir, { recursive: true, force: true });
@@ -459,10 +470,12 @@ test('failed exchanges upload no half: 400 upstream and mid-stream error are ign
459
470
  });
460
471
 
461
472
  test('a converted stream cut mid-way uploads no half, a complete one does', async () => {
462
- const chat = (text) => fetch(`http://127.0.0.1:${proxies.on.port}/v1/chat/completions`, {
473
+ // R5: /v1/chat/completions no longer exists. A converted stream is now an Anthropic request
474
+ // whose upstream speaks the other protocol, which is the same conversion by another door.
475
+ const chat = (text) => fetch(`http://127.0.0.1:${proxies.on.port}/v1/messages`, {
463
476
  method: 'POST',
464
477
  headers: { 'Content-Type': 'application/json', 'user-agent': 'codex/1.0.0' },
465
- body: JSON.stringify({ model: 'claude-opus-4-6', stream: true, messages: [{ role: 'user', content: text }] })
478
+ body: JSON.stringify({ model: 'claude-opus-4-6', max_tokens: 16, stream: true, messages: [{ role: 'user', content: text }] })
466
479
  });
467
480
  let traceOk;
468
481
  await waitFor(async () => {
@@ -636,10 +649,12 @@ test('Anthropic probe with thinking sets max_tokens greater than budget_tokens',
636
649
 
637
650
  test('the probe models are every mapped model of the active profiles, once each', () => {
638
651
  const cfg = {
639
- activeProfiles: { anthropic: 'a', responses: 'a', 'openai-chat': 'b', vertex: null },
652
+ // R5: `openai-chat` and `vertex` are retired pointer keys. They name a real profile here on
653
+ // purpose - a retired key must not activate anything, so `never-probed` has to stay unprobed.
654
+ activeProfiles: { claude: 'a', codex: 'b', 'openai-chat': 'unused', vertex: null },
640
655
  profiles: {
641
656
  a: { inFormat: 'auto', defaultModels: { opus: 'up-opus', sonnet: 'up-sonnet', haiku: '', fable: 'up-opus' } },
642
- b: { inFormat: 'openai-chat', defaultModels: { default: 'up-chat' } },
657
+ b: { inFormat: 'auto', defaultModels: { opus: 'up-chat' } },
643
658
  unused: { inFormat: 'auto', defaultModels: { opus: 'never-probed' } }
644
659
  }
645
660
  };
@@ -2,7 +2,7 @@
2
2
  import { test } from 'node:test';
3
3
  import assert from 'node:assert/strict';
4
4
  import {
5
- anthropicToIR, chatToIR, responsesToIR, vertexToIR,
5
+ anthropicToIR, responsesToIR,
6
6
  healToolPairs, irToChatBody, irToAnthropicBody, irToVertexBody, toGeminiSchema,
7
7
  createUpstreamNormalizer, createCollector, createThinkTagSplitter,
8
8
  createAnthropicStream, createResponsesStream, createVertexStream,
@@ -62,11 +62,6 @@ test('healAnthropicPayload: native Anthropic passthrough keeps the billing heade
62
62
  assert.equal(payload.system[0].text, header);
63
63
  });
64
64
 
65
- test('chatToIR: reasoning_effort "none" disables thinking', () => {
66
- const ir = chatToIR({ model: 'x', messages: [{ role: 'user', content: 'hi' }], reasoning_effort: 'none' });
67
- assert.equal(ir.thinking.type, 'disabled');
68
- });
69
-
70
65
  test('healToolPairs: orphan result -> user text, missing result -> placeholder, adjacency kept', () => {
71
66
  const healed = healToolPairs([
72
67
  { role: 'user', content: 'start' },
@@ -118,21 +113,6 @@ test('responsesToIR: parallel function_call items merge into one assistant turn;
118
113
  assert.equal(messages[2].tool_calls.length, 2);
119
114
  });
120
115
 
121
- test('vertexToIR: functionCall/functionResponse are paired by generated ids', () => {
122
- const ir = vertexToIR({
123
- contents: [
124
- { role: 'user', parts: [{ text: 'weather?' }] },
125
- { role: 'model', parts: [{ functionCall: { name: 'w', args: { c: 'A' } } }, { functionCall: { name: 'w', args: { c: 'B' } } }] },
126
- { role: 'function', parts: [{ functionResponse: { name: 'w', response: { t: 1 } } }, { functionResponse: { name: 'w', response: { t: 2 } } }] }
127
- ]
128
- });
129
- const calls = ir.messages[1].toolCalls.map(t => t.id);
130
- const results = ir.messages.filter(m => m.role === 'tool').map(m => m.toolCallId);
131
- assert.deepEqual(results, calls);
132
- const { messages } = irToChatBody(ir, 'gpt-4o');
133
- assert.equal(messages.filter(m => m.role === 'tool').length, 2);
134
- });
135
-
136
116
  test('irToVertexBody: functionResponse uses the function name and parallel responses share one content', () => {
137
117
  const ir = anthropicToIR({
138
118
  model: 'x', messages: [
@@ -164,19 +144,19 @@ test('toGeminiSchema strips unsupported JSON Schema keywords', () => {
164
144
  });
165
145
 
166
146
  test('irToAnthropicBody: thinking constraints (budget < max_tokens, no top_k, temperature 1)', () => {
167
- const ir = chatToIR({ model: 'x', messages: [{ role: 'user', content: 'hi' }], max_tokens: 3000, temperature: 0.2, reasoning_effort: 'high' });
147
+ const ir = anthropicToIR({ model: 'x', max_tokens: 3000, temperature: 0.2, messages: [{ role: 'user', content: 'hi' }], thinking: { type: 'enabled', budget_tokens: 2048 } });
168
148
  ir.params.topK = 5;
169
149
  const body = irToAnthropicBody(ir, 'claude-opus-4-6');
170
150
  assert.ok(body.thinking.budget_tokens < body.max_tokens);
171
151
  assert.equal(body.temperature, undefined);
172
152
  assert.equal(body.top_k, undefined);
173
153
 
174
- const small = irToAnthropicBody(chatToIR({ model: 'x', messages: [{ role: 'user', content: 'hi' }], max_tokens: 100, reasoning_effort: 'high' }), 'claude');
154
+ const small = irToAnthropicBody(anthropicToIR({ model: 'x', max_tokens: 100, messages: [{ role: 'user', content: 'hi' }], thinking: { type: 'enabled', budget_tokens: 2048 } }), 'claude');
175
155
  assert.equal(small.thinking, undefined, 'thinking dropped when max_tokens <= 1024');
176
156
  });
177
157
 
178
158
  test('irToAnthropicBody: never emits empty text blocks', () => {
179
- const ir = chatToIR({ model: 'x', messages: [{ role: 'user', content: [{ type: 'text', text: '' }, { type: 'text', text: 'hi' }] }] });
159
+ const ir = anthropicToIR({ model: 'x', max_tokens: 10, messages: [{ role: 'user', content: [{ type: 'text', text: '' }, { type: 'text', text: 'hi' }] }] });
180
160
  const body = irToAnthropicBody(ir, 'claude');
181
161
  assert.ok(body.messages.every(m => m.content.every(b => b.type !== 'text' || b.text)));
182
162
  });
@@ -699,26 +679,16 @@ test('Responses tool_choice: allowed_tools restricts the tools and keeps its mod
699
679
  assert.deepEqual(responsesToIR({ ...base, tool_choice: { type: 'function', name: 'b' } }).toolChoice, { name: 'b' });
700
680
  });
701
681
 
702
- test('Chat tool_choice: allowed_tools restricts the tools and keeps its mode', () => {
703
- const tools = ['a', 'b'].map(name => ({ type: 'function', function: { name, parameters: { type: 'object', properties: {} } } }));
704
- const ir = chatToIR({ model: 'm', messages: [{ role: 'user', content: 'x' }], tools,
705
- tool_choice: { type: 'allowed_tools', allowed_tools: { mode: 'auto', tools: [{ type: 'function', function: { name: 'b' } }] } } });
706
- assert.deepEqual(ir.tools.map(t => t.name), ['b']);
707
- assert.equal(ir.toolChoice, 'auto');
708
- });
709
-
710
682
  test('smartText and smartReasoning return each string once', () => {
711
683
  assert.equal(smartText({ content: ['line 1', 'line 2'] }), 'line 1\nline 2');
712
684
  assert.equal(smartReasoning({ reasoning_content: 'step 1', parts: [{ text: 'step 1', thought: true }] }).text, 'step 1');
713
685
  });
714
686
 
715
- test('an empty stop string is dropped in every input format', () => {
687
+ test('an empty stop string is dropped, and stop reaches the body only when it has values', () => {
716
688
  const msgs = [{ role: 'user', content: 'x' }];
717
- assert.deepEqual(chatToIR({ model: 'm', messages: msgs, stop: '' }).params.stop, []);
718
- assert.deepEqual(chatToIR({ model: 'm', messages: msgs, stop: ['', 'END'] }).params.stop, ['END']);
689
+ assert.deepEqual(anthropicToIR({ model: 'm', max_tokens: 10, messages: msgs, stop_sequences: [''] }).params.stop, []);
719
690
  assert.deepEqual(anthropicToIR({ model: 'm', max_tokens: 10, messages: msgs, stop_sequences: ['', 'END'] }).params.stop, ['END']);
720
- assert.deepEqual(vertexToIR({ contents: [{ role: 'user', parts: [{ text: 'x' }] }], generationConfig: { stopSequences: [''] } }).params.stop, []);
721
- const body = irToAnthropicBody(chatToIR({ model: 'm', messages: msgs, stop: '' }), 'claude-x');
691
+ const body = irToAnthropicBody(anthropicToIR({ model: 'm', max_tokens: 10, messages: msgs, stop_sequences: [''] }), 'claude-x');
722
692
  assert.equal(body.stop_sequences, undefined);
723
693
  });
724
694
 
@@ -736,7 +706,7 @@ test('Vertex usage: thoughts count as output, and the Vertex reply splits them b
736
706
  });
737
707
 
738
708
  test('reasoning model families get no <think> guide', () => {
739
- const ir = chatToIR({ model: 'm', messages: [{ role: 'user', content: 'x' }], reasoning_effort: 'high' });
709
+ const ir = anthropicToIR({ model: 'm', max_tokens: 100, messages: [{ role: 'user', content: 'x' }], thinking: { type: 'enabled', budget_tokens: 2048 } });
740
710
  for (const model of ['o3-mini', 'openai/o1', 'o4-mini-high', 'deepseek-r1', 'deepseek-reasoner', 'qwq-32b']) {
741
711
  const sys = irToChatBody(ir, model).messages.find(m => m.role === 'system')?.content || '';
742
712
  assert.ok(!sys.includes('<think>'), `${model} must not get the guide`);
@@ -781,15 +751,10 @@ test('healAnthropicPayload keeps an unchanged user turn as the same object', ()
781
751
  });
782
752
 
783
753
  test('allowed_tools that matches no declared tool gives a request without tools, not a crash', () => {
784
- const chat = chatToIR({ model: 'm', messages: [{ role: 'user', content: 'x' }],
785
- tools: [{ type: 'function', function: { name: 'a', parameters: { type: 'object', properties: {} } } }],
786
- tool_choice: { type: 'allowed_tools', allowed_tools: { mode: 'auto', tools: [{ type: 'function', function: { name: 'b' } }] } } });
787
754
  const hosted = responsesToIR({ model: 'm', input: 'x', tools: [fn('a')], tool_choice: { type: 'allowed_tools', mode: 'required', tools: [{ type: 'web_search' }] } });
788
- for (const ir of [chat, hosted]) {
789
- for (const body of [irToChatBody(ir, 'm'), irToAnthropicBody(ir, 'claude-x'), irToVertexBody(ir, 'gemini-x')]) {
790
- assert.equal(body.tools, undefined);
791
- assert.equal(body.tool_choice, undefined);
792
- }
755
+ for (const body of [irToChatBody(hosted, 'm'), irToAnthropicBody(hosted, 'claude-x'), irToVertexBody(hosted, 'gemini-x')]) {
756
+ assert.equal(body.tools, undefined);
757
+ assert.equal(body.tool_choice, undefined);
793
758
  }
794
759
  });
795
760
 
@@ -815,3 +780,55 @@ test('tool schemas keep 0, false, "" and null values', () => {
815
780
  assert.equal(p.properties.tags.items.default, '');
816
781
  assert.equal(p.properties.note.default, null);
817
782
  });
783
+
784
+ // Claude Code sends `advisor` with `input_schema: {}`. That survived as `parameters: {}`, so
785
+ // Vertex failed the whole request with `tools.18.custom.input_schema.type: Field required` —
786
+ // while the same call with a real tool list succeeded. An empty shape must be healed exactly
787
+ // like a missing one, on the chat path a profile actually uses.
788
+ test('an empty tool schema is healed, not forwarded as {}', () => {
789
+ assert.deepEqual(sanitizeJsonSchema({}), { type: 'object', properties: {} });
790
+ assert.deepEqual(
791
+ sanitizeJsonSchema({ type: 'object', properties: { nested: {} } }),
792
+ { type: 'object', properties: { nested: { type: 'object', properties: {} } } }
793
+ );
794
+ assert.deepEqual(toGeminiSchema({}), { type: 'object', properties: {} });
795
+
796
+ const body = irToChatBody(
797
+ anthropicToIR({
798
+ model: 'm', max_tokens: 10, messages: [{ role: 'user', content: 'x' }],
799
+ tools: [{ name: 'advisor', description: '', input_schema: {} }]
800
+ }),
801
+ 'antigravity/claude-opus-4-6-thinking'
802
+ );
803
+ assert.deepEqual(body.tools[0].function.parameters, { type: 'object', properties: {} });
804
+ });
805
+
806
+ // Structured output: a JSON schema the client asks the answer to follow must reach every upstream.
807
+ const TITLE_SCHEMA = { type: 'object', properties: { title: { type: 'string' } }, required: ['title'], additionalProperties: false };
808
+
809
+ test('structured output: Anthropic output_config.format reaches chat, Anthropic and Vertex upstreams', () => {
810
+ const ir = anthropicToIR({ model: 'x', max_tokens: 100, messages: [{ role: 'user', content: 'hi' }], tools: [],
811
+ output_config: { format: { type: 'json_schema', schema: TITLE_SCHEMA } } });
812
+ assert.deepEqual(irToChatBody(ir, 'm').response_format,
813
+ { type: 'json_schema', json_schema: { name: 'response', schema: TITLE_SCHEMA } });
814
+ assert.deepEqual(irToAnthropicBody(ir, 'm').output_config, { format: { type: 'json_schema', schema: TITLE_SCHEMA } });
815
+ const gc = irToVertexBody(ir, 'm').generationConfig;
816
+ assert.equal(gc.responseMimeType, 'application/json');
817
+ assert.deepEqual(gc.responseSchema, toGeminiSchema(TITLE_SCHEMA));
818
+ // The deprecated top-level field still works.
819
+ const old = anthropicToIR({ model: 'x', messages: [{ role: 'user', content: 'hi' }], output_format: { type: 'json_schema', schema: TITLE_SCHEMA } });
820
+ assert.equal(irToChatBody(old, 'm').response_format.json_schema.schema, TITLE_SCHEMA);
821
+ });
822
+
823
+ test('structured output: Responses text.format keeps name and strict; JSON mode and plain text', () => {
824
+ const resp = responsesToIR({ model: 'x', input: 'hi', text: { format: { type: 'json_schema', name: 'title', schema: TITLE_SCHEMA, strict: true } } });
825
+ assert.deepEqual(irToChatBody(resp, 'm').response_format,
826
+ { type: 'json_schema', json_schema: { name: 'title', schema: TITLE_SCHEMA, strict: true } });
827
+ const jm = responsesToIR({ model: 'x', input: 'hi', text: { format: { type: 'json_object' } } });
828
+ assert.deepEqual(irToChatBody(jm, 'm').response_format, { type: 'json_object' });
829
+ assert.equal(irToVertexBody(jm, 'm').generationConfig.responseMimeType, 'application/json');
830
+ assert.equal(irToAnthropicBody(jm, 'm').output_config, undefined);
831
+ const txt = responsesToIR({ model: 'x', input: 'hi', text: { format: { type: 'text' } } });
832
+ assert.equal(irToChatBody(txt, 'm').response_format, undefined);
833
+ assert.equal(irToVertexBody(txt, 'm').generationConfig, undefined);
834
+ });