llm-switcher 1.1.11 → 1.2.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +63 -0
- package/README.md +202 -257
- package/README.vi.md +200 -256
- package/blindfold/blindfold.mjs +200 -53
- package/blindfold/make-certs.sh +26 -7
- package/catalog.mjs +246 -0
- package/classifier.mjs +238 -0
- package/config.example.json +12 -34
- package/docs/codex-blindfold.md +28 -17
- package/docs/cross-platform.md +16 -7
- package/docs/diagrams/ir-healer-pipeline.mmd +16 -0
- package/docs/diagrams/ir-healer-pipeline.png +0 -0
- package/docs/diagrams/ir-healer-pipeline.svg +90 -0
- package/docs/diagrams/ir-translation-pipeline.html +14925 -0
- package/docs/diagrams/ir-translation-pipeline.sequence.json +31 -0
- package/docs/diagrams/ir-translation-pipeline.svg +5128 -0
- package/docs/diagrams/system-architecture.architecture.json +76 -0
- package/docs/diagrams/system-architecture.html +14978 -0
- package/docs/diagrams/system-architecture.svg +5147 -0
- package/docs/diagrams/system-topology.mmd +30 -0
- package/docs/diagrams/system-topology.png +0 -0
- package/docs/diagrams/system-topology.svg +125 -0
- package/ensure-ca-bundle.mjs +28 -0
- package/formats.mjs +43 -156
- package/icons/antigravity.png +0 -0
- package/icons/claude.png +0 -0
- package/icons/codex.png +0 -0
- package/icons/deepseek.png +0 -0
- package/icons/gemini.png +0 -0
- package/icons/github.png +0 -0
- package/icons/groq.png +0 -0
- package/icons/intact.svg +1 -0
- package/icons/ollama.png +0 -0
- package/icons/openai.png +0 -0
- package/icons/openrouter.png +0 -0
- package/icons/qwen.png +0 -0
- package/icons/vertex.png +0 -0
- package/mcp.mjs +39 -11
- package/package.json +1 -1
- package/proxy.mjs +114 -37
- package/shim.mjs +200 -57
- package/skills/llm-switcher/SKILL.md +15 -10
- package/state.mjs +1100 -191
- package/switch.cmd +2 -2
- package/switch.mjs +228 -53
- package/tests/blindfold-e2e.test.mjs +380 -0
- package/tests/blindfold-task5.test.mjs +429 -0
- package/tests/blindfold.task3.test.mjs +700 -0
- package/tests/blindfold.test.mjs +10 -5
- package/tests/catalog.test.mjs +147 -0
- package/tests/classifier.test.mjs +210 -0
- package/tests/contract-lab.test.mjs +22 -7
- package/tests/formats.test.mjs +63 -46
- package/tests/gateway.e2e.test.mjs +136 -36
- package/tests/lifecycle.test.mjs +16 -10
- package/tests/mcp.test.mjs +78 -2
- package/tests/real-user-sim.test.mjs +464 -0
- package/tests/shim.test.mjs +159 -66
- package/tests/state.test.mjs +975 -193
- package/tests/switch.test.mjs +446 -2
- package/ui.html +1710 -1726
package/tests/blindfold.test.mjs
CHANGED
|
@@ -50,11 +50,16 @@ test('a real Codex API path is routed, and its query string survives', () => {
|
|
|
50
50
|
assert.equal(toGatewayPath('/backend-api/codex/responses'), `${GATEWAY_PREFIX}/responses`);
|
|
51
51
|
});
|
|
52
52
|
|
|
53
|
-
// A CONNECT to any
|
|
54
|
-
// only copies bytes and never holds that host's plaintext.
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
for (const
|
|
53
|
+
// A CONNECT to any host outside the table must be tunneled, not intercepted: the process then
|
|
54
|
+
// only copies bytes and never holds that host's plaintext. The table of R3 names exactly three
|
|
55
|
+
// hosts, and it is an exact lookup — a name that merely ends in one of them is not in it.
|
|
56
|
+
test('the host table names exactly three hosts; every other public host is tunneled', () => {
|
|
57
|
+
for (const host of ['api.anthropic.com', 'api.openai.com', 'chatgpt.com']) {
|
|
58
|
+
assert.equal(isInterceptedHost(host), true, `must intercept ${host}`);
|
|
59
|
+
assert.equal(isInterceptedHost(host.toUpperCase()), true, 'DNS case never matters');
|
|
60
|
+
}
|
|
61
|
+
for (const other of ['auth.openai.com', 'api.chatgpt.com', 'openai.com', 'example.com',
|
|
62
|
+
'chatgpt.com.evil.test', 'anthropic.com']) {
|
|
58
63
|
assert.equal(isInterceptedHost(other), false, `must not intercept ${other}`);
|
|
59
64
|
}
|
|
60
65
|
});
|
|
@@ -0,0 +1,147 @@
|
|
|
1
|
+
// Unit tests for catalog.mjs (dynamic model discovery, caching and auto-role classification)
|
|
2
|
+
import test from 'node:test';
|
|
3
|
+
import assert from 'node:assert/strict';
|
|
4
|
+
import fs from 'node:fs';
|
|
5
|
+
import os from 'node:os';
|
|
6
|
+
import path from 'node:path';
|
|
7
|
+
import http from 'node:http';
|
|
8
|
+
import {
|
|
9
|
+
classifyClaudeTier,
|
|
10
|
+
classifyCodexRole,
|
|
11
|
+
loadCatalogCache,
|
|
12
|
+
saveCatalogCache,
|
|
13
|
+
fetchToolModels,
|
|
14
|
+
refreshCatalog,
|
|
15
|
+
detectToolVersion,
|
|
16
|
+
checkVersionAndRefresh,
|
|
17
|
+
BASELINE_MODELS
|
|
18
|
+
} from '../catalog.mjs';
|
|
19
|
+
|
|
20
|
+
test('classifyClaudeTier categorizes all Claude models into correct tiers', () => {
|
|
21
|
+
assert.equal(classifyClaudeTier('claude-opus-5-5'), 'opus');
|
|
22
|
+
assert.equal(classifyClaudeTier('claude-opus-4-6'), 'opus');
|
|
23
|
+
assert.equal(classifyClaudeTier('claude-3-opus-20240229'), 'opus');
|
|
24
|
+
|
|
25
|
+
assert.equal(classifyClaudeTier('claude-sonnet-4'), 'sonnet');
|
|
26
|
+
assert.equal(classifyClaudeTier('claude-3-7-sonnet-20250219'), 'sonnet');
|
|
27
|
+
assert.equal(classifyClaudeTier('claude-3-5-sonnet-20241022'), 'sonnet');
|
|
28
|
+
|
|
29
|
+
assert.equal(classifyClaudeTier('claude-haiku-4'), 'haiku');
|
|
30
|
+
assert.equal(classifyClaudeTier('claude-3-5-haiku-20241022'), 'haiku');
|
|
31
|
+
|
|
32
|
+
assert.equal(classifyClaudeTier('claude-fable-4'), 'fable');
|
|
33
|
+
assert.equal(classifyClaudeTier('custom-model'), 'sonnet', 'defaults unknown Claude models to sonnet tier');
|
|
34
|
+
});
|
|
35
|
+
|
|
36
|
+
test('classifyCodexRole categorizes all Codex models into correct roles', () => {
|
|
37
|
+
assert.equal(classifyCodexRole('gpt-6-sol'), 'main');
|
|
38
|
+
assert.equal(classifyCodexRole('gpt-5.6-sol'), 'main');
|
|
39
|
+
assert.equal(classifyCodexRole('gpt-5.2'), 'main');
|
|
40
|
+
assert.equal(classifyCodexRole('o3'), 'main');
|
|
41
|
+
|
|
42
|
+
assert.equal(classifyCodexRole('gpt-6-terra'), 'review');
|
|
43
|
+
assert.equal(classifyCodexRole('gpt-5.6-terra'), 'review');
|
|
44
|
+
assert.equal(classifyCodexRole('codex-review-model'), 'review');
|
|
45
|
+
|
|
46
|
+
assert.equal(classifyCodexRole('gpt-6-luna'), 'subagent');
|
|
47
|
+
assert.equal(classifyCodexRole('gpt-5.6-luna'), 'subagent');
|
|
48
|
+
assert.equal(classifyCodexRole('codex-subagent-worker'), 'subagent');
|
|
49
|
+
});
|
|
50
|
+
|
|
51
|
+
test('loadCatalogCache returns baseline models when cache file does not exist', () => {
|
|
52
|
+
const tmp = fs.mkdtempSync(path.join(os.tmpdir(), 'llmsw-cat-test-'));
|
|
53
|
+
try {
|
|
54
|
+
const catalog = loadCatalogCache(tmp);
|
|
55
|
+
assert.ok(catalog.claude.models.length > 0);
|
|
56
|
+
assert.ok(catalog.codex.models.length > 0);
|
|
57
|
+
assert.ok(catalog.claude.models.some(m => m.id === 'claude-opus-5-5'));
|
|
58
|
+
assert.ok(catalog.codex.models.some(m => m.id === 'gpt-6-sol'));
|
|
59
|
+
} finally {
|
|
60
|
+
fs.rmSync(tmp, { recursive: true, force: true });
|
|
61
|
+
}
|
|
62
|
+
});
|
|
63
|
+
|
|
64
|
+
test('saveCatalogCache writes cache atomically and loadCatalogCache reads it back', () => {
|
|
65
|
+
const tmp = fs.mkdtempSync(path.join(os.tmpdir(), 'llmsw-cat-test-'));
|
|
66
|
+
try {
|
|
67
|
+
const sample = {
|
|
68
|
+
updatedAt: 123456789,
|
|
69
|
+
claude: { models: [{ id: 'claude-future-opus', tier: 'opus' }] },
|
|
70
|
+
codex: { models: [{ id: 'gpt-7-sol', role: 'main' }] }
|
|
71
|
+
};
|
|
72
|
+
const saved = saveCatalogCache(tmp, sample);
|
|
73
|
+
assert.equal(saved, true);
|
|
74
|
+
const loaded = loadCatalogCache(tmp);
|
|
75
|
+
assert.equal(loaded.updatedAt, 123456789);
|
|
76
|
+
assert.equal(loaded.claude.models[0].id, 'claude-future-opus');
|
|
77
|
+
assert.equal(loaded.codex.models[0].id, 'gpt-7-sol');
|
|
78
|
+
} finally {
|
|
79
|
+
fs.rmSync(tmp, { recursive: true, force: true });
|
|
80
|
+
}
|
|
81
|
+
});
|
|
82
|
+
|
|
83
|
+
test('fetchToolModels queries endpoint and parses model list', async () => {
|
|
84
|
+
const server = http.createServer((req, res) => {
|
|
85
|
+
if (req.url === '/v1/models') {
|
|
86
|
+
res.writeHead(200, { 'Content-Type': 'application/json' });
|
|
87
|
+
res.end(JSON.stringify({
|
|
88
|
+
data: [
|
|
89
|
+
{ id: 'gpt-6-sol', object: 'model' },
|
|
90
|
+
{ id: 'gpt-6-terra', object: 'model' },
|
|
91
|
+
{ id: 'gpt-6-luna', object: 'model' }
|
|
92
|
+
]
|
|
93
|
+
}));
|
|
94
|
+
return;
|
|
95
|
+
}
|
|
96
|
+
res.writeHead(404);
|
|
97
|
+
res.end('{}');
|
|
98
|
+
});
|
|
99
|
+
|
|
100
|
+
const port = await new Promise(r => server.listen(0, '127.0.0.1', () => r(server.address().port)));
|
|
101
|
+
try {
|
|
102
|
+
const res = await fetchToolModels('codex', { url: `http://127.0.0.1:${port}/v1/models` });
|
|
103
|
+
assert.equal(res.ok, true);
|
|
104
|
+
assert.equal(res.models.length, 3);
|
|
105
|
+
assert.equal(res.models.find(m => m.id === 'gpt-6-sol').role, 'main');
|
|
106
|
+
assert.equal(res.models.find(m => m.id === 'gpt-6-terra').role, 'review');
|
|
107
|
+
assert.equal(res.models.find(m => m.id === 'gpt-6-luna').role, 'subagent');
|
|
108
|
+
} finally {
|
|
109
|
+
server.close();
|
|
110
|
+
}
|
|
111
|
+
});
|
|
112
|
+
|
|
113
|
+
test('fetchToolModels handles network errors gracefully without throwing', async () => {
|
|
114
|
+
const res = await fetchToolModels('claude', { url: 'http://127.0.0.1:1/nonexistent', timeout: 50 });
|
|
115
|
+
assert.equal(res.ok, false);
|
|
116
|
+
assert.ok(res.error, 'reports error reason without crashing');
|
|
117
|
+
});
|
|
118
|
+
|
|
119
|
+
test('detectToolVersion extracts semantic version from various User-Agent strings', () => {
|
|
120
|
+
assert.equal(detectToolVersion({ 'user-agent': 'claude-cli/2.1.280 (external, cli)' }), '2.1.280');
|
|
121
|
+
assert.equal(detectToolVersion({ 'user-agent': 'codex-cli/0.157.1 (Windows NT 10.0; Win64; x64)' }), '0.157.1');
|
|
122
|
+
assert.equal(detectToolVersion({ 'user-agent': 'claude-code/2.2.0 darwin' }), '2.2.0');
|
|
123
|
+
assert.equal(detectToolVersion({ 'user-agent': 'curl/7.68.0' }), '');
|
|
124
|
+
assert.equal(detectToolVersion({}), '');
|
|
125
|
+
});
|
|
126
|
+
|
|
127
|
+
test('checkVersionAndRefresh records new version and triggers update only on version bumps', async () => {
|
|
128
|
+
const tmp = fs.mkdtempSync(path.join(os.tmpdir(), 'llmsw-bump-test-'));
|
|
129
|
+
try {
|
|
130
|
+
// Initial request from version 1.1.0
|
|
131
|
+
checkVersionAndRefresh('claude', { 'user-agent': 'claude-cli/1.1.0' }, tmp);
|
|
132
|
+
const cat1 = loadCatalogCache(tmp);
|
|
133
|
+
assert.equal(cat1.claude.lastSeenVersion, '1.1.0');
|
|
134
|
+
|
|
135
|
+
// Repeated request from the same version 1.1.0 -> no change, no duplicate trigger
|
|
136
|
+
checkVersionAndRefresh('claude', { 'user-agent': 'claude-cli/1.1.0' }, tmp);
|
|
137
|
+
const cat2 = loadCatalogCache(tmp);
|
|
138
|
+
assert.equal(cat2.claude.lastSeenVersion, '1.1.0');
|
|
139
|
+
|
|
140
|
+
// Tool updates to version 1.2.0 -> version bump detected and saved
|
|
141
|
+
checkVersionAndRefresh('claude', { 'user-agent': 'claude-cli/1.2.0' }, tmp);
|
|
142
|
+
const cat3 = loadCatalogCache(tmp);
|
|
143
|
+
assert.equal(cat3.claude.lastSeenVersion, '1.2.0');
|
|
144
|
+
} finally {
|
|
145
|
+
fs.rmSync(tmp, { recursive: true, force: true });
|
|
146
|
+
}
|
|
147
|
+
});
|
|
@@ -0,0 +1,210 @@
|
|
|
1
|
+
import { describe, it } from 'node:test';
|
|
2
|
+
import assert from 'node:assert/strict';
|
|
3
|
+
import http from 'node:http';
|
|
4
|
+
import {
|
|
5
|
+
heuristicClassify,
|
|
6
|
+
findJevKey,
|
|
7
|
+
classifyPrompt,
|
|
8
|
+
checkSemanticEquivalence,
|
|
9
|
+
DEFAULT_JEV_URL,
|
|
10
|
+
OPENROUTER_JEV_URL,
|
|
11
|
+
} from '../classifier.mjs';
|
|
12
|
+
|
|
13
|
+
function createServer(handler) {
|
|
14
|
+
const server = http.createServer(handler);
|
|
15
|
+
return new Promise((resolve, reject) => {
|
|
16
|
+
server.listen(0, '127.0.0.1', () => {
|
|
17
|
+
const addr = server.address();
|
|
18
|
+
resolve({
|
|
19
|
+
url: `http://127.0.0.1:${addr.port}`,
|
|
20
|
+
close: () => new Promise((res) => server.close(res)),
|
|
21
|
+
});
|
|
22
|
+
});
|
|
23
|
+
server.on('error', reject);
|
|
24
|
+
});
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
describe('classifier module', () => {
|
|
28
|
+
describe('heuristicClassify', () => {
|
|
29
|
+
it('classifies complex tasks to opus', () => {
|
|
30
|
+
assert.equal(heuristicClassify('Design the system architecture for high throughput'), 'opus');
|
|
31
|
+
assert.equal(heuristicClassify('Find race condition in concurrent worker pool'), 'opus');
|
|
32
|
+
assert.equal(heuristicClassify('Conduct a security audit of our auth boundary'), 'opus');
|
|
33
|
+
});
|
|
34
|
+
|
|
35
|
+
it('classifies simple short tasks to haiku', () => {
|
|
36
|
+
assert.equal(heuristicClassify('Fix typo in comment'), 'haiku');
|
|
37
|
+
assert.equal(heuristicClassify('Format json output to string'), 'haiku');
|
|
38
|
+
assert.equal(heuristicClassify('Extract date from this line'), 'haiku');
|
|
39
|
+
});
|
|
40
|
+
|
|
41
|
+
it('defaults general programming tasks to sonnet', () => {
|
|
42
|
+
assert.equal(heuristicClassify('Refactor this helper function to accept options object and write tests'), 'sonnet');
|
|
43
|
+
});
|
|
44
|
+
});
|
|
45
|
+
|
|
46
|
+
describe('findJevKey', () => {
|
|
47
|
+
it('resolves TYPESAFE_API_KEY first', () => {
|
|
48
|
+
const res = findJevKey({ TYPESAFE_API_KEY: 'ts_key', JEV_API_KEY: 'jev_key' });
|
|
49
|
+
assert.deepEqual(res, { key: 'ts_key', source: 'typesafe' });
|
|
50
|
+
});
|
|
51
|
+
|
|
52
|
+
it('resolves JEV_API_KEY when TYPESAFE_API_KEY is absent', () => {
|
|
53
|
+
const res = findJevKey({ JEV_API_KEY: 'jev_key' });
|
|
54
|
+
assert.deepEqual(res, { key: 'jev_key', source: 'openrouter' });
|
|
55
|
+
});
|
|
56
|
+
|
|
57
|
+
it('returns null on empty env', () => {
|
|
58
|
+
assert.equal(findJevKey({}), null);
|
|
59
|
+
});
|
|
60
|
+
});
|
|
61
|
+
|
|
62
|
+
describe('classifyPrompt', () => {
|
|
63
|
+
it('returns heuristic tier with reason no-key when no key is present', async () => {
|
|
64
|
+
const res = await classifyPrompt({
|
|
65
|
+
prompt: 'Design architecture for database',
|
|
66
|
+
env: {},
|
|
67
|
+
});
|
|
68
|
+
assert.equal(res.tier, 'opus');
|
|
69
|
+
assert.equal(res.source, 'heuristic');
|
|
70
|
+
assert.equal(res.reason, 'no-key');
|
|
71
|
+
});
|
|
72
|
+
|
|
73
|
+
it('calls Jev endpoint and parses choice response', async () => {
|
|
74
|
+
let receivedAuth = null;
|
|
75
|
+
let receivedBody = null;
|
|
76
|
+
|
|
77
|
+
const server = await createServer(async (req, res) => {
|
|
78
|
+
receivedAuth = req.headers.authorization;
|
|
79
|
+
const chunks = [];
|
|
80
|
+
for await (const chunk of req) chunks.push(chunk);
|
|
81
|
+
receivedBody = JSON.parse(Buffer.concat(chunks).toString('utf8'));
|
|
82
|
+
|
|
83
|
+
res.writeHead(200, { 'Content-Type': 'application/json' });
|
|
84
|
+
res.end(
|
|
85
|
+
JSON.stringify({
|
|
86
|
+
model: 'jev-latest',
|
|
87
|
+
answers: {
|
|
88
|
+
recommended_tier: {
|
|
89
|
+
choice: 'haiku',
|
|
90
|
+
confidence: 0.94,
|
|
91
|
+
distribution: { haiku: 0.94, sonnet: 0.05, opus: 0.01 },
|
|
92
|
+
},
|
|
93
|
+
},
|
|
94
|
+
}),
|
|
95
|
+
);
|
|
96
|
+
});
|
|
97
|
+
|
|
98
|
+
try {
|
|
99
|
+
const res = await classifyPrompt({
|
|
100
|
+
prompt: 'Translate hello to French',
|
|
101
|
+
apiKey: 'test-ts-key',
|
|
102
|
+
url: server.url,
|
|
103
|
+
});
|
|
104
|
+
|
|
105
|
+
assert.equal(res.tier, 'haiku');
|
|
106
|
+
assert.equal(res.confidence, 0.94);
|
|
107
|
+
assert.equal(res.source, 'jev');
|
|
108
|
+
assert.equal(res.model, 'jev-latest');
|
|
109
|
+
assert.equal(receivedAuth, 'Bearer test-ts-key');
|
|
110
|
+
assert.equal(receivedBody.state, 'Translate hello to French');
|
|
111
|
+
assert.ok('recommended_tier' in receivedBody.questions);
|
|
112
|
+
} finally {
|
|
113
|
+
await server.close();
|
|
114
|
+
}
|
|
115
|
+
});
|
|
116
|
+
|
|
117
|
+
it('falls back gracefully to heuristic on HTTP error', async () => {
|
|
118
|
+
const server = await createServer((req, res) => {
|
|
119
|
+
res.writeHead(500, { 'Content-Type': 'application/json' });
|
|
120
|
+
res.end(JSON.stringify({ error: 'internal error' }));
|
|
121
|
+
});
|
|
122
|
+
|
|
123
|
+
try {
|
|
124
|
+
const res = await classifyPrompt({
|
|
125
|
+
prompt: 'Fix typo',
|
|
126
|
+
apiKey: 'key',
|
|
127
|
+
url: server.url,
|
|
128
|
+
});
|
|
129
|
+
|
|
130
|
+
assert.equal(res.tier, 'haiku');
|
|
131
|
+
assert.equal(res.source, 'heuristic');
|
|
132
|
+
assert.equal(res.reason, 'http-500');
|
|
133
|
+
} finally {
|
|
134
|
+
await server.close();
|
|
135
|
+
}
|
|
136
|
+
});
|
|
137
|
+
|
|
138
|
+
it('falls back gracefully to heuristic on timeout', async () => {
|
|
139
|
+
const server = await createServer((req, res) => {
|
|
140
|
+
// Hang indefinitely
|
|
141
|
+
});
|
|
142
|
+
|
|
143
|
+
try {
|
|
144
|
+
const res = await classifyPrompt({
|
|
145
|
+
prompt: 'Fix typo',
|
|
146
|
+
apiKey: 'key',
|
|
147
|
+
url: server.url,
|
|
148
|
+
timeoutMs: 50,
|
|
149
|
+
});
|
|
150
|
+
|
|
151
|
+
assert.equal(res.tier, 'haiku');
|
|
152
|
+
assert.equal(res.source, 'heuristic');
|
|
153
|
+
assert.equal(res.reason, 'timeout');
|
|
154
|
+
} finally {
|
|
155
|
+
await server.close();
|
|
156
|
+
}
|
|
157
|
+
});
|
|
158
|
+
});
|
|
159
|
+
|
|
160
|
+
describe('checkSemanticEquivalence', () => {
|
|
161
|
+
it('returns exact-match immediately when strings match', async () => {
|
|
162
|
+
const res = await checkSemanticEquivalence({
|
|
163
|
+
candidate: 'What is Node.js?',
|
|
164
|
+
target: 'What is Node.js?',
|
|
165
|
+
});
|
|
166
|
+
assert.equal(res.equivalent, true);
|
|
167
|
+
assert.equal(res.confidence, 1.0);
|
|
168
|
+
assert.equal(res.reason, 'exact-match');
|
|
169
|
+
});
|
|
170
|
+
|
|
171
|
+
it('bypasses immediately with false when no key is configured', async () => {
|
|
172
|
+
const res = await checkSemanticEquivalence({
|
|
173
|
+
candidate: 'Say hello',
|
|
174
|
+
target: 'Greet me',
|
|
175
|
+
env: {},
|
|
176
|
+
});
|
|
177
|
+
assert.equal(res.equivalent, false);
|
|
178
|
+
assert.equal(res.reason, 'no-key-fallback-bypass');
|
|
179
|
+
});
|
|
180
|
+
|
|
181
|
+
it('evaluates Noul probability from Jev and confirms equivalence >= 0.85', async () => {
|
|
182
|
+
const server = await createServer(async (req, res) => {
|
|
183
|
+
res.writeHead(200, { 'Content-Type': 'application/json' });
|
|
184
|
+
res.end(
|
|
185
|
+
JSON.stringify({
|
|
186
|
+
model: 'jev-latest',
|
|
187
|
+
answers: {
|
|
188
|
+
is_equivalent: 0.92,
|
|
189
|
+
},
|
|
190
|
+
}),
|
|
191
|
+
);
|
|
192
|
+
});
|
|
193
|
+
|
|
194
|
+
try {
|
|
195
|
+
const res = await checkSemanticEquivalence({
|
|
196
|
+
candidate: 'How do I read a file in Node?',
|
|
197
|
+
target: 'Read file contents in NodeJS',
|
|
198
|
+
apiKey: 'test-key',
|
|
199
|
+
url: server.url,
|
|
200
|
+
});
|
|
201
|
+
|
|
202
|
+
assert.equal(res.equivalent, true);
|
|
203
|
+
assert.equal(res.confidence, 0.92);
|
|
204
|
+
assert.equal(res.reason, 'jev-confirmed');
|
|
205
|
+
} finally {
|
|
206
|
+
await server.close();
|
|
207
|
+
}
|
|
208
|
+
});
|
|
209
|
+
});
|
|
210
|
+
});
|
|
@@ -10,7 +10,7 @@ import fs from 'node:fs';
|
|
|
10
10
|
import os from 'node:os';
|
|
11
11
|
import path from 'node:path';
|
|
12
12
|
import { spawn, execFileSync } from 'node:child_process';
|
|
13
|
-
import { fileURLToPath } from 'node:url';
|
|
13
|
+
import { fileURLToPath, pathToFileURL } from 'node:url';
|
|
14
14
|
import { createFrameReader } from '../blindfold/wsframe.mjs';
|
|
15
15
|
import {
|
|
16
16
|
createContractLab, createHalfTap, newTraceId, switcherVersion, toolVersionFromUA, capJson,
|
|
@@ -85,7 +85,7 @@ test('contractLab is absent by default, and a saved block survives a rewrite', (
|
|
|
85
85
|
fs.writeFileSync(cfgPath, JSON.stringify(cfg));
|
|
86
86
|
const out = JSON.parse(spawnNodeSync(`
|
|
87
87
|
import fs from 'node:fs';
|
|
88
|
-
import { loadConfig, saveConfig, contractLabSettings, redactConfig } from '${ROOT
|
|
88
|
+
import { loadConfig, saveConfig, contractLabSettings, redactConfig } from '${pathToFileURL(path.join(ROOT, 'state.mjs')).href}';
|
|
89
89
|
saveConfig({ ...loadConfig(), debug: true });
|
|
90
90
|
const again = JSON.parse(fs.readFileSync(process.env.LLM_SWITCHER_CONFIG, 'utf8'));
|
|
91
91
|
console.log(JSON.stringify({ settings: contractLabSettings(again), masked: redactConfig(again).contractLab }));
|
|
@@ -312,11 +312,13 @@ function startIntact() {
|
|
|
312
312
|
|
|
313
313
|
async function startProxy(name, contractLab) {
|
|
314
314
|
const port = await freePort();
|
|
315
|
+
const bfPort = await freePort();
|
|
315
316
|
const dir = path.join(tmpDir, name);
|
|
316
317
|
fs.mkdirSync(dir, { recursive: true });
|
|
317
318
|
const models = { opus: 'up-opus', sonnet: 'up-sonnet', haiku: 'up-haiku', fable: 'up-fable' };
|
|
318
319
|
const cfg = {
|
|
319
320
|
port,
|
|
321
|
+
blindfold: { port: bfPort },
|
|
320
322
|
activeProfiles: { anthropic: 'ant', responses: 'ant', 'openai-chat': 'ant', vertex: 'ant' },
|
|
321
323
|
profiles: {
|
|
322
324
|
ant: { name: 'Mock Anthropic', mode: 'direct', inFormat: 'auto', outFormat: 'anthropic', baseURL: `http://127.0.0.1:${upstreamPort}/ant`, apiKey: 'sk-secret-ant', defaultModels: models }
|
|
@@ -365,7 +367,16 @@ before(async () => {
|
|
|
365
367
|
});
|
|
366
368
|
|
|
367
369
|
after(() => {
|
|
368
|
-
for (const p of Object.
|
|
370
|
+
for (const [name, p] of Object.entries(proxies)) {
|
|
371
|
+
try {
|
|
372
|
+
const bfState = JSON.parse(fs.readFileSync(path.join(tmpDir, name, 'blindfold.json'), 'utf8'));
|
|
373
|
+
if (bfState?.pid) {
|
|
374
|
+
if (process.platform === 'win32') execFileSync('taskkill', ['/F', '/PID', String(bfState.pid)], { stdio: 'ignore' });
|
|
375
|
+
else process.kill(bfState.pid, 'SIGTERM');
|
|
376
|
+
}
|
|
377
|
+
} catch {}
|
|
378
|
+
p.child?.kill();
|
|
379
|
+
}
|
|
369
380
|
upstream?.close();
|
|
370
381
|
intact?.close();
|
|
371
382
|
if (tmpDir) fs.rmSync(tmpDir, { recursive: true, force: true });
|
|
@@ -459,10 +470,12 @@ test('failed exchanges upload no half: 400 upstream and mid-stream error are ign
|
|
|
459
470
|
});
|
|
460
471
|
|
|
461
472
|
test('a converted stream cut mid-way uploads no half, a complete one does', async () => {
|
|
462
|
-
|
|
473
|
+
// R5: /v1/chat/completions no longer exists. A converted stream is now an Anthropic request
|
|
474
|
+
// whose upstream speaks the other protocol, which is the same conversion by another door.
|
|
475
|
+
const chat = (text) => fetch(`http://127.0.0.1:${proxies.on.port}/v1/messages`, {
|
|
463
476
|
method: 'POST',
|
|
464
477
|
headers: { 'Content-Type': 'application/json', 'user-agent': 'codex/1.0.0' },
|
|
465
|
-
body: JSON.stringify({ model: 'claude-opus-4-6', stream: true, messages: [{ role: 'user', content: text }] })
|
|
478
|
+
body: JSON.stringify({ model: 'claude-opus-4-6', max_tokens: 16, stream: true, messages: [{ role: 'user', content: text }] })
|
|
466
479
|
});
|
|
467
480
|
let traceOk;
|
|
468
481
|
await waitFor(async () => {
|
|
@@ -636,10 +649,12 @@ test('Anthropic probe with thinking sets max_tokens greater than budget_tokens',
|
|
|
636
649
|
|
|
637
650
|
test('the probe models are every mapped model of the active profiles, once each', () => {
|
|
638
651
|
const cfg = {
|
|
639
|
-
|
|
652
|
+
// R5: `openai-chat` and `vertex` are retired pointer keys. They name a real profile here on
|
|
653
|
+
// purpose - a retired key must not activate anything, so `never-probed` has to stay unprobed.
|
|
654
|
+
activeProfiles: { claude: 'a', codex: 'b', 'openai-chat': 'unused', vertex: null },
|
|
640
655
|
profiles: {
|
|
641
656
|
a: { inFormat: 'auto', defaultModels: { opus: 'up-opus', sonnet: 'up-sonnet', haiku: '', fable: 'up-opus' } },
|
|
642
|
-
b: { inFormat: '
|
|
657
|
+
b: { inFormat: 'auto', defaultModels: { opus: 'up-chat' } },
|
|
643
658
|
unused: { inFormat: 'auto', defaultModels: { opus: 'never-probed' } }
|
|
644
659
|
}
|
|
645
660
|
};
|
package/tests/formats.test.mjs
CHANGED
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
import { test } from 'node:test';
|
|
3
3
|
import assert from 'node:assert/strict';
|
|
4
4
|
import {
|
|
5
|
-
anthropicToIR,
|
|
5
|
+
anthropicToIR, responsesToIR,
|
|
6
6
|
healToolPairs, irToChatBody, irToAnthropicBody, irToVertexBody, toGeminiSchema,
|
|
7
7
|
createUpstreamNormalizer, createCollector, createThinkTagSplitter,
|
|
8
8
|
createAnthropicStream, createResponsesStream, createVertexStream,
|
|
@@ -62,11 +62,6 @@ test('healAnthropicPayload: native Anthropic passthrough keeps the billing heade
|
|
|
62
62
|
assert.equal(payload.system[0].text, header);
|
|
63
63
|
});
|
|
64
64
|
|
|
65
|
-
test('chatToIR: reasoning_effort "none" disables thinking', () => {
|
|
66
|
-
const ir = chatToIR({ model: 'x', messages: [{ role: 'user', content: 'hi' }], reasoning_effort: 'none' });
|
|
67
|
-
assert.equal(ir.thinking.type, 'disabled');
|
|
68
|
-
});
|
|
69
|
-
|
|
70
65
|
test('healToolPairs: orphan result -> user text, missing result -> placeholder, adjacency kept', () => {
|
|
71
66
|
const healed = healToolPairs([
|
|
72
67
|
{ role: 'user', content: 'start' },
|
|
@@ -118,21 +113,6 @@ test('responsesToIR: parallel function_call items merge into one assistant turn;
|
|
|
118
113
|
assert.equal(messages[2].tool_calls.length, 2);
|
|
119
114
|
});
|
|
120
115
|
|
|
121
|
-
test('vertexToIR: functionCall/functionResponse are paired by generated ids', () => {
|
|
122
|
-
const ir = vertexToIR({
|
|
123
|
-
contents: [
|
|
124
|
-
{ role: 'user', parts: [{ text: 'weather?' }] },
|
|
125
|
-
{ role: 'model', parts: [{ functionCall: { name: 'w', args: { c: 'A' } } }, { functionCall: { name: 'w', args: { c: 'B' } } }] },
|
|
126
|
-
{ role: 'function', parts: [{ functionResponse: { name: 'w', response: { t: 1 } } }, { functionResponse: { name: 'w', response: { t: 2 } } }] }
|
|
127
|
-
]
|
|
128
|
-
});
|
|
129
|
-
const calls = ir.messages[1].toolCalls.map(t => t.id);
|
|
130
|
-
const results = ir.messages.filter(m => m.role === 'tool').map(m => m.toolCallId);
|
|
131
|
-
assert.deepEqual(results, calls);
|
|
132
|
-
const { messages } = irToChatBody(ir, 'gpt-4o');
|
|
133
|
-
assert.equal(messages.filter(m => m.role === 'tool').length, 2);
|
|
134
|
-
});
|
|
135
|
-
|
|
136
116
|
test('irToVertexBody: functionResponse uses the function name and parallel responses share one content', () => {
|
|
137
117
|
const ir = anthropicToIR({
|
|
138
118
|
model: 'x', messages: [
|
|
@@ -164,19 +144,19 @@ test('toGeminiSchema strips unsupported JSON Schema keywords', () => {
|
|
|
164
144
|
});
|
|
165
145
|
|
|
166
146
|
test('irToAnthropicBody: thinking constraints (budget < max_tokens, no top_k, temperature 1)', () => {
|
|
167
|
-
const ir =
|
|
147
|
+
const ir = anthropicToIR({ model: 'x', max_tokens: 3000, temperature: 0.2, messages: [{ role: 'user', content: 'hi' }], thinking: { type: 'enabled', budget_tokens: 2048 } });
|
|
168
148
|
ir.params.topK = 5;
|
|
169
149
|
const body = irToAnthropicBody(ir, 'claude-opus-4-6');
|
|
170
150
|
assert.ok(body.thinking.budget_tokens < body.max_tokens);
|
|
171
151
|
assert.equal(body.temperature, undefined);
|
|
172
152
|
assert.equal(body.top_k, undefined);
|
|
173
153
|
|
|
174
|
-
const small = irToAnthropicBody(
|
|
154
|
+
const small = irToAnthropicBody(anthropicToIR({ model: 'x', max_tokens: 100, messages: [{ role: 'user', content: 'hi' }], thinking: { type: 'enabled', budget_tokens: 2048 } }), 'claude');
|
|
175
155
|
assert.equal(small.thinking, undefined, 'thinking dropped when max_tokens <= 1024');
|
|
176
156
|
});
|
|
177
157
|
|
|
178
158
|
test('irToAnthropicBody: never emits empty text blocks', () => {
|
|
179
|
-
const ir =
|
|
159
|
+
const ir = anthropicToIR({ model: 'x', max_tokens: 10, messages: [{ role: 'user', content: [{ type: 'text', text: '' }, { type: 'text', text: 'hi' }] }] });
|
|
180
160
|
const body = irToAnthropicBody(ir, 'claude');
|
|
181
161
|
assert.ok(body.messages.every(m => m.content.every(b => b.type !== 'text' || b.text)));
|
|
182
162
|
});
|
|
@@ -699,26 +679,16 @@ test('Responses tool_choice: allowed_tools restricts the tools and keeps its mod
|
|
|
699
679
|
assert.deepEqual(responsesToIR({ ...base, tool_choice: { type: 'function', name: 'b' } }).toolChoice, { name: 'b' });
|
|
700
680
|
});
|
|
701
681
|
|
|
702
|
-
test('Chat tool_choice: allowed_tools restricts the tools and keeps its mode', () => {
|
|
703
|
-
const tools = ['a', 'b'].map(name => ({ type: 'function', function: { name, parameters: { type: 'object', properties: {} } } }));
|
|
704
|
-
const ir = chatToIR({ model: 'm', messages: [{ role: 'user', content: 'x' }], tools,
|
|
705
|
-
tool_choice: { type: 'allowed_tools', allowed_tools: { mode: 'auto', tools: [{ type: 'function', function: { name: 'b' } }] } } });
|
|
706
|
-
assert.deepEqual(ir.tools.map(t => t.name), ['b']);
|
|
707
|
-
assert.equal(ir.toolChoice, 'auto');
|
|
708
|
-
});
|
|
709
|
-
|
|
710
682
|
test('smartText and smartReasoning return each string once', () => {
|
|
711
683
|
assert.equal(smartText({ content: ['line 1', 'line 2'] }), 'line 1\nline 2');
|
|
712
684
|
assert.equal(smartReasoning({ reasoning_content: 'step 1', parts: [{ text: 'step 1', thought: true }] }).text, 'step 1');
|
|
713
685
|
});
|
|
714
686
|
|
|
715
|
-
test('an empty stop string is dropped
|
|
687
|
+
test('an empty stop string is dropped, and stop reaches the body only when it has values', () => {
|
|
716
688
|
const msgs = [{ role: 'user', content: 'x' }];
|
|
717
|
-
assert.deepEqual(
|
|
718
|
-
assert.deepEqual(chatToIR({ model: 'm', messages: msgs, stop: ['', 'END'] }).params.stop, ['END']);
|
|
689
|
+
assert.deepEqual(anthropicToIR({ model: 'm', max_tokens: 10, messages: msgs, stop_sequences: [''] }).params.stop, []);
|
|
719
690
|
assert.deepEqual(anthropicToIR({ model: 'm', max_tokens: 10, messages: msgs, stop_sequences: ['', 'END'] }).params.stop, ['END']);
|
|
720
|
-
|
|
721
|
-
const body = irToAnthropicBody(chatToIR({ model: 'm', messages: msgs, stop: '' }), 'claude-x');
|
|
691
|
+
const body = irToAnthropicBody(anthropicToIR({ model: 'm', max_tokens: 10, messages: msgs, stop_sequences: [''] }), 'claude-x');
|
|
722
692
|
assert.equal(body.stop_sequences, undefined);
|
|
723
693
|
});
|
|
724
694
|
|
|
@@ -736,7 +706,7 @@ test('Vertex usage: thoughts count as output, and the Vertex reply splits them b
|
|
|
736
706
|
});
|
|
737
707
|
|
|
738
708
|
test('reasoning model families get no <think> guide', () => {
|
|
739
|
-
const ir =
|
|
709
|
+
const ir = anthropicToIR({ model: 'm', max_tokens: 100, messages: [{ role: 'user', content: 'x' }], thinking: { type: 'enabled', budget_tokens: 2048 } });
|
|
740
710
|
for (const model of ['o3-mini', 'openai/o1', 'o4-mini-high', 'deepseek-r1', 'deepseek-reasoner', 'qwq-32b']) {
|
|
741
711
|
const sys = irToChatBody(ir, model).messages.find(m => m.role === 'system')?.content || '';
|
|
742
712
|
assert.ok(!sys.includes('<think>'), `${model} must not get the guide`);
|
|
@@ -781,15 +751,10 @@ test('healAnthropicPayload keeps an unchanged user turn as the same object', ()
|
|
|
781
751
|
});
|
|
782
752
|
|
|
783
753
|
test('allowed_tools that matches no declared tool gives a request without tools, not a crash', () => {
|
|
784
|
-
const chat = chatToIR({ model: 'm', messages: [{ role: 'user', content: 'x' }],
|
|
785
|
-
tools: [{ type: 'function', function: { name: 'a', parameters: { type: 'object', properties: {} } } }],
|
|
786
|
-
tool_choice: { type: 'allowed_tools', allowed_tools: { mode: 'auto', tools: [{ type: 'function', function: { name: 'b' } }] } } });
|
|
787
754
|
const hosted = responsesToIR({ model: 'm', input: 'x', tools: [fn('a')], tool_choice: { type: 'allowed_tools', mode: 'required', tools: [{ type: 'web_search' }] } });
|
|
788
|
-
for (const
|
|
789
|
-
|
|
790
|
-
|
|
791
|
-
assert.equal(body.tool_choice, undefined);
|
|
792
|
-
}
|
|
755
|
+
for (const body of [irToChatBody(hosted, 'm'), irToAnthropicBody(hosted, 'claude-x'), irToVertexBody(hosted, 'gemini-x')]) {
|
|
756
|
+
assert.equal(body.tools, undefined);
|
|
757
|
+
assert.equal(body.tool_choice, undefined);
|
|
793
758
|
}
|
|
794
759
|
});
|
|
795
760
|
|
|
@@ -815,3 +780,55 @@ test('tool schemas keep 0, false, "" and null values', () => {
|
|
|
815
780
|
assert.equal(p.properties.tags.items.default, '');
|
|
816
781
|
assert.equal(p.properties.note.default, null);
|
|
817
782
|
});
|
|
783
|
+
|
|
784
|
+
// Claude Code sends `advisor` with `input_schema: {}`. That survived as `parameters: {}`, so
|
|
785
|
+
// Vertex failed the whole request with `tools.18.custom.input_schema.type: Field required` —
|
|
786
|
+
// while the same call with a real tool list succeeded. An empty shape must be healed exactly
|
|
787
|
+
// like a missing one, on the chat path a profile actually uses.
|
|
788
|
+
test('an empty tool schema is healed, not forwarded as {}', () => {
|
|
789
|
+
assert.deepEqual(sanitizeJsonSchema({}), { type: 'object', properties: {} });
|
|
790
|
+
assert.deepEqual(
|
|
791
|
+
sanitizeJsonSchema({ type: 'object', properties: { nested: {} } }),
|
|
792
|
+
{ type: 'object', properties: { nested: { type: 'object', properties: {} } } }
|
|
793
|
+
);
|
|
794
|
+
assert.deepEqual(toGeminiSchema({}), { type: 'object', properties: {} });
|
|
795
|
+
|
|
796
|
+
const body = irToChatBody(
|
|
797
|
+
anthropicToIR({
|
|
798
|
+
model: 'm', max_tokens: 10, messages: [{ role: 'user', content: 'x' }],
|
|
799
|
+
tools: [{ name: 'advisor', description: '', input_schema: {} }]
|
|
800
|
+
}),
|
|
801
|
+
'antigravity/claude-opus-4-6-thinking'
|
|
802
|
+
);
|
|
803
|
+
assert.deepEqual(body.tools[0].function.parameters, { type: 'object', properties: {} });
|
|
804
|
+
});
|
|
805
|
+
|
|
806
|
+
// Structured output: a JSON schema the client asks the answer to follow must reach every upstream.
|
|
807
|
+
const TITLE_SCHEMA = { type: 'object', properties: { title: { type: 'string' } }, required: ['title'], additionalProperties: false };
|
|
808
|
+
|
|
809
|
+
test('structured output: Anthropic output_config.format reaches chat, Anthropic and Vertex upstreams', () => {
|
|
810
|
+
const ir = anthropicToIR({ model: 'x', max_tokens: 100, messages: [{ role: 'user', content: 'hi' }], tools: [],
|
|
811
|
+
output_config: { format: { type: 'json_schema', schema: TITLE_SCHEMA } } });
|
|
812
|
+
assert.deepEqual(irToChatBody(ir, 'm').response_format,
|
|
813
|
+
{ type: 'json_schema', json_schema: { name: 'response', schema: TITLE_SCHEMA } });
|
|
814
|
+
assert.deepEqual(irToAnthropicBody(ir, 'm').output_config, { format: { type: 'json_schema', schema: TITLE_SCHEMA } });
|
|
815
|
+
const gc = irToVertexBody(ir, 'm').generationConfig;
|
|
816
|
+
assert.equal(gc.responseMimeType, 'application/json');
|
|
817
|
+
assert.deepEqual(gc.responseSchema, toGeminiSchema(TITLE_SCHEMA));
|
|
818
|
+
// The deprecated top-level field still works.
|
|
819
|
+
const old = anthropicToIR({ model: 'x', messages: [{ role: 'user', content: 'hi' }], output_format: { type: 'json_schema', schema: TITLE_SCHEMA } });
|
|
820
|
+
assert.equal(irToChatBody(old, 'm').response_format.json_schema.schema, TITLE_SCHEMA);
|
|
821
|
+
});
|
|
822
|
+
|
|
823
|
+
test('structured output: Responses text.format keeps name and strict; JSON mode and plain text', () => {
|
|
824
|
+
const resp = responsesToIR({ model: 'x', input: 'hi', text: { format: { type: 'json_schema', name: 'title', schema: TITLE_SCHEMA, strict: true } } });
|
|
825
|
+
assert.deepEqual(irToChatBody(resp, 'm').response_format,
|
|
826
|
+
{ type: 'json_schema', json_schema: { name: 'title', schema: TITLE_SCHEMA, strict: true } });
|
|
827
|
+
const jm = responsesToIR({ model: 'x', input: 'hi', text: { format: { type: 'json_object' } } });
|
|
828
|
+
assert.deepEqual(irToChatBody(jm, 'm').response_format, { type: 'json_object' });
|
|
829
|
+
assert.equal(irToVertexBody(jm, 'm').generationConfig.responseMimeType, 'application/json');
|
|
830
|
+
assert.equal(irToAnthropicBody(jm, 'm').output_config, undefined);
|
|
831
|
+
const txt = responsesToIR({ model: 'x', input: 'hi', text: { format: { type: 'text' } } });
|
|
832
|
+
assert.equal(irToChatBody(txt, 'm').response_format, undefined);
|
|
833
|
+
assert.equal(irToVertexBody(txt, 'm').generationConfig, undefined);
|
|
834
|
+
});
|