claude-autorouter 0.3.6 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +12 -7
- package/CONTRIBUTING.md +37 -0
- package/README.md +43 -70
- package/bin/autorouter.mjs +40 -56
- package/docs/development.md +50 -2
- package/docs/hardware-benchmark.md +29 -0
- package/docs/hardware-comparison.md +55 -0
- package/docs/hardware-results-16gb.json +4002 -0
- package/docs/hardware-results-16gb.md +26 -0
- package/docs/hardware-results-64gb.json +4020 -0
- package/docs/reference.md +83 -40
- package/docs/releasing.md +76 -34
- package/docs/router-performance.json +1697 -0
- package/docs/router-performance.md +50 -0
- package/docs/status-performance.json +363 -0
- package/docs/status-performance.md +44 -0
- package/package.json +57 -9
- package/src/auto-routing.mjs +214 -0
- package/src/bounded-json.mjs +57 -0
- package/src/cli-help.mjs +87 -0
- package/src/config-command.mjs +141 -0
- package/src/config.mjs +52 -27
- package/src/contracts.mjs +123 -0
- package/src/evaluation-report.mjs +114 -0
- package/src/local-diagnostic.mjs +191 -0
- package/src/model-catalog.mjs +96 -0
- package/src/model-request.mjs +10 -6
- package/src/ollama-evaluator.mjs +9 -27
- package/src/onboarding.mjs +82 -23
- package/src/request-validation.mjs +54 -0
- package/src/response-observer.mjs +126 -18
- package/src/router.mjs +174 -70
- package/src/savings.mjs +74 -16
- package/src/server.mjs +79 -12
- package/src/session-history.mjs +261 -0
- package/src/session-log.mjs +9 -58
- package/src/status-state.mjs +110 -62
- package/src/statusline.mjs +57 -27
- package/src/telemetry-event.mjs +196 -0
- package/src/token-counter.mjs +3 -1
- package/src/turn-state.mjs +132 -0
- package/src/user-config.mjs +18 -8
|
@@ -0,0 +1,96 @@
|
|
|
1
|
+
// Exact Claude API IDs only. A name containing "sonnet" or "opus" is not
|
|
2
|
+
// evidence that a gateway alias implements that model's request contract.
|
|
3
|
+
export const MODEL_CATALOG_REVIEWED_AT = '2026-10-05';
|
|
4
|
+
export const MODEL_CAPABILITY_SOURCES = Object.freeze({
|
|
5
|
+
thinking: 'https://platform.claude.com/docs/en/build-with-claude/thinking-troubleshooting',
|
|
6
|
+
effort: 'https://platform.claude.com/docs/en/build-with-claude/effort',
|
|
7
|
+
sonnet55: 'https://platform.claude.com/docs/en/models/sonnet-5-5/migration-guide',
|
|
8
|
+
opus55: 'https://platform.claude.com/docs/en/models/opus-5-5/whats-new-opus-5-5',
|
|
9
|
+
auto: 'https://code.claude.com/docs/en/permission-modes',
|
|
10
|
+
subscriptionContext: 'https://code.claude.com/docs/en/model-config',
|
|
11
|
+
});
|
|
12
|
+
|
|
13
|
+
/**
|
|
14
|
+
* @typedef {object} ModelCapabilities
|
|
15
|
+
* @property {'haiku'|'sonnet'|'opus'} family
|
|
16
|
+
* @property {number} maxOutputTokens Standard Messages API limit, not Batch beta.
|
|
17
|
+
* @property {number|undefined} contextWindow Unambiguous window across supported auth modes.
|
|
18
|
+
* @property {boolean} capacityUpgradeSource Existing small/opt-in-window models.
|
|
19
|
+
* @property {boolean} toolReferences
|
|
20
|
+
* @property {boolean} autoMode Claude client Auto eligibility, not arbitrary request compatibility.
|
|
21
|
+
* @property {boolean} sharedAuto Verified modern Auto execution pair.
|
|
22
|
+
* @property {readonly string[]} thinkingTypes
|
|
23
|
+
* @property {readonly string[]} effortLevels
|
|
24
|
+
* @property {boolean} forcedToolChoice
|
|
25
|
+
* @property {boolean} assistantPrefill
|
|
26
|
+
* @property {boolean} defaultSamplingOnly
|
|
27
|
+
* @property {boolean} midConversationSystem
|
|
28
|
+
* @property {boolean} perMessageEffort
|
|
29
|
+
* @property {boolean} taskBudget
|
|
30
|
+
* @property {'adaptive'|'between_tools'|undefined} [disabledThinkingAdaptation]
|
|
31
|
+
* @property {string} reviewedAt
|
|
32
|
+
* @property {string} source
|
|
33
|
+
*/
|
|
34
|
+
|
|
35
|
+
const BASIC_EFFORT = Object.freeze(['low', 'medium', 'high']);
|
|
36
|
+
const FOUR_EFFORT = Object.freeze([...BASIC_EFFORT, 'max']);
|
|
37
|
+
const FIVE_EFFORT = Object.freeze([...BASIC_EFFORT, 'xhigh', 'max']);
|
|
38
|
+
const EXTENDED = Object.freeze(['disabled', 'enabled']);
|
|
39
|
+
const BOTH = Object.freeze(['disabled', 'enabled', 'adaptive']);
|
|
40
|
+
const ADAPTIVE = Object.freeze(['disabled', 'adaptive']);
|
|
41
|
+
const registry = {};
|
|
42
|
+
|
|
43
|
+
function add(ids, facts) {
|
|
44
|
+
const canonical = ids[0].replace(/^claude-/, '');
|
|
45
|
+
const entry = Object.freeze({
|
|
46
|
+
contextWindow: 200000, maxOutputTokens: 64000, capacityUpgradeSource: true,
|
|
47
|
+
toolReferences: true, autoMode: false, sharedAuto: false,
|
|
48
|
+
thinkingTypes: EXTENDED, effortLevels: Object.freeze([]),
|
|
49
|
+
forcedToolChoice: true, assistantPrefill: true, defaultSamplingOnly: false,
|
|
50
|
+
midConversationSystem: false, perMessageEffort: false, taskBudget: false,
|
|
51
|
+
reviewedAt: MODEL_CATALOG_REVIEWED_AT,
|
|
52
|
+
source: `https://platform.claude.com/docs/en/models/${canonical}/overview`,
|
|
53
|
+
...facts,
|
|
54
|
+
});
|
|
55
|
+
for (const id of ids) registry[id] = entry;
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
add(['claude-haiku-4-5', 'claude-haiku-4-5-20251001'], { family: 'haiku' });
|
|
59
|
+
add(['claude-sonnet-4-5', 'claude-sonnet-4-5-20250929'], { family: 'sonnet' });
|
|
60
|
+
add(['claude-opus-4-5', 'claude-opus-4-5-20251101'], { family: 'opus', effortLevels: BASIC_EFFORT });
|
|
61
|
+
// 4.6 has a 1M API window, but subscription usage can need a client opt-in.
|
|
62
|
+
// Retain the existing conservative capacity policy instead of promising 1M.
|
|
63
|
+
for (const family of ['sonnet', 'opus']) add([`claude-${family}-4-6`], {
|
|
64
|
+
family, contextWindow: undefined, maxOutputTokens: 128000, autoMode: true,
|
|
65
|
+
thinkingTypes: BOTH, effortLevels: FOUR_EFFORT, assistantPrefill: false,
|
|
66
|
+
});
|
|
67
|
+
for (const version of ['4-7', '4-8']) add([`claude-opus-${version}`], {
|
|
68
|
+
family: 'opus', contextWindow: 1000000, maxOutputTokens: 128000,
|
|
69
|
+
capacityUpgradeSource: false, autoMode: true, thinkingTypes: ADAPTIVE,
|
|
70
|
+
effortLevels: FIVE_EFFORT, assistantPrefill: false, defaultSamplingOnly: true,
|
|
71
|
+
});
|
|
72
|
+
for (const family of ['sonnet', 'opus']) for (const version of ['5', '5-5']) {
|
|
73
|
+
const latest = version === '5-5';
|
|
74
|
+
add([`claude-${family}-${version}`], {
|
|
75
|
+
family, contextWindow: 1000000, maxOutputTokens: 128000,
|
|
76
|
+
capacityUpgradeSource: false, autoMode: true, sharedAuto: true,
|
|
77
|
+
thinkingTypes: latest ? Object.freeze(family === 'sonnet' ? ['adaptive', 'between_tools'] : ['adaptive']) : ADAPTIVE,
|
|
78
|
+
effortLevels: FIVE_EFFORT, forcedToolChoice: !latest,
|
|
79
|
+
assistantPrefill: false, defaultSamplingOnly: true,
|
|
80
|
+
midConversationSystem: family === 'opus' || latest,
|
|
81
|
+
perMessageEffort: family === 'opus' || latest,
|
|
82
|
+
taskBudget: family === 'opus' || latest,
|
|
83
|
+
// Preserve the existing explicit adaptation for Opus 5 as well as 5.5.
|
|
84
|
+
disabledThinkingAdaptation: family === 'opus' ? 'adaptive' : latest ? 'between_tools' : undefined,
|
|
85
|
+
});
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
export const MODEL_CATALOG = Object.freeze(registry);
|
|
89
|
+
/** @returns {Readonly<ModelCapabilities>|undefined} */
|
|
90
|
+
export const modelCapabilities = model => typeof model === 'string' && Object.hasOwn(MODEL_CATALOG, model)
|
|
91
|
+
? MODEL_CATALOG[model] : undefined;
|
|
92
|
+
export const modelContextWindow = model => modelCapabilities(model)?.contextWindow;
|
|
93
|
+
export const hasNativeMillionContext = model => modelContextWindow(model) === 1000000;
|
|
94
|
+
export const canUpgradeContext = model => modelCapabilities(model)?.capacityUpgradeSource === true;
|
|
95
|
+
export const supportsToolReferences = model => modelCapabilities(model)?.toolReferences === true;
|
|
96
|
+
export const supportsAutoMode = model => modelCapabilities(model)?.autoMode === true;
|
package/src/model-request.mjs
CHANGED
|
@@ -1,6 +1,4 @@
|
|
|
1
|
-
|
|
2
|
-
// thinking contracts. Preserve the existing adaptive Opus upgrade behavior.
|
|
3
|
-
const ADAPTIVE_TARGETS = new Set(['claude-opus-5', 'claude-opus-5-5']);
|
|
1
|
+
import { modelCapabilities } from './model-catalog.mjs';
|
|
4
2
|
|
|
5
3
|
function sonnetNeedsAdaptive(body) {
|
|
6
4
|
const effort = body.output_config?.effort ?? 'high';
|
|
@@ -14,12 +12,18 @@ function sonnetNeedsAdaptive(body) {
|
|
|
14
12
|
export function prepareRequest(body, model) {
|
|
15
13
|
const request = { ...body, model };
|
|
16
14
|
const adjustments = [];
|
|
17
|
-
|
|
18
|
-
|
|
15
|
+
const adaptation = modelCapabilities(model)?.disabledThinkingAdaptation;
|
|
16
|
+
if (model !== body.model && body.model === 'claude-sonnet-5-5' && body.thinking?.type === 'between_tools'
|
|
17
|
+
&& Object.keys(body.thinking).length === 1 && adaptation === 'adaptive') {
|
|
18
|
+
request.thinking = { type: 'adaptive' };
|
|
19
|
+
adjustments.push('adaptive_thinking_required');
|
|
20
|
+
}
|
|
21
|
+
if (model !== body.model && body.thinking?.type === 'disabled' && Object.keys(body.thinking).length === 1) {
|
|
22
|
+
if (adaptation === 'between_tools') {
|
|
19
23
|
const type = sonnetNeedsAdaptive(body) ? 'adaptive' : 'between_tools';
|
|
20
24
|
request.thinking = { type };
|
|
21
25
|
adjustments.push(type === 'adaptive' ? 'adaptive_thinking_required' : 'between_tools_thinking_required');
|
|
22
|
-
} else if (
|
|
26
|
+
} else if (adaptation === 'adaptive') {
|
|
23
27
|
request.thinking = { type: 'adaptive' };
|
|
24
28
|
adjustments.push('adaptive_thinking_required');
|
|
25
29
|
}
|
package/src/ollama-evaluator.mjs
CHANGED
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import { validateOllamaEndpoint, validateOllamaModel } from './ollama-models.mjs';
|
|
2
2
|
import { buildState } from './prompt-state.mjs';
|
|
3
|
+
import { cancelResponseBody, readBoundedJson, MODEL_METADATA_LIMIT } from './bounded-json.mjs';
|
|
3
4
|
|
|
4
5
|
// Bound UTF-8 bytes as well as serialized characters to keep local decision
|
|
5
6
|
// excerpts small, including when the prompt contains non-ASCII text.
|
|
@@ -51,44 +52,20 @@ export function buildOllamaRequest(state, config) {
|
|
|
51
52
|
return { model: config.ollamaModel, state, questions: OLLAMA_QUESTIONS, keep_alive: config.ollamaKeepAlive };
|
|
52
53
|
}
|
|
53
54
|
|
|
54
|
-
async function readJson(response, signal, limit = 64 * 1024) {
|
|
55
|
-
const reader = response.body?.getReader();
|
|
56
|
-
if (!reader) throw new Error('classifier_invalid_response');
|
|
57
|
-
let total = 0;
|
|
58
|
-
const chunks = [];
|
|
59
|
-
const abort = () => { void reader.cancel(signal.reason).catch(() => {}); };
|
|
60
|
-
signal.addEventListener('abort', abort, { once: true });
|
|
61
|
-
try {
|
|
62
|
-
while (true) {
|
|
63
|
-
signal.throwIfAborted();
|
|
64
|
-
const { value, done } = await reader.read();
|
|
65
|
-
if (done) break;
|
|
66
|
-
total += value.byteLength;
|
|
67
|
-
if (total > limit) throw new Error('classifier_invalid_response');
|
|
68
|
-
chunks.push(Buffer.from(value));
|
|
69
|
-
}
|
|
70
|
-
signal.throwIfAborted();
|
|
71
|
-
return JSON.parse(Buffer.concat(chunks).toString('utf8'));
|
|
72
|
-
} finally {
|
|
73
|
-
signal.removeEventListener('abort', abort);
|
|
74
|
-
await reader.cancel().catch(() => {}); reader.releaseLock();
|
|
75
|
-
}
|
|
76
|
-
}
|
|
77
|
-
|
|
78
55
|
async function request(config, path, body, { fetchImpl, signal, responseLimit }) {
|
|
79
56
|
const response = await fetchImpl(`${config.ollamaEndpoint}${path}`, {
|
|
80
57
|
method: 'POST', redirect: 'error', signal,
|
|
81
58
|
headers: { 'content-type': 'application/json' }, body: JSON.stringify(body),
|
|
82
59
|
});
|
|
83
60
|
if (!response.ok) {
|
|
84
|
-
|
|
61
|
+
cancelResponseBody(response);
|
|
85
62
|
const needsVersion = path === '/v1/systemone' && response.status === 404;
|
|
86
63
|
const error = new Error(needsVersion ? OLLAMA_VERSION_MESSAGE : 'classifier_http_error');
|
|
87
64
|
if (needsVersion) error.code = 'OLLAMA_VERSION';
|
|
88
65
|
error.classifierStatus = response.status;
|
|
89
66
|
throw error;
|
|
90
67
|
}
|
|
91
|
-
return
|
|
68
|
+
return readBoundedJson(response, { signal, limit: responseLimit });
|
|
92
69
|
}
|
|
93
70
|
|
|
94
71
|
const record = value => value !== null && typeof value === 'object' && !Array.isArray(value);
|
|
@@ -120,13 +97,18 @@ export async function checkLocalOllamaModel(config, options) {
|
|
|
120
97
|
validateOllamaModel(config.ollamaModel);
|
|
121
98
|
// /show includes tensor metadata and licenses that can exceed 64 KiB. Keep
|
|
122
99
|
// its separate limit bounded while decision responses stay at 64 KiB.
|
|
123
|
-
const model = await request(config, '/api/show', { model: config.ollamaModel }, { ...options, responseLimit:
|
|
100
|
+
const model = await request(config, '/api/show', { model: config.ollamaModel }, { ...options, responseLimit: MODEL_METADATA_LIMIT });
|
|
124
101
|
if (!model || typeof model !== 'object' || model.remote_host || model.remote_model
|
|
125
102
|
|| typeof model.details?.parameter_size !== 'string' || !model.details.parameter_size) {
|
|
126
103
|
throw new Error('classifier_invalid_response');
|
|
127
104
|
}
|
|
128
105
|
}
|
|
129
106
|
|
|
107
|
+
/**
|
|
108
|
+
* @param {any} state
|
|
109
|
+
* @param {any} config
|
|
110
|
+
* @param {{fetchImpl?:typeof fetch,signal?:AbortSignal}} [options]
|
|
111
|
+
*/
|
|
130
112
|
export async function evaluateOllama(state, config, { fetchImpl = fetch, signal } = {}) {
|
|
131
113
|
let combined = signal;
|
|
132
114
|
if (config.ollamaTimeoutMs !== 0) {
|
package/src/onboarding.mjs
CHANGED
|
@@ -1,11 +1,10 @@
|
|
|
1
1
|
import { execFile } from 'node:child_process';
|
|
2
|
-
import { existsSync } from 'node:fs';
|
|
3
2
|
import { createInterface } from 'node:readline';
|
|
4
3
|
import { Writable } from 'node:stream';
|
|
5
4
|
import { promisify } from 'node:util';
|
|
6
5
|
import { CLIENT_PROFILES, readConfig, requireKeys, parseStopHookBlockCap, parseSessionLogDir } from './config.mjs';
|
|
7
6
|
import { buildClaudeEnv, conflictingProviders, LOCAL_AUTH_HEADER } from './auth.mjs';
|
|
8
|
-
import {
|
|
7
|
+
import { CONFIG_KEYS, SECRET_CONFIG_KEYS, loadUserConfig, saveUserConfig } from './user-config.mjs';
|
|
9
8
|
import { DEFAULT_OLLAMA_MODEL, validateOllamaModel } from './ollama-models.mjs';
|
|
10
9
|
import { inspectOllama, setupOllama } from './ollama-setup.mjs';
|
|
11
10
|
|
|
@@ -40,19 +39,29 @@ export async function askSecret(label, { input = process.stdin, output = process
|
|
|
40
39
|
export async function setup(args, {
|
|
41
40
|
env = process.env, write = console.log, prompt = askSecret, fetchImpl = fetch, signal,
|
|
42
41
|
} = {}) {
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
42
|
+
if (signal?.aborted) throw new Error('Setup cancelled');
|
|
43
|
+
const loaded = loadUserConfig(env, { allowMissing: true });
|
|
44
|
+
const replace = args.includes('--replace');
|
|
45
|
+
const mergeExisting = loaded.exists && !replace;
|
|
46
|
+
// Updating one preference must not turn unrelated runtime overrides into
|
|
47
|
+
// saved defaults. First setup/replacement still supports environment-only
|
|
48
|
+
// configuration; an explicitly selected backend can use its supplied key.
|
|
49
|
+
const effectiveEnv = mergeExisting ? loaded.values : env;
|
|
50
|
+
let explicitEvaluator = false, explicitAuthMode = false;
|
|
51
|
+
let authMode = effectiveEnv.AUTOROUTER_AUTH_MODE ?? 'subscription';
|
|
52
|
+
let clientProfile = effectiveEnv.AUTOROUTER_CLIENT_PROFILE ?? 'compatible';
|
|
53
|
+
let evaluator = effectiveEnv.AUTOROUTER_EVALUATOR ?? 'jev';
|
|
46
54
|
let model;
|
|
47
55
|
let ollamaTimeoutMs;
|
|
48
|
-
let stopHookBlockCap =
|
|
49
|
-
let sessionLogDir =
|
|
56
|
+
let stopHookBlockCap = effectiveEnv.CLAUDE_CODE_STOP_HOOK_BLOCK_CAP;
|
|
57
|
+
let sessionLogDir = effectiveEnv.AUTOROUTER_SESSION_LOG_DIR;
|
|
58
|
+
let sessionLogMode = effectiveEnv.AUTOROUTER_SESSION_LOG_MODE;
|
|
50
59
|
let pull = false;
|
|
51
|
-
let overwrite =
|
|
60
|
+
let overwrite = replace;
|
|
52
61
|
for (let i = 0; i < args.length; i++) {
|
|
53
|
-
if (args[i] === '--auth-mode') authMode = args[++i];
|
|
62
|
+
if (args[i] === '--auth-mode') { authMode = args[++i]; explicitAuthMode = true; }
|
|
54
63
|
else if (args[i] === '--client-profile') clientProfile = args[++i];
|
|
55
|
-
else if (args[i] === '--evaluator') evaluator = args[++i];
|
|
64
|
+
else if (args[i] === '--evaluator') { evaluator = args[++i]; explicitEvaluator = true; }
|
|
56
65
|
else if (args[i] === '--ollama-model') { model = args[++i]; if (model === undefined) throw new Error('--ollama-model requires a model tag'); }
|
|
57
66
|
else if (args[i] === '--ollama-timeout-ms') {
|
|
58
67
|
const value = args[++i];
|
|
@@ -69,9 +78,14 @@ export async function setup(args, {
|
|
|
69
78
|
if (sessionLogDir === undefined || sessionLogDir.startsWith('--')) throw new Error('--session-log-dir requires a directory path');
|
|
70
79
|
parseSessionLogDir(sessionLogDir, '--session-log-dir');
|
|
71
80
|
}
|
|
81
|
+
else if (args[i] === '--session-log-mode') {
|
|
82
|
+
sessionLogMode = args[++i];
|
|
83
|
+
if (!['metadata', 'prompts'].includes(sessionLogMode)) throw new Error('--session-log-mode must be metadata or prompts');
|
|
84
|
+
}
|
|
72
85
|
else if (args[i] === '--pull') pull = true;
|
|
73
86
|
else if (args[i] === '--force') overwrite = true;
|
|
74
|
-
else
|
|
87
|
+
else if (args[i] === '--replace') continue;
|
|
88
|
+
else throw new Error('Usage: claude-autorouter setup [--auth-mode subscription|api-key] [--client-profile compatible|native|auto] [--evaluator jev|ollama] [--ollama-model TAG] [--ollama-timeout-ms N] [--stop-hook-block-cap N] [--session-log-dir DIR] [--session-log-mode metadata|prompts] [--pull] [--force|--replace]');
|
|
75
89
|
}
|
|
76
90
|
if (!['subscription', 'api-key'].includes(authMode)) throw new Error('--auth-mode must be subscription or api-key');
|
|
77
91
|
if (!CLIENT_PROFILES.includes(clientProfile)) throw new Error('--client-profile must be compatible, native or auto');
|
|
@@ -79,32 +93,47 @@ export async function setup(args, {
|
|
|
79
93
|
if (evaluator !== 'ollama' && (model !== undefined || ollamaTimeoutMs !== undefined || pull)) throw new Error('Ollama model, deadline and download options require --evaluator ollama');
|
|
80
94
|
if (stopHookBlockCap !== undefined) stopHookBlockCap = parseStopHookBlockCap(stopHookBlockCap);
|
|
81
95
|
if (sessionLogDir !== undefined) sessionLogDir = parseSessionLogDir(sessionLogDir) ?? '';
|
|
82
|
-
const path =
|
|
83
|
-
if (!overwrite &&
|
|
96
|
+
const path = loaded.path;
|
|
97
|
+
if (!overwrite && loaded.exists) throw new Error('AutoRouter configuration already exists. Use setup --force to update it while preserving unrelated settings.');
|
|
84
98
|
write(evaluator === 'ollama'
|
|
85
99
|
? 'AutoRouter evaluates bounded prompt excerpts locally with Ollama. Complete requests still go to Anthropic.'
|
|
86
100
|
: 'AutoRouter sends bounded prompt excerpts to TypeSafe Jev and complete requests to Anthropic.');
|
|
87
|
-
const values =
|
|
101
|
+
const values = replace ? {} : { ...loaded.values };
|
|
102
|
+
const selectedBackendKeys = explicitEvaluator ? CONFIG_KEYS.filter(key => evaluator === 'ollama'
|
|
103
|
+
? key.startsWith('AUTOROUTER_OLLAMA_') : key.startsWith('AUTOROUTER_JEV_') || key === 'AUTOROUTER_MIN_CONFIDENCE') : [];
|
|
104
|
+
for (const key of mergeExisting ? selectedBackendKeys : CONFIG_KEYS) {
|
|
105
|
+
if (!SECRET_CONFIG_KEYS.includes(key) && env[key] !== undefined) values[key] = env[key];
|
|
106
|
+
}
|
|
107
|
+
Object.assign(values, { AUTOROUTER_AUTH_MODE: authMode, AUTOROUTER_CLIENT_PROFILE: clientProfile, AUTOROUTER_EVALUATOR: evaluator });
|
|
88
108
|
if (stopHookBlockCap !== undefined) values.CLAUDE_CODE_STOP_HOOK_BLOCK_CAP = String(stopHookBlockCap);
|
|
89
109
|
if (sessionLogDir !== undefined) values.AUTOROUTER_SESSION_LOG_DIR = sessionLogDir;
|
|
110
|
+
if (sessionLogMode !== undefined) values.AUTOROUTER_SESSION_LOG_MODE = sessionLogMode;
|
|
90
111
|
if (evaluator === 'ollama') {
|
|
91
|
-
values.AUTOROUTER_OLLAMA_MODEL = validateOllamaModel(model
|
|
112
|
+
values.AUTOROUTER_OLLAMA_MODEL = validateOllamaModel(model
|
|
113
|
+
?? (explicitEvaluator ? env.AUTOROUTER_OLLAMA_MODEL : undefined) ?? effectiveEnv.AUTOROUTER_OLLAMA_MODEL ?? DEFAULT_OLLAMA_MODEL);
|
|
92
114
|
for (const key of ['AUTOROUTER_OLLAMA_URL', 'AUTOROUTER_OLLAMA_TIMEOUT_MS', 'AUTOROUTER_OLLAMA_KEEP_ALIVE']) {
|
|
93
|
-
if (env[key] !== undefined) values[key] = env[key];
|
|
115
|
+
if ((!mergeExisting || explicitEvaluator) && env[key] !== undefined) values[key] = env[key];
|
|
94
116
|
}
|
|
95
117
|
if (ollamaTimeoutMs !== undefined) values.AUTOROUTER_OLLAMA_TIMEOUT_MS = ollamaTimeoutMs;
|
|
96
118
|
}
|
|
97
119
|
const keys = [...(evaluator === 'jev' ? ['TYPESAFE_API_KEY'] : []), ...(authMode === 'api-key' ? ['ANTHROPIC_API_KEY'] : [])];
|
|
120
|
+
// Reject invalid settings before inviting secret input or making local calls.
|
|
121
|
+
readConfig(values);
|
|
98
122
|
for (const key of keys) {
|
|
99
|
-
|
|
123
|
+
if (signal?.aborted) throw new Error('Setup cancelled');
|
|
124
|
+
const explicitlySelected = !mergeExisting || (key === 'TYPESAFE_API_KEY' ? explicitEvaluator : explicitAuthMode);
|
|
125
|
+
const value = ((explicitlySelected ? env[key]?.trim() : '') || effectiveEnv[key]?.trim() || env[key]?.trim() || await prompt(key)).trim();
|
|
126
|
+
if (signal?.aborted) throw new Error('Setup cancelled');
|
|
100
127
|
if (!value || /[\r\n\0]/.test(value)) throw new Error(`${key} must be a nonempty, single-line key`);
|
|
101
128
|
values[key] = value;
|
|
102
129
|
}
|
|
103
130
|
const config = readConfig(values);
|
|
104
131
|
requireKeys(config);
|
|
132
|
+
if (signal?.aborted) throw new Error('Setup cancelled');
|
|
105
133
|
if (config.clientProfile === 'auto') write('Auto-compatible profile: Sonnet/Opus task routing. Claude controls permission-mode availability and safety checks.');
|
|
106
134
|
if (config.stopHookBlockCap !== undefined) write(stopHookCapText(config.stopHookBlockCap));
|
|
107
|
-
if (config.sessionLogDir) write('Session
|
|
135
|
+
if (config.sessionLogDir) write(config.sessionLogMode === 'metadata' ? 'Session logs enabled with metadata only; prompt excerpts are omitted.'
|
|
136
|
+
: 'Session decision logs enabled; files include up to 500 characters of user prompt text per decision.');
|
|
108
137
|
if (evaluator === 'ollama') {
|
|
109
138
|
write(`Local evaluator: ${config.ollamaModel}; ${ollamaDeadlineText(config.ollamaTimeoutMs)}.`);
|
|
110
139
|
const controller = new AbortController();
|
|
@@ -114,13 +143,37 @@ export async function setup(args, {
|
|
|
114
143
|
finally { if (!signal) for (const name of ['SIGINT', 'SIGTERM']) process.removeListener(name, cancel); }
|
|
115
144
|
}
|
|
116
145
|
if (signal?.aborted) throw new Error('Setup cancelled');
|
|
117
|
-
saveUserConfig(values, { env, overwrite });
|
|
146
|
+
saveUserConfig(values, { env, overwrite, expectedRevision: loaded.revision });
|
|
118
147
|
write(`Saved ${authMode} configuration to ${path}`);
|
|
119
148
|
write(`${keys.length ? 'Keys and settings are' : 'Settings are'} stored locally in this file with owner-only permissions. Environment variables take precedence.`);
|
|
120
149
|
write('Next: claude-autorouter doctor, then claude-autorouter claude from your project.');
|
|
121
150
|
}
|
|
122
151
|
|
|
123
|
-
export async function doctor({ env = process.env, write = console.log, run = execute, fetchImpl = fetch, signal
|
|
152
|
+
export async function doctor({ env = process.env, write = console.log, run = execute, fetchImpl = fetch, signal,
|
|
153
|
+
evaluateLocal = false, json = false, diagnosticRunner } = {}) {
|
|
154
|
+
if (evaluateLocal) {
|
|
155
|
+
try {
|
|
156
|
+
const config = readConfig(loadUserConfig(env).env);
|
|
157
|
+
if (config.evaluator !== 'ollama') throw new Error('Local evaluation requires AUTOROUTER_EVALUATOR=ollama. It does not call Jev or Anthropic.');
|
|
158
|
+
const diagnostic = await import('./local-diagnostic.mjs');
|
|
159
|
+
const onProgress = json ? () => {} : progress => {
|
|
160
|
+
if (progress.event === 'preflight') write('Checking the local evaluator; no Claude authentication or cloud requests are used.');
|
|
161
|
+
else if (progress.event === 'startup') write('Preparing the local model with a separate 60-second startup deadline…');
|
|
162
|
+
else if (progress.event === 'case_start') write('Checking a synthetic routing case…');
|
|
163
|
+
else if (progress.event === 'case_complete') write('Synthetic routing case finished.');
|
|
164
|
+
};
|
|
165
|
+
const result = await (diagnosticRunner ?? diagnostic.runLocalDiagnostic)(config, { fetchImpl, signal, onProgress });
|
|
166
|
+
if (json) write(JSON.stringify(result));
|
|
167
|
+
else for (const line of diagnostic.formatLocalDiagnostic(result)) write(line);
|
|
168
|
+
return result.passed;
|
|
169
|
+
} catch (error) {
|
|
170
|
+
if (signal?.aborted) throw error;
|
|
171
|
+
if (json) write(JSON.stringify({ schema_version: 1, type: 'local_evaluator_diagnostic', passed: false,
|
|
172
|
+
error: { code: 'configuration_error', message: error.message } }));
|
|
173
|
+
else write(`FAIL ${error.message}`);
|
|
174
|
+
return false;
|
|
175
|
+
}
|
|
176
|
+
}
|
|
124
177
|
let healthy = true;
|
|
125
178
|
const report = (ok, message) => { if (!ok) healthy = false; write(`${ok ? 'OK' : 'FAIL'} ${message}`); };
|
|
126
179
|
report(Number(process.versions.node.split('.')[0]) >= 22, `Node.js ${process.versions.node} (requires 22+)`);
|
|
@@ -134,14 +187,19 @@ export async function doctor({ env = process.env, write = console.log, run = exe
|
|
|
134
187
|
requireKeys(config);
|
|
135
188
|
report(true, `Configuration and required keys present (${config.authMode})`);
|
|
136
189
|
} catch (error) {
|
|
137
|
-
report(false, `${error.message}.
|
|
190
|
+
report(false, `${error.message}. Use claude-autorouter config show --check-all to inspect settings, or setup for first-time configuration.`);
|
|
138
191
|
}
|
|
139
192
|
for (const key of conflictingProviders(effectiveEnv)) {
|
|
140
193
|
report(false, `Unset ${key}; AutoRouter uses the Anthropic Messages API`);
|
|
141
194
|
}
|
|
142
195
|
if (config?.stopHookBlockCap !== undefined) write(stopHookCapText(config.stopHookBlockCap));
|
|
196
|
+
if (config) {
|
|
197
|
+
write(`Routing profile: ${config.clientProfile}; Haiku ${config.models.haiku}; Sonnet ${config.models.sonnet}; Opus ${config.models.opus}.`);
|
|
198
|
+
if (config.evaluator === 'jev') write(`Evaluator: Jev ${config.jevModel}; routing deadline ${config.jevTimeoutMs} ms; confidence floor ${config.minConfidence}.`);
|
|
199
|
+
}
|
|
143
200
|
if (config?.clientProfile === 'auto') write('Auto-compatible profile: Sonnet/Opus task routing. Claude controls permission-mode availability and safety checks.');
|
|
144
|
-
if (config?.sessionLogDir) write('Session
|
|
201
|
+
if (config?.sessionLogDir) write(config.sessionLogMode === 'metadata' ? 'Session logs enabled with metadata only; prompt excerpts are omitted.'
|
|
202
|
+
: 'Session decision logs enabled; files include up to 500 characters of user prompt text per decision.');
|
|
145
203
|
if (config?.evaluator === 'ollama') {
|
|
146
204
|
write(`Local evaluator: ${config.ollamaModel}; ${ollamaDeadlineText(config.ollamaTimeoutMs)}.`);
|
|
147
205
|
write('Model availability is checked below; classification speed and accuracy are not tested.');
|
|
@@ -149,7 +207,7 @@ export async function doctor({ env = process.env, write = console.log, run = exe
|
|
|
149
207
|
const result = await inspectOllama(config, { fetchImpl, signal });
|
|
150
208
|
report(result.installed, result.installed
|
|
151
209
|
? `Local Ollama model available (${result.model})`
|
|
152
|
-
: `Ollama model missing (${result.model}). Run ollama
|
|
210
|
+
: `Ollama model missing (${result.model}). Run claude-autorouter setup --evaluator ollama --ollama-model ${result.model} --pull --force.`);
|
|
153
211
|
} catch (error) {
|
|
154
212
|
report(false, error.message);
|
|
155
213
|
}
|
|
@@ -168,6 +226,7 @@ export async function doctor({ env = process.env, write = console.log, run = exe
|
|
|
168
226
|
const { stdout } = await run('claude', ['--version'], { env: childEnv, timeout: 10000, maxBuffer: 64 * 1024 });
|
|
169
227
|
const version = /\b\d+\.\d+\.\d+\b/.exec(stdout)?.[0];
|
|
170
228
|
report(Boolean(version), version ? `Claude Code ${version}` : 'Could not recognize Claude Code version');
|
|
229
|
+
if (version) write(`Historical integration observations cover Claude Code 2.1.284 and 2.1.285; finding an executable does not certify its full compatibility${['2.1.284', '2.1.285'].includes(version) ? '.' : ' (installed version differs).'}`);
|
|
171
230
|
claudeAvailable = Boolean(version);
|
|
172
231
|
} catch {
|
|
173
232
|
report(false, 'Claude Code unavailable. Install claude and ensure it is on PATH.');
|
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
const object = value => value !== null && typeof value === 'object' && !Array.isArray(value);
|
|
2
|
+
const text = value => typeof value === 'string' && value.length > 0;
|
|
3
|
+
const valid = Object.freeze({ valid: true });
|
|
4
|
+
const invalid = field => ({ valid: false, error: `Invalid Messages API request shape: ${field}` });
|
|
5
|
+
|
|
6
|
+
// Validate containers and scalar fields read by routing/excerpt extraction.
|
|
7
|
+
// This intentionally is not the provider's full schema: unknown fields and
|
|
8
|
+
// block/tool types remain untouched, and opaque inputs/schemas are not walked.
|
|
9
|
+
export function validateRequestShape(body) {
|
|
10
|
+
if (!object(body) || !text(body.model) || !body.model.trim()) return invalid('model');
|
|
11
|
+
if (!Array.isArray(body.messages)) return invalid('messages');
|
|
12
|
+
const pending = [];
|
|
13
|
+
const content = (value, field, depth = 0) => {
|
|
14
|
+
if (typeof value === 'string') return true;
|
|
15
|
+
if (!Array.isArray(value)) return false;
|
|
16
|
+
pending.push({ blocks: value, field, depth });
|
|
17
|
+
return true;
|
|
18
|
+
};
|
|
19
|
+
if (body.system !== undefined && !content(body.system, 'system')) return invalid('system');
|
|
20
|
+
for (const message of body.messages) {
|
|
21
|
+
if (!object(message) || !['user', 'assistant', 'system'].includes(message.role)) return invalid('messages');
|
|
22
|
+
if (!content(message.content, 'message content')) return invalid('message content');
|
|
23
|
+
if (message.output_config !== undefined && message.output_config !== null && !object(message.output_config)) return invalid('message output_config');
|
|
24
|
+
}
|
|
25
|
+
if (body.tools !== undefined && (!Array.isArray(body.tools) || body.tools.some(tool =>
|
|
26
|
+
!object(tool) || (tool.type !== undefined && tool.type !== null && !text(tool.type))
|
|
27
|
+
|| (tool.name !== undefined && typeof tool.name !== 'string')))) return invalid('tools');
|
|
28
|
+
for (const field of ['thinking', 'tool_choice', 'output_config', 'context_management']) {
|
|
29
|
+
// Context management is explicitly nullable in the beta Messages schema.
|
|
30
|
+
if (field === 'context_management' && body[field] === null) continue;
|
|
31
|
+
if (body[field] !== undefined && !object(body[field])) return invalid(field);
|
|
32
|
+
}
|
|
33
|
+
for (const field of ['thinking', 'tool_choice']) {
|
|
34
|
+
if (body[field] !== undefined && !text(body[field].type)) return invalid(field);
|
|
35
|
+
}
|
|
36
|
+
// The Messages API permits zero for cache population without generation.
|
|
37
|
+
// Reviewed 2026-10-05: https://platform.claude.com/docs/en/api/beta/messages/create
|
|
38
|
+
if (body.max_tokens !== undefined && (!Number.isSafeInteger(body.max_tokens) || body.max_tokens < 0)) return invalid('max_tokens');
|
|
39
|
+
if (body.stream !== undefined && typeof body.stream !== 'boolean') return invalid('stream');
|
|
40
|
+
while (pending.length) {
|
|
41
|
+
const { blocks, field, depth } = pending.pop();
|
|
42
|
+
// Excerpt extraction recursively reads tool results. Bound this known
|
|
43
|
+
// content nesting before it can exhaust the stack; never echo field data.
|
|
44
|
+
if (depth > 32) return invalid('content nesting');
|
|
45
|
+
for (const block of blocks) {
|
|
46
|
+
if (!object(block) || !text(block.type)) return invalid(field);
|
|
47
|
+
if (block.type === 'text' && typeof block.text !== 'string') return invalid('text content');
|
|
48
|
+
if (block.type === 'tool_use' && block.name !== undefined && typeof block.name !== 'string') return invalid('tool name');
|
|
49
|
+
if (block.type === 'tool_result' && block.content !== undefined
|
|
50
|
+
&& !content(block.content, 'tool result content', depth + 1)) return invalid('tool result content');
|
|
51
|
+
}
|
|
52
|
+
}
|
|
53
|
+
return valid;
|
|
54
|
+
}
|