claude-autorouter 0.3.7 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. package/.env.example +4 -2
  2. package/CONTRIBUTING.md +37 -0
  3. package/README.md +43 -70
  4. package/bin/autorouter.mjs +40 -57
  5. package/docs/development.md +48 -2
  6. package/docs/hardware-benchmark.md +29 -0
  7. package/docs/hardware-comparison.md +55 -0
  8. package/docs/hardware-results-16gb.json +4002 -0
  9. package/docs/hardware-results-16gb.md +26 -0
  10. package/docs/hardware-results-64gb.json +4020 -0
  11. package/docs/reference.md +71 -32
  12. package/docs/releasing.md +74 -34
  13. package/docs/router-performance.json +1697 -0
  14. package/docs/router-performance.md +50 -0
  15. package/docs/status-performance.json +363 -0
  16. package/docs/status-performance.md +44 -0
  17. package/package.json +57 -9
  18. package/src/auto-routing.mjs +184 -24
  19. package/src/bounded-json.mjs +57 -0
  20. package/src/cli-help.mjs +87 -0
  21. package/src/config-command.mjs +141 -0
  22. package/src/config.mjs +52 -27
  23. package/src/contracts.mjs +123 -0
  24. package/src/evaluation-report.mjs +114 -0
  25. package/src/local-diagnostic.mjs +191 -0
  26. package/src/model-catalog.mjs +96 -0
  27. package/src/model-request.mjs +6 -7
  28. package/src/ollama-evaluator.mjs +9 -27
  29. package/src/onboarding.mjs +82 -23
  30. package/src/request-validation.mjs +54 -0
  31. package/src/response-observer.mjs +126 -18
  32. package/src/router.mjs +151 -61
  33. package/src/savings.mjs +74 -16
  34. package/src/server.mjs +79 -12
  35. package/src/session-history.mjs +261 -0
  36. package/src/session-log.mjs +9 -58
  37. package/src/status-state.mjs +110 -62
  38. package/src/statusline.mjs +57 -27
  39. package/src/telemetry-event.mjs +196 -0
  40. package/src/token-counter.mjs +3 -1
  41. package/src/turn-state.mjs +132 -0
  42. package/src/user-config.mjs +18 -8
@@ -0,0 +1,96 @@
1
+ // Exact Claude API IDs only. A name containing "sonnet" or "opus" is not
2
+ // evidence that a gateway alias implements that model's request contract.
3
+ export const MODEL_CATALOG_REVIEWED_AT = '2026-10-05';
4
+ export const MODEL_CAPABILITY_SOURCES = Object.freeze({
5
+ thinking: 'https://platform.claude.com/docs/en/build-with-claude/thinking-troubleshooting',
6
+ effort: 'https://platform.claude.com/docs/en/build-with-claude/effort',
7
+ sonnet55: 'https://platform.claude.com/docs/en/models/sonnet-5-5/migration-guide',
8
+ opus55: 'https://platform.claude.com/docs/en/models/opus-5-5/whats-new-opus-5-5',
9
+ auto: 'https://code.claude.com/docs/en/permission-modes',
10
+ subscriptionContext: 'https://code.claude.com/docs/en/model-config',
11
+ });
12
+
13
+ /**
14
+ * @typedef {object} ModelCapabilities
15
+ * @property {'haiku'|'sonnet'|'opus'} family
16
+ * @property {number} maxOutputTokens Standard Messages API limit, not Batch beta.
17
+ * @property {number|undefined} contextWindow Unambiguous window across supported auth modes.
18
+ * @property {boolean} capacityUpgradeSource Existing small/opt-in-window models.
19
+ * @property {boolean} toolReferences
20
+ * @property {boolean} autoMode Claude client Auto eligibility, not arbitrary request compatibility.
21
+ * @property {boolean} sharedAuto Verified modern Auto execution pair.
22
+ * @property {readonly string[]} thinkingTypes
23
+ * @property {readonly string[]} effortLevels
24
+ * @property {boolean} forcedToolChoice
25
+ * @property {boolean} assistantPrefill
26
+ * @property {boolean} defaultSamplingOnly
27
+ * @property {boolean} midConversationSystem
28
+ * @property {boolean} perMessageEffort
29
+ * @property {boolean} taskBudget
30
+ * @property {'adaptive'|'between_tools'|undefined} [disabledThinkingAdaptation]
31
+ * @property {string} reviewedAt
32
+ * @property {string} source
33
+ */
34
+
35
+ const BASIC_EFFORT = Object.freeze(['low', 'medium', 'high']);
36
+ const FOUR_EFFORT = Object.freeze([...BASIC_EFFORT, 'max']);
37
+ const FIVE_EFFORT = Object.freeze([...BASIC_EFFORT, 'xhigh', 'max']);
38
+ const EXTENDED = Object.freeze(['disabled', 'enabled']);
39
+ const BOTH = Object.freeze(['disabled', 'enabled', 'adaptive']);
40
+ const ADAPTIVE = Object.freeze(['disabled', 'adaptive']);
41
+ const registry = {};
42
+
43
+ function add(ids, facts) {
44
+ const canonical = ids[0].replace(/^claude-/, '');
45
+ const entry = Object.freeze({
46
+ contextWindow: 200000, maxOutputTokens: 64000, capacityUpgradeSource: true,
47
+ toolReferences: true, autoMode: false, sharedAuto: false,
48
+ thinkingTypes: EXTENDED, effortLevels: Object.freeze([]),
49
+ forcedToolChoice: true, assistantPrefill: true, defaultSamplingOnly: false,
50
+ midConversationSystem: false, perMessageEffort: false, taskBudget: false,
51
+ reviewedAt: MODEL_CATALOG_REVIEWED_AT,
52
+ source: `https://platform.claude.com/docs/en/models/${canonical}/overview`,
53
+ ...facts,
54
+ });
55
+ for (const id of ids) registry[id] = entry;
56
+ }
57
+
58
+ add(['claude-haiku-4-5', 'claude-haiku-4-5-20251001'], { family: 'haiku' });
59
+ add(['claude-sonnet-4-5', 'claude-sonnet-4-5-20250929'], { family: 'sonnet' });
60
+ add(['claude-opus-4-5', 'claude-opus-4-5-20251101'], { family: 'opus', effortLevels: BASIC_EFFORT });
61
+ // 4.6 has a 1M API window, but subscription usage can need a client opt-in.
62
+ // Retain the existing conservative capacity policy instead of promising 1M.
63
+ for (const family of ['sonnet', 'opus']) add([`claude-${family}-4-6`], {
64
+ family, contextWindow: undefined, maxOutputTokens: 128000, autoMode: true,
65
+ thinkingTypes: BOTH, effortLevels: FOUR_EFFORT, assistantPrefill: false,
66
+ });
67
+ for (const version of ['4-7', '4-8']) add([`claude-opus-${version}`], {
68
+ family: 'opus', contextWindow: 1000000, maxOutputTokens: 128000,
69
+ capacityUpgradeSource: false, autoMode: true, thinkingTypes: ADAPTIVE,
70
+ effortLevels: FIVE_EFFORT, assistantPrefill: false, defaultSamplingOnly: true,
71
+ });
72
+ for (const family of ['sonnet', 'opus']) for (const version of ['5', '5-5']) {
73
+ const latest = version === '5-5';
74
+ add([`claude-${family}-${version}`], {
75
+ family, contextWindow: 1000000, maxOutputTokens: 128000,
76
+ capacityUpgradeSource: false, autoMode: true, sharedAuto: true,
77
+ thinkingTypes: latest ? Object.freeze(family === 'sonnet' ? ['adaptive', 'between_tools'] : ['adaptive']) : ADAPTIVE,
78
+ effortLevels: FIVE_EFFORT, forcedToolChoice: !latest,
79
+ assistantPrefill: false, defaultSamplingOnly: true,
80
+ midConversationSystem: family === 'opus' || latest,
81
+ perMessageEffort: family === 'opus' || latest,
82
+ taskBudget: family === 'opus' || latest,
83
+ // Preserve the existing explicit adaptation for Opus 5 as well as 5.5.
84
+ disabledThinkingAdaptation: family === 'opus' ? 'adaptive' : latest ? 'between_tools' : undefined,
85
+ });
86
+ }
87
+
88
+ export const MODEL_CATALOG = Object.freeze(registry);
89
+ /** @returns {Readonly<ModelCapabilities>|undefined} */
90
+ export const modelCapabilities = model => typeof model === 'string' && Object.hasOwn(MODEL_CATALOG, model)
91
+ ? MODEL_CATALOG[model] : undefined;
92
+ export const modelContextWindow = model => modelCapabilities(model)?.contextWindow;
93
+ export const hasNativeMillionContext = model => modelContextWindow(model) === 1000000;
94
+ export const canUpgradeContext = model => modelCapabilities(model)?.capacityUpgradeSource === true;
95
+ export const supportsToolReferences = model => modelCapabilities(model)?.toolReferences === true;
96
+ export const supportsAutoMode = model => modelCapabilities(model)?.autoMode === true;
@@ -1,6 +1,4 @@
1
- // Keep adaptations explicit: models in the same family can have different
2
- // thinking contracts. Preserve the existing adaptive Opus upgrade behavior.
3
- const ADAPTIVE_TARGETS = new Set(['claude-opus-5', 'claude-opus-5-5']);
1
+ import { modelCapabilities } from './model-catalog.mjs';
4
2
 
5
3
  function sonnetNeedsAdaptive(body) {
6
4
  const effort = body.output_config?.effort ?? 'high';
@@ -14,17 +12,18 @@ function sonnetNeedsAdaptive(body) {
14
12
  export function prepareRequest(body, model) {
15
13
  const request = { ...body, model };
16
14
  const adjustments = [];
15
+ const adaptation = modelCapabilities(model)?.disabledThinkingAdaptation;
17
16
  if (model !== body.model && body.model === 'claude-sonnet-5-5' && body.thinking?.type === 'between_tools'
18
- && Object.keys(body.thinking).length === 1 && ADAPTIVE_TARGETS.has(model)) {
17
+ && Object.keys(body.thinking).length === 1 && adaptation === 'adaptive') {
19
18
  request.thinking = { type: 'adaptive' };
20
19
  adjustments.push('adaptive_thinking_required');
21
20
  }
22
- if (model !== body.model && body.thinking?.type === 'disabled') {
23
- if (model === 'claude-sonnet-5-5') {
21
+ if (model !== body.model && body.thinking?.type === 'disabled' && Object.keys(body.thinking).length === 1) {
22
+ if (adaptation === 'between_tools') {
24
23
  const type = sonnetNeedsAdaptive(body) ? 'adaptive' : 'between_tools';
25
24
  request.thinking = { type };
26
25
  adjustments.push(type === 'adaptive' ? 'adaptive_thinking_required' : 'between_tools_thinking_required');
27
- } else if (ADAPTIVE_TARGETS.has(model)) {
26
+ } else if (adaptation === 'adaptive') {
28
27
  request.thinking = { type: 'adaptive' };
29
28
  adjustments.push('adaptive_thinking_required');
30
29
  }
@@ -1,5 +1,6 @@
1
1
  import { validateOllamaEndpoint, validateOllamaModel } from './ollama-models.mjs';
2
2
  import { buildState } from './prompt-state.mjs';
3
+ import { cancelResponseBody, readBoundedJson, MODEL_METADATA_LIMIT } from './bounded-json.mjs';
3
4
 
4
5
  // Bound UTF-8 bytes as well as serialized characters to keep local decision
5
6
  // excerpts small, including when the prompt contains non-ASCII text.
@@ -51,44 +52,20 @@ export function buildOllamaRequest(state, config) {
51
52
  return { model: config.ollamaModel, state, questions: OLLAMA_QUESTIONS, keep_alive: config.ollamaKeepAlive };
52
53
  }
53
54
 
54
- async function readJson(response, signal, limit = 64 * 1024) {
55
- const reader = response.body?.getReader();
56
- if (!reader) throw new Error('classifier_invalid_response');
57
- let total = 0;
58
- const chunks = [];
59
- const abort = () => { void reader.cancel(signal.reason).catch(() => {}); };
60
- signal.addEventListener('abort', abort, { once: true });
61
- try {
62
- while (true) {
63
- signal.throwIfAborted();
64
- const { value, done } = await reader.read();
65
- if (done) break;
66
- total += value.byteLength;
67
- if (total > limit) throw new Error('classifier_invalid_response');
68
- chunks.push(Buffer.from(value));
69
- }
70
- signal.throwIfAborted();
71
- return JSON.parse(Buffer.concat(chunks).toString('utf8'));
72
- } finally {
73
- signal.removeEventListener('abort', abort);
74
- await reader.cancel().catch(() => {}); reader.releaseLock();
75
- }
76
- }
77
-
78
55
  async function request(config, path, body, { fetchImpl, signal, responseLimit }) {
79
56
  const response = await fetchImpl(`${config.ollamaEndpoint}${path}`, {
80
57
  method: 'POST', redirect: 'error', signal,
81
58
  headers: { 'content-type': 'application/json' }, body: JSON.stringify(body),
82
59
  });
83
60
  if (!response.ok) {
84
- await response.body?.cancel();
61
+ cancelResponseBody(response);
85
62
  const needsVersion = path === '/v1/systemone' && response.status === 404;
86
63
  const error = new Error(needsVersion ? OLLAMA_VERSION_MESSAGE : 'classifier_http_error');
87
64
  if (needsVersion) error.code = 'OLLAMA_VERSION';
88
65
  error.classifierStatus = response.status;
89
66
  throw error;
90
67
  }
91
- return readJson(response, signal, responseLimit);
68
+ return readBoundedJson(response, { signal, limit: responseLimit });
92
69
  }
93
70
 
94
71
  const record = value => value !== null && typeof value === 'object' && !Array.isArray(value);
@@ -120,13 +97,18 @@ export async function checkLocalOllamaModel(config, options) {
120
97
  validateOllamaModel(config.ollamaModel);
121
98
  // /show includes tensor metadata and licenses that can exceed 64 KiB. Keep
122
99
  // its separate limit bounded while decision responses stay at 64 KiB.
123
- const model = await request(config, '/api/show', { model: config.ollamaModel }, { ...options, responseLimit: 1024 * 1024 });
100
+ const model = await request(config, '/api/show', { model: config.ollamaModel }, { ...options, responseLimit: MODEL_METADATA_LIMIT });
124
101
  if (!model || typeof model !== 'object' || model.remote_host || model.remote_model
125
102
  || typeof model.details?.parameter_size !== 'string' || !model.details.parameter_size) {
126
103
  throw new Error('classifier_invalid_response');
127
104
  }
128
105
  }
129
106
 
107
+ /**
108
+ * @param {any} state
109
+ * @param {any} config
110
+ * @param {{fetchImpl?:typeof fetch,signal?:AbortSignal}} [options]
111
+ */
130
112
  export async function evaluateOllama(state, config, { fetchImpl = fetch, signal } = {}) {
131
113
  let combined = signal;
132
114
  if (config.ollamaTimeoutMs !== 0) {
@@ -1,11 +1,10 @@
1
1
  import { execFile } from 'node:child_process';
2
- import { existsSync } from 'node:fs';
3
2
  import { createInterface } from 'node:readline';
4
3
  import { Writable } from 'node:stream';
5
4
  import { promisify } from 'node:util';
6
5
  import { CLIENT_PROFILES, readConfig, requireKeys, parseStopHookBlockCap, parseSessionLogDir } from './config.mjs';
7
6
  import { buildClaudeEnv, conflictingProviders, LOCAL_AUTH_HEADER } from './auth.mjs';
8
- import { getConfigPath, loadUserConfig, saveUserConfig } from './user-config.mjs';
7
+ import { CONFIG_KEYS, SECRET_CONFIG_KEYS, loadUserConfig, saveUserConfig } from './user-config.mjs';
9
8
  import { DEFAULT_OLLAMA_MODEL, validateOllamaModel } from './ollama-models.mjs';
10
9
  import { inspectOllama, setupOllama } from './ollama-setup.mjs';
11
10
 
@@ -40,19 +39,29 @@ export async function askSecret(label, { input = process.stdin, output = process
40
39
  export async function setup(args, {
41
40
  env = process.env, write = console.log, prompt = askSecret, fetchImpl = fetch, signal,
42
41
  } = {}) {
43
- let authMode = env.AUTOROUTER_AUTH_MODE ?? 'subscription';
44
- let clientProfile = env.AUTOROUTER_CLIENT_PROFILE ?? 'compatible';
45
- let evaluator = env.AUTOROUTER_EVALUATOR ?? 'jev';
42
+ if (signal?.aborted) throw new Error('Setup cancelled');
43
+ const loaded = loadUserConfig(env, { allowMissing: true });
44
+ const replace = args.includes('--replace');
45
+ const mergeExisting = loaded.exists && !replace;
46
+ // Updating one preference must not turn unrelated runtime overrides into
47
+ // saved defaults. First setup/replacement still supports environment-only
48
+ // configuration; an explicitly selected backend can use its supplied key.
49
+ const effectiveEnv = mergeExisting ? loaded.values : env;
50
+ let explicitEvaluator = false, explicitAuthMode = false;
51
+ let authMode = effectiveEnv.AUTOROUTER_AUTH_MODE ?? 'subscription';
52
+ let clientProfile = effectiveEnv.AUTOROUTER_CLIENT_PROFILE ?? 'compatible';
53
+ let evaluator = effectiveEnv.AUTOROUTER_EVALUATOR ?? 'jev';
46
54
  let model;
47
55
  let ollamaTimeoutMs;
48
- let stopHookBlockCap = env.CLAUDE_CODE_STOP_HOOK_BLOCK_CAP;
49
- let sessionLogDir = env.AUTOROUTER_SESSION_LOG_DIR;
56
+ let stopHookBlockCap = effectiveEnv.CLAUDE_CODE_STOP_HOOK_BLOCK_CAP;
57
+ let sessionLogDir = effectiveEnv.AUTOROUTER_SESSION_LOG_DIR;
58
+ let sessionLogMode = effectiveEnv.AUTOROUTER_SESSION_LOG_MODE;
50
59
  let pull = false;
51
- let overwrite = false;
60
+ let overwrite = replace;
52
61
  for (let i = 0; i < args.length; i++) {
53
- if (args[i] === '--auth-mode') authMode = args[++i];
62
+ if (args[i] === '--auth-mode') { authMode = args[++i]; explicitAuthMode = true; }
54
63
  else if (args[i] === '--client-profile') clientProfile = args[++i];
55
- else if (args[i] === '--evaluator') evaluator = args[++i];
64
+ else if (args[i] === '--evaluator') { evaluator = args[++i]; explicitEvaluator = true; }
56
65
  else if (args[i] === '--ollama-model') { model = args[++i]; if (model === undefined) throw new Error('--ollama-model requires a model tag'); }
57
66
  else if (args[i] === '--ollama-timeout-ms') {
58
67
  const value = args[++i];
@@ -69,9 +78,14 @@ export async function setup(args, {
69
78
  if (sessionLogDir === undefined || sessionLogDir.startsWith('--')) throw new Error('--session-log-dir requires a directory path');
70
79
  parseSessionLogDir(sessionLogDir, '--session-log-dir');
71
80
  }
81
+ else if (args[i] === '--session-log-mode') {
82
+ sessionLogMode = args[++i];
83
+ if (!['metadata', 'prompts'].includes(sessionLogMode)) throw new Error('--session-log-mode must be metadata or prompts');
84
+ }
72
85
  else if (args[i] === '--pull') pull = true;
73
86
  else if (args[i] === '--force') overwrite = true;
74
- else throw new Error('Usage: claude-autorouter setup [--auth-mode subscription|api-key] [--client-profile compatible|native|auto] [--evaluator jev|ollama] [--ollama-model TAG] [--ollama-timeout-ms N] [--stop-hook-block-cap N] [--session-log-dir DIR] [--pull] [--force]');
87
+ else if (args[i] === '--replace') continue;
88
+ else throw new Error('Usage: claude-autorouter setup [--auth-mode subscription|api-key] [--client-profile compatible|native|auto] [--evaluator jev|ollama] [--ollama-model TAG] [--ollama-timeout-ms N] [--stop-hook-block-cap N] [--session-log-dir DIR] [--session-log-mode metadata|prompts] [--pull] [--force|--replace]');
75
89
  }
76
90
  if (!['subscription', 'api-key'].includes(authMode)) throw new Error('--auth-mode must be subscription or api-key');
77
91
  if (!CLIENT_PROFILES.includes(clientProfile)) throw new Error('--client-profile must be compatible, native or auto');
@@ -79,32 +93,47 @@ export async function setup(args, {
79
93
  if (evaluator !== 'ollama' && (model !== undefined || ollamaTimeoutMs !== undefined || pull)) throw new Error('Ollama model, deadline and download options require --evaluator ollama');
80
94
  if (stopHookBlockCap !== undefined) stopHookBlockCap = parseStopHookBlockCap(stopHookBlockCap);
81
95
  if (sessionLogDir !== undefined) sessionLogDir = parseSessionLogDir(sessionLogDir) ?? '';
82
- const path = getConfigPath(env);
83
- if (!overwrite && existsSync(path)) throw new Error('AutoRouter configuration already exists. Use setup --force to replace it.');
96
+ const path = loaded.path;
97
+ if (!overwrite && loaded.exists) throw new Error('AutoRouter configuration already exists. Use setup --force to update it while preserving unrelated settings.');
84
98
  write(evaluator === 'ollama'
85
99
  ? 'AutoRouter evaluates bounded prompt excerpts locally with Ollama. Complete requests still go to Anthropic.'
86
100
  : 'AutoRouter sends bounded prompt excerpts to TypeSafe Jev and complete requests to Anthropic.');
87
- const values = { AUTOROUTER_AUTH_MODE: authMode, AUTOROUTER_CLIENT_PROFILE: clientProfile, AUTOROUTER_EVALUATOR: evaluator };
101
+ const values = replace ? {} : { ...loaded.values };
102
+ const selectedBackendKeys = explicitEvaluator ? CONFIG_KEYS.filter(key => evaluator === 'ollama'
103
+ ? key.startsWith('AUTOROUTER_OLLAMA_') : key.startsWith('AUTOROUTER_JEV_') || key === 'AUTOROUTER_MIN_CONFIDENCE') : [];
104
+ for (const key of mergeExisting ? selectedBackendKeys : CONFIG_KEYS) {
105
+ if (!SECRET_CONFIG_KEYS.includes(key) && env[key] !== undefined) values[key] = env[key];
106
+ }
107
+ Object.assign(values, { AUTOROUTER_AUTH_MODE: authMode, AUTOROUTER_CLIENT_PROFILE: clientProfile, AUTOROUTER_EVALUATOR: evaluator });
88
108
  if (stopHookBlockCap !== undefined) values.CLAUDE_CODE_STOP_HOOK_BLOCK_CAP = String(stopHookBlockCap);
89
109
  if (sessionLogDir !== undefined) values.AUTOROUTER_SESSION_LOG_DIR = sessionLogDir;
110
+ if (sessionLogMode !== undefined) values.AUTOROUTER_SESSION_LOG_MODE = sessionLogMode;
90
111
  if (evaluator === 'ollama') {
91
- values.AUTOROUTER_OLLAMA_MODEL = validateOllamaModel(model ?? env.AUTOROUTER_OLLAMA_MODEL ?? DEFAULT_OLLAMA_MODEL);
112
+ values.AUTOROUTER_OLLAMA_MODEL = validateOllamaModel(model
113
+ ?? (explicitEvaluator ? env.AUTOROUTER_OLLAMA_MODEL : undefined) ?? effectiveEnv.AUTOROUTER_OLLAMA_MODEL ?? DEFAULT_OLLAMA_MODEL);
92
114
  for (const key of ['AUTOROUTER_OLLAMA_URL', 'AUTOROUTER_OLLAMA_TIMEOUT_MS', 'AUTOROUTER_OLLAMA_KEEP_ALIVE']) {
93
- if (env[key] !== undefined) values[key] = env[key];
115
+ if ((!mergeExisting || explicitEvaluator) && env[key] !== undefined) values[key] = env[key];
94
116
  }
95
117
  if (ollamaTimeoutMs !== undefined) values.AUTOROUTER_OLLAMA_TIMEOUT_MS = ollamaTimeoutMs;
96
118
  }
97
119
  const keys = [...(evaluator === 'jev' ? ['TYPESAFE_API_KEY'] : []), ...(authMode === 'api-key' ? ['ANTHROPIC_API_KEY'] : [])];
120
+ // Reject invalid settings before inviting secret input or making local calls.
121
+ readConfig(values);
98
122
  for (const key of keys) {
99
- const value = (env[key]?.trim() || await prompt(key)).trim();
123
+ if (signal?.aborted) throw new Error('Setup cancelled');
124
+ const explicitlySelected = !mergeExisting || (key === 'TYPESAFE_API_KEY' ? explicitEvaluator : explicitAuthMode);
125
+ const value = ((explicitlySelected ? env[key]?.trim() : '') || effectiveEnv[key]?.trim() || env[key]?.trim() || await prompt(key)).trim();
126
+ if (signal?.aborted) throw new Error('Setup cancelled');
100
127
  if (!value || /[\r\n\0]/.test(value)) throw new Error(`${key} must be a nonempty, single-line key`);
101
128
  values[key] = value;
102
129
  }
103
130
  const config = readConfig(values);
104
131
  requireKeys(config);
132
+ if (signal?.aborted) throw new Error('Setup cancelled');
105
133
  if (config.clientProfile === 'auto') write('Auto-compatible profile: Sonnet/Opus task routing. Claude controls permission-mode availability and safety checks.');
106
134
  if (config.stopHookBlockCap !== undefined) write(stopHookCapText(config.stopHookBlockCap));
107
- if (config.sessionLogDir) write('Session decision logs enabled; files include up to 500 characters of user prompt text per decision.');
135
+ if (config.sessionLogDir) write(config.sessionLogMode === 'metadata' ? 'Session logs enabled with metadata only; prompt excerpts are omitted.'
136
+ : 'Session decision logs enabled; files include up to 500 characters of user prompt text per decision.');
108
137
  if (evaluator === 'ollama') {
109
138
  write(`Local evaluator: ${config.ollamaModel}; ${ollamaDeadlineText(config.ollamaTimeoutMs)}.`);
110
139
  const controller = new AbortController();
@@ -114,13 +143,37 @@ export async function setup(args, {
114
143
  finally { if (!signal) for (const name of ['SIGINT', 'SIGTERM']) process.removeListener(name, cancel); }
115
144
  }
116
145
  if (signal?.aborted) throw new Error('Setup cancelled');
117
- saveUserConfig(values, { env, overwrite });
146
+ saveUserConfig(values, { env, overwrite, expectedRevision: loaded.revision });
118
147
  write(`Saved ${authMode} configuration to ${path}`);
119
148
  write(`${keys.length ? 'Keys and settings are' : 'Settings are'} stored locally in this file with owner-only permissions. Environment variables take precedence.`);
120
149
  write('Next: claude-autorouter doctor, then claude-autorouter claude from your project.');
121
150
  }
122
151
 
123
- export async function doctor({ env = process.env, write = console.log, run = execute, fetchImpl = fetch, signal } = {}) {
152
+ export async function doctor({ env = process.env, write = console.log, run = execute, fetchImpl = fetch, signal,
153
+ evaluateLocal = false, json = false, diagnosticRunner } = {}) {
154
+ if (evaluateLocal) {
155
+ try {
156
+ const config = readConfig(loadUserConfig(env).env);
157
+ if (config.evaluator !== 'ollama') throw new Error('Local evaluation requires AUTOROUTER_EVALUATOR=ollama. It does not call Jev or Anthropic.');
158
+ const diagnostic = await import('./local-diagnostic.mjs');
159
+ const onProgress = json ? () => {} : progress => {
160
+ if (progress.event === 'preflight') write('Checking the local evaluator; no Claude authentication or cloud requests are used.');
161
+ else if (progress.event === 'startup') write('Preparing the local model with a separate 60-second startup deadline…');
162
+ else if (progress.event === 'case_start') write('Checking a synthetic routing case…');
163
+ else if (progress.event === 'case_complete') write('Synthetic routing case finished.');
164
+ };
165
+ const result = await (diagnosticRunner ?? diagnostic.runLocalDiagnostic)(config, { fetchImpl, signal, onProgress });
166
+ if (json) write(JSON.stringify(result));
167
+ else for (const line of diagnostic.formatLocalDiagnostic(result)) write(line);
168
+ return result.passed;
169
+ } catch (error) {
170
+ if (signal?.aborted) throw error;
171
+ if (json) write(JSON.stringify({ schema_version: 1, type: 'local_evaluator_diagnostic', passed: false,
172
+ error: { code: 'configuration_error', message: error.message } }));
173
+ else write(`FAIL ${error.message}`);
174
+ return false;
175
+ }
176
+ }
124
177
  let healthy = true;
125
178
  const report = (ok, message) => { if (!ok) healthy = false; write(`${ok ? 'OK' : 'FAIL'} ${message}`); };
126
179
  report(Number(process.versions.node.split('.')[0]) >= 22, `Node.js ${process.versions.node} (requires 22+)`);
@@ -134,14 +187,19 @@ export async function doctor({ env = process.env, write = console.log, run = exe
134
187
  requireKeys(config);
135
188
  report(true, `Configuration and required keys present (${config.authMode})`);
136
189
  } catch (error) {
137
- report(false, `${error.message}. Run claude-autorouter setup.`);
190
+ report(false, `${error.message}. Use claude-autorouter config show --check-all to inspect settings, or setup for first-time configuration.`);
138
191
  }
139
192
  for (const key of conflictingProviders(effectiveEnv)) {
140
193
  report(false, `Unset ${key}; AutoRouter uses the Anthropic Messages API`);
141
194
  }
142
195
  if (config?.stopHookBlockCap !== undefined) write(stopHookCapText(config.stopHookBlockCap));
196
+ if (config) {
197
+ write(`Routing profile: ${config.clientProfile}; Haiku ${config.models.haiku}; Sonnet ${config.models.sonnet}; Opus ${config.models.opus}.`);
198
+ if (config.evaluator === 'jev') write(`Evaluator: Jev ${config.jevModel}; routing deadline ${config.jevTimeoutMs} ms; confidence floor ${config.minConfidence}.`);
199
+ }
143
200
  if (config?.clientProfile === 'auto') write('Auto-compatible profile: Sonnet/Opus task routing. Claude controls permission-mode availability and safety checks.');
144
- if (config?.sessionLogDir) write('Session decision logs enabled; files include up to 500 characters of user prompt text per decision.');
201
+ if (config?.sessionLogDir) write(config.sessionLogMode === 'metadata' ? 'Session logs enabled with metadata only; prompt excerpts are omitted.'
202
+ : 'Session decision logs enabled; files include up to 500 characters of user prompt text per decision.');
145
203
  if (config?.evaluator === 'ollama') {
146
204
  write(`Local evaluator: ${config.ollamaModel}; ${ollamaDeadlineText(config.ollamaTimeoutMs)}.`);
147
205
  write('Model availability is checked below; classification speed and accuracy are not tested.');
@@ -149,7 +207,7 @@ export async function doctor({ env = process.env, write = console.log, run = exe
149
207
  const result = await inspectOllama(config, { fetchImpl, signal });
150
208
  report(result.installed, result.installed
151
209
  ? `Local Ollama model available (${result.model})`
152
- : `Ollama model missing (${result.model}). Run ollama pull ${result.model}, or setup --evaluator ollama --pull --force.`);
210
+ : `Ollama model missing (${result.model}). Run claude-autorouter setup --evaluator ollama --ollama-model ${result.model} --pull --force.`);
153
211
  } catch (error) {
154
212
  report(false, error.message);
155
213
  }
@@ -168,6 +226,7 @@ export async function doctor({ env = process.env, write = console.log, run = exe
168
226
  const { stdout } = await run('claude', ['--version'], { env: childEnv, timeout: 10000, maxBuffer: 64 * 1024 });
169
227
  const version = /\b\d+\.\d+\.\d+\b/.exec(stdout)?.[0];
170
228
  report(Boolean(version), version ? `Claude Code ${version}` : 'Could not recognize Claude Code version');
229
+ if (version) write(`Historical integration observations cover Claude Code 2.1.284 and 2.1.285; finding an executable does not certify its full compatibility${['2.1.284', '2.1.285'].includes(version) ? '.' : ' (installed version differs).'}`);
171
230
  claudeAvailable = Boolean(version);
172
231
  } catch {
173
232
  report(false, 'Claude Code unavailable. Install claude and ensure it is on PATH.');
@@ -0,0 +1,54 @@
1
+ const object = value => value !== null && typeof value === 'object' && !Array.isArray(value);
2
+ const text = value => typeof value === 'string' && value.length > 0;
3
+ const valid = Object.freeze({ valid: true });
4
+ const invalid = field => ({ valid: false, error: `Invalid Messages API request shape: ${field}` });
5
+
6
+ // Validate containers and scalar fields read by routing/excerpt extraction.
7
+ // This intentionally is not the provider's full schema: unknown fields and
8
+ // block/tool types remain untouched, and opaque inputs/schemas are not walked.
9
+ export function validateRequestShape(body) {
10
+ if (!object(body) || !text(body.model) || !body.model.trim()) return invalid('model');
11
+ if (!Array.isArray(body.messages)) return invalid('messages');
12
+ const pending = [];
13
+ const content = (value, field, depth = 0) => {
14
+ if (typeof value === 'string') return true;
15
+ if (!Array.isArray(value)) return false;
16
+ pending.push({ blocks: value, field, depth });
17
+ return true;
18
+ };
19
+ if (body.system !== undefined && !content(body.system, 'system')) return invalid('system');
20
+ for (const message of body.messages) {
21
+ if (!object(message) || !['user', 'assistant', 'system'].includes(message.role)) return invalid('messages');
22
+ if (!content(message.content, 'message content')) return invalid('message content');
23
+ if (message.output_config !== undefined && message.output_config !== null && !object(message.output_config)) return invalid('message output_config');
24
+ }
25
+ if (body.tools !== undefined && (!Array.isArray(body.tools) || body.tools.some(tool =>
26
+ !object(tool) || (tool.type !== undefined && tool.type !== null && !text(tool.type))
27
+ || (tool.name !== undefined && typeof tool.name !== 'string')))) return invalid('tools');
28
+ for (const field of ['thinking', 'tool_choice', 'output_config', 'context_management']) {
29
+ // Context management is explicitly nullable in the beta Messages schema.
30
+ if (field === 'context_management' && body[field] === null) continue;
31
+ if (body[field] !== undefined && !object(body[field])) return invalid(field);
32
+ }
33
+ for (const field of ['thinking', 'tool_choice']) {
34
+ if (body[field] !== undefined && !text(body[field].type)) return invalid(field);
35
+ }
36
+ // The Messages API permits zero for cache population without generation.
37
+ // Reviewed 2026-10-05: https://platform.claude.com/docs/en/api/beta/messages/create
38
+ if (body.max_tokens !== undefined && (!Number.isSafeInteger(body.max_tokens) || body.max_tokens < 0)) return invalid('max_tokens');
39
+ if (body.stream !== undefined && typeof body.stream !== 'boolean') return invalid('stream');
40
+ while (pending.length) {
41
+ const { blocks, field, depth } = pending.pop();
42
+ // Excerpt extraction recursively reads tool results. Bound this known
43
+ // content nesting before it can exhaust the stack; never echo field data.
44
+ if (depth > 32) return invalid('content nesting');
45
+ for (const block of blocks) {
46
+ if (!object(block) || !text(block.type)) return invalid(field);
47
+ if (block.type === 'text' && typeof block.text !== 'string') return invalid('text content');
48
+ if (block.type === 'tool_use' && block.name !== undefined && typeof block.name !== 'string') return invalid('tool name');
49
+ if (block.type === 'tool_result' && block.content !== undefined
50
+ && !content(block.content, 'tool result content', depth + 1)) return invalid('tool result content');
51
+ }
52
+ }
53
+ return valid;
54
+ }