claude-autorouter 0.3.7 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. package/.env.example +6 -3
  2. package/CODE_OF_CONDUCT.md +9 -0
  3. package/CONTRIBUTING.md +57 -0
  4. package/README.md +47 -70
  5. package/SECURITY.md +23 -0
  6. package/SUPPORT.md +18 -0
  7. package/bin/autorouter.mjs +40 -57
  8. package/docs/development.md +48 -2
  9. package/docs/hardware-benchmark.md +29 -0
  10. package/docs/hardware-comparison.md +55 -0
  11. package/docs/hardware-results-16gb.json +4002 -0
  12. package/docs/hardware-results-16gb.md +26 -0
  13. package/docs/hardware-results-64gb.json +4020 -0
  14. package/docs/reference.md +92 -37
  15. package/docs/releasing.md +79 -37
  16. package/docs/router-performance.json +1697 -0
  17. package/docs/router-performance.md +50 -0
  18. package/docs/status-performance.json +363 -0
  19. package/docs/status-performance.md +44 -0
  20. package/docs/subscription-integration.md +27 -0
  21. package/package.json +66 -10
  22. package/src/auto-routing.mjs +184 -24
  23. package/src/bounded-json.mjs +57 -0
  24. package/src/cli-help.mjs +90 -0
  25. package/src/config-command.mjs +158 -0
  26. package/src/config.mjs +53 -28
  27. package/src/contracts.mjs +123 -0
  28. package/src/evaluation-report.mjs +114 -0
  29. package/src/keychain.mjs +58 -0
  30. package/src/local-diagnostic.mjs +191 -0
  31. package/src/model-catalog.mjs +96 -0
  32. package/src/model-request.mjs +6 -7
  33. package/src/ollama-evaluator.mjs +9 -27
  34. package/src/onboarding.mjs +130 -26
  35. package/src/prompt-state.mjs +22 -7
  36. package/src/redaction.mjs +97 -0
  37. package/src/request-validation.mjs +54 -0
  38. package/src/response-observer.mjs +126 -18
  39. package/src/router.mjs +151 -61
  40. package/src/savings.mjs +74 -16
  41. package/src/server.mjs +79 -12
  42. package/src/session-history.mjs +262 -0
  43. package/src/session-log.mjs +9 -58
  44. package/src/status-state.mjs +110 -62
  45. package/src/statusline.mjs +57 -27
  46. package/src/telemetry-event.mjs +200 -0
  47. package/src/token-counter.mjs +3 -1
  48. package/src/turn-state.mjs +132 -0
  49. package/src/user-config.mjs +81 -10
@@ -0,0 +1,96 @@
1
+ // Exact Claude API IDs only. A name containing "sonnet" or "opus" is not
2
+ // evidence that a gateway alias implements that model's request contract.
3
+ export const MODEL_CATALOG_REVIEWED_AT = '2026-10-05';
4
+ export const MODEL_CAPABILITY_SOURCES = Object.freeze({
5
+ thinking: 'https://platform.claude.com/docs/en/build-with-claude/thinking-troubleshooting',
6
+ effort: 'https://platform.claude.com/docs/en/build-with-claude/effort',
7
+ sonnet55: 'https://platform.claude.com/docs/en/models/sonnet-5-5/migration-guide',
8
+ opus55: 'https://platform.claude.com/docs/en/models/opus-5-5/whats-new-opus-5-5',
9
+ auto: 'https://code.claude.com/docs/en/permission-modes',
10
+ subscriptionContext: 'https://code.claude.com/docs/en/model-config',
11
+ });
12
+
13
+ /**
14
+ * @typedef {object} ModelCapabilities
15
+ * @property {'haiku'|'sonnet'|'opus'} family
16
+ * @property {number} maxOutputTokens Standard Messages API limit, not Batch beta.
17
+ * @property {number|undefined} contextWindow Unambiguous window across supported auth modes.
18
+ * @property {boolean} capacityUpgradeSource Existing small/opt-in-window models.
19
+ * @property {boolean} toolReferences
20
+ * @property {boolean} autoMode Claude client Auto eligibility, not arbitrary request compatibility.
21
+ * @property {boolean} sharedAuto Verified modern Auto execution pair.
22
+ * @property {readonly string[]} thinkingTypes
23
+ * @property {readonly string[]} effortLevels
24
+ * @property {boolean} forcedToolChoice
25
+ * @property {boolean} assistantPrefill
26
+ * @property {boolean} defaultSamplingOnly
27
+ * @property {boolean} midConversationSystem
28
+ * @property {boolean} perMessageEffort
29
+ * @property {boolean} taskBudget
30
+ * @property {'adaptive'|'between_tools'|undefined} [disabledThinkingAdaptation]
31
+ * @property {string} reviewedAt
32
+ * @property {string} source
33
+ */
34
+
35
+ const BASIC_EFFORT = Object.freeze(['low', 'medium', 'high']);
36
+ const FOUR_EFFORT = Object.freeze([...BASIC_EFFORT, 'max']);
37
+ const FIVE_EFFORT = Object.freeze([...BASIC_EFFORT, 'xhigh', 'max']);
38
+ const EXTENDED = Object.freeze(['disabled', 'enabled']);
39
+ const BOTH = Object.freeze(['disabled', 'enabled', 'adaptive']);
40
+ const ADAPTIVE = Object.freeze(['disabled', 'adaptive']);
41
+ const registry = {};
42
+
43
+ function add(ids, facts) {
44
+ const canonical = ids[0].replace(/^claude-/, '');
45
+ const entry = Object.freeze({
46
+ contextWindow: 200000, maxOutputTokens: 64000, capacityUpgradeSource: true,
47
+ toolReferences: true, autoMode: false, sharedAuto: false,
48
+ thinkingTypes: EXTENDED, effortLevels: Object.freeze([]),
49
+ forcedToolChoice: true, assistantPrefill: true, defaultSamplingOnly: false,
50
+ midConversationSystem: false, perMessageEffort: false, taskBudget: false,
51
+ reviewedAt: MODEL_CATALOG_REVIEWED_AT,
52
+ source: `https://platform.claude.com/docs/en/models/${canonical}/overview`,
53
+ ...facts,
54
+ });
55
+ for (const id of ids) registry[id] = entry;
56
+ }
57
+
58
+ add(['claude-haiku-4-5', 'claude-haiku-4-5-20251001'], { family: 'haiku' });
59
+ add(['claude-sonnet-4-5', 'claude-sonnet-4-5-20250929'], { family: 'sonnet' });
60
+ add(['claude-opus-4-5', 'claude-opus-4-5-20251101'], { family: 'opus', effortLevels: BASIC_EFFORT });
61
+ // 4.6 has a 1M API window, but subscription usage can need a client opt-in.
62
+ // Retain the existing conservative capacity policy instead of promising 1M.
63
+ for (const family of ['sonnet', 'opus']) add([`claude-${family}-4-6`], {
64
+ family, contextWindow: undefined, maxOutputTokens: 128000, autoMode: true,
65
+ thinkingTypes: BOTH, effortLevels: FOUR_EFFORT, assistantPrefill: false,
66
+ });
67
+ for (const version of ['4-7', '4-8']) add([`claude-opus-${version}`], {
68
+ family: 'opus', contextWindow: 1000000, maxOutputTokens: 128000,
69
+ capacityUpgradeSource: false, autoMode: true, thinkingTypes: ADAPTIVE,
70
+ effortLevels: FIVE_EFFORT, assistantPrefill: false, defaultSamplingOnly: true,
71
+ });
72
+ for (const family of ['sonnet', 'opus']) for (const version of ['5', '5-5']) {
73
+ const latest = version === '5-5';
74
+ add([`claude-${family}-${version}`], {
75
+ family, contextWindow: 1000000, maxOutputTokens: 128000,
76
+ capacityUpgradeSource: false, autoMode: true, sharedAuto: true,
77
+ thinkingTypes: latest ? Object.freeze(family === 'sonnet' ? ['adaptive', 'between_tools'] : ['adaptive']) : ADAPTIVE,
78
+ effortLevels: FIVE_EFFORT, forcedToolChoice: !latest,
79
+ assistantPrefill: false, defaultSamplingOnly: true,
80
+ midConversationSystem: family === 'opus' || latest,
81
+ perMessageEffort: family === 'opus' || latest,
82
+ taskBudget: family === 'opus' || latest,
83
+ // Preserve the existing explicit adaptation for Opus 5 as well as 5.5.
84
+ disabledThinkingAdaptation: family === 'opus' ? 'adaptive' : latest ? 'between_tools' : undefined,
85
+ });
86
+ }
87
+
88
+ export const MODEL_CATALOG = Object.freeze(registry);
89
+ /** @returns {Readonly<ModelCapabilities>|undefined} */
90
+ export const modelCapabilities = model => typeof model === 'string' && Object.hasOwn(MODEL_CATALOG, model)
91
+ ? MODEL_CATALOG[model] : undefined;
92
+ export const modelContextWindow = model => modelCapabilities(model)?.contextWindow;
93
+ export const hasNativeMillionContext = model => modelContextWindow(model) === 1000000;
94
+ export const canUpgradeContext = model => modelCapabilities(model)?.capacityUpgradeSource === true;
95
+ export const supportsToolReferences = model => modelCapabilities(model)?.toolReferences === true;
96
+ export const supportsAutoMode = model => modelCapabilities(model)?.autoMode === true;
@@ -1,6 +1,4 @@
1
- // Keep adaptations explicit: models in the same family can have different
2
- // thinking contracts. Preserve the existing adaptive Opus upgrade behavior.
3
- const ADAPTIVE_TARGETS = new Set(['claude-opus-5', 'claude-opus-5-5']);
1
+ import { modelCapabilities } from './model-catalog.mjs';
4
2
 
5
3
  function sonnetNeedsAdaptive(body) {
6
4
  const effort = body.output_config?.effort ?? 'high';
@@ -14,17 +12,18 @@ function sonnetNeedsAdaptive(body) {
14
12
  export function prepareRequest(body, model) {
15
13
  const request = { ...body, model };
16
14
  const adjustments = [];
15
+ const adaptation = modelCapabilities(model)?.disabledThinkingAdaptation;
17
16
  if (model !== body.model && body.model === 'claude-sonnet-5-5' && body.thinking?.type === 'between_tools'
18
- && Object.keys(body.thinking).length === 1 && ADAPTIVE_TARGETS.has(model)) {
17
+ && Object.keys(body.thinking).length === 1 && adaptation === 'adaptive') {
19
18
  request.thinking = { type: 'adaptive' };
20
19
  adjustments.push('adaptive_thinking_required');
21
20
  }
22
- if (model !== body.model && body.thinking?.type === 'disabled') {
23
- if (model === 'claude-sonnet-5-5') {
21
+ if (model !== body.model && body.thinking?.type === 'disabled' && Object.keys(body.thinking).length === 1) {
22
+ if (adaptation === 'between_tools') {
24
23
  const type = sonnetNeedsAdaptive(body) ? 'adaptive' : 'between_tools';
25
24
  request.thinking = { type };
26
25
  adjustments.push(type === 'adaptive' ? 'adaptive_thinking_required' : 'between_tools_thinking_required');
27
- } else if (ADAPTIVE_TARGETS.has(model)) {
26
+ } else if (adaptation === 'adaptive') {
28
27
  request.thinking = { type: 'adaptive' };
29
28
  adjustments.push('adaptive_thinking_required');
30
29
  }
@@ -1,5 +1,6 @@
1
1
  import { validateOllamaEndpoint, validateOllamaModel } from './ollama-models.mjs';
2
2
  import { buildState } from './prompt-state.mjs';
3
+ import { cancelResponseBody, readBoundedJson, MODEL_METADATA_LIMIT } from './bounded-json.mjs';
3
4
 
4
5
  // Bound UTF-8 bytes as well as serialized characters to keep local decision
5
6
  // excerpts small, including when the prompt contains non-ASCII text.
@@ -51,44 +52,20 @@ export function buildOllamaRequest(state, config) {
51
52
  return { model: config.ollamaModel, state, questions: OLLAMA_QUESTIONS, keep_alive: config.ollamaKeepAlive };
52
53
  }
53
54
 
54
- async function readJson(response, signal, limit = 64 * 1024) {
55
- const reader = response.body?.getReader();
56
- if (!reader) throw new Error('classifier_invalid_response');
57
- let total = 0;
58
- const chunks = [];
59
- const abort = () => { void reader.cancel(signal.reason).catch(() => {}); };
60
- signal.addEventListener('abort', abort, { once: true });
61
- try {
62
- while (true) {
63
- signal.throwIfAborted();
64
- const { value, done } = await reader.read();
65
- if (done) break;
66
- total += value.byteLength;
67
- if (total > limit) throw new Error('classifier_invalid_response');
68
- chunks.push(Buffer.from(value));
69
- }
70
- signal.throwIfAborted();
71
- return JSON.parse(Buffer.concat(chunks).toString('utf8'));
72
- } finally {
73
- signal.removeEventListener('abort', abort);
74
- await reader.cancel().catch(() => {}); reader.releaseLock();
75
- }
76
- }
77
-
78
55
  async function request(config, path, body, { fetchImpl, signal, responseLimit }) {
79
56
  const response = await fetchImpl(`${config.ollamaEndpoint}${path}`, {
80
57
  method: 'POST', redirect: 'error', signal,
81
58
  headers: { 'content-type': 'application/json' }, body: JSON.stringify(body),
82
59
  });
83
60
  if (!response.ok) {
84
- await response.body?.cancel();
61
+ cancelResponseBody(response);
85
62
  const needsVersion = path === '/v1/systemone' && response.status === 404;
86
63
  const error = new Error(needsVersion ? OLLAMA_VERSION_MESSAGE : 'classifier_http_error');
87
64
  if (needsVersion) error.code = 'OLLAMA_VERSION';
88
65
  error.classifierStatus = response.status;
89
66
  throw error;
90
67
  }
91
- return readJson(response, signal, responseLimit);
68
+ return readBoundedJson(response, { signal, limit: responseLimit });
92
69
  }
93
70
 
94
71
  const record = value => value !== null && typeof value === 'object' && !Array.isArray(value);
@@ -120,13 +97,18 @@ export async function checkLocalOllamaModel(config, options) {
120
97
  validateOllamaModel(config.ollamaModel);
121
98
  // /show includes tensor metadata and licenses that can exceed 64 KiB. Keep
122
99
  // its separate limit bounded while decision responses stay at 64 KiB.
123
- const model = await request(config, '/api/show', { model: config.ollamaModel }, { ...options, responseLimit: 1024 * 1024 });
100
+ const model = await request(config, '/api/show', { model: config.ollamaModel }, { ...options, responseLimit: MODEL_METADATA_LIMIT });
124
101
  if (!model || typeof model !== 'object' || model.remote_host || model.remote_model
125
102
  || typeof model.details?.parameter_size !== 'string' || !model.details.parameter_size) {
126
103
  throw new Error('classifier_invalid_response');
127
104
  }
128
105
  }
129
106
 
107
+ /**
108
+ * @param {any} state
109
+ * @param {any} config
110
+ * @param {{fetchImpl?:typeof fetch,signal?:AbortSignal}} [options]
111
+ */
130
112
  export async function evaluateOllama(state, config, { fetchImpl = fetch, signal } = {}) {
131
113
  let combined = signal;
132
114
  if (config.ollamaTimeoutMs !== 0) {
@@ -1,11 +1,10 @@
1
1
  import { execFile } from 'node:child_process';
2
- import { existsSync } from 'node:fs';
3
2
  import { createInterface } from 'node:readline';
4
3
  import { Writable } from 'node:stream';
5
4
  import { promisify } from 'node:util';
6
5
  import { CLIENT_PROFILES, readConfig, requireKeys, parseStopHookBlockCap, parseSessionLogDir } from './config.mjs';
7
6
  import { buildClaudeEnv, conflictingProviders, LOCAL_AUTH_HEADER } from './auth.mjs';
8
- import { getConfigPath, loadUserConfig, saveUserConfig } from './user-config.mjs';
7
+ import { CONFIG_KEYS, SECRET_CONFIG_KEYS, SECRET_STORES, keychainRemovals, loadUserConfig, saveUserConfig } from './user-config.mjs';
9
8
  import { DEFAULT_OLLAMA_MODEL, validateOllamaModel } from './ollama-models.mjs';
10
9
  import { inspectOllama, setupOllama } from './ollama-setup.mjs';
11
10
 
@@ -38,21 +37,34 @@ export async function askSecret(label, { input = process.stdin, output = process
38
37
  }
39
38
 
40
39
  export async function setup(args, {
41
- env = process.env, write = console.log, prompt = askSecret, fetchImpl = fetch, signal,
40
+ env = process.env, write = console.log, prompt = askSecret, fetchImpl = fetch, signal, keychain,
41
+ platform = process.platform,
42
42
  } = {}) {
43
- let authMode = env.AUTOROUTER_AUTH_MODE ?? 'subscription';
44
- let clientProfile = env.AUTOROUTER_CLIENT_PROFILE ?? 'compatible';
45
- let evaluator = env.AUTOROUTER_EVALUATOR ?? 'jev';
43
+ if (signal?.aborted) throw new Error('Setup cancelled');
44
+ const store = keychain ? { keychain } : {};
45
+ const loaded = loadUserConfig(env, { allowMissing: true, ...store });
46
+ const replace = args.includes('--replace');
47
+ const mergeExisting = loaded.exists && !replace;
48
+ // Updating one preference must not turn unrelated runtime overrides into
49
+ // saved defaults. First setup/replacement still supports environment-only
50
+ // configuration; an explicitly selected backend can use its supplied key.
51
+ const effectiveEnv = mergeExisting ? loaded.values : env;
52
+ let explicitEvaluator = false, explicitAuthMode = false;
53
+ let authMode = effectiveEnv.AUTOROUTER_AUTH_MODE ?? 'subscription';
54
+ let clientProfile = effectiveEnv.AUTOROUTER_CLIENT_PROFILE ?? 'compatible';
55
+ let evaluator = effectiveEnv.AUTOROUTER_EVALUATOR ?? 'ollama';
46
56
  let model;
47
57
  let ollamaTimeoutMs;
48
- let stopHookBlockCap = env.CLAUDE_CODE_STOP_HOOK_BLOCK_CAP;
49
- let sessionLogDir = env.AUTOROUTER_SESSION_LOG_DIR;
58
+ let stopHookBlockCap = effectiveEnv.CLAUDE_CODE_STOP_HOOK_BLOCK_CAP;
59
+ let sessionLogDir = effectiveEnv.AUTOROUTER_SESSION_LOG_DIR;
60
+ let sessionLogMode = effectiveEnv.AUTOROUTER_SESSION_LOG_MODE;
61
+ let secretStore;
50
62
  let pull = false;
51
- let overwrite = false;
63
+ let overwrite = replace;
52
64
  for (let i = 0; i < args.length; i++) {
53
- if (args[i] === '--auth-mode') authMode = args[++i];
65
+ if (args[i] === '--auth-mode') { authMode = args[++i]; explicitAuthMode = true; }
54
66
  else if (args[i] === '--client-profile') clientProfile = args[++i];
55
- else if (args[i] === '--evaluator') evaluator = args[++i];
67
+ else if (args[i] === '--evaluator') { evaluator = args[++i]; explicitEvaluator = true; }
56
68
  else if (args[i] === '--ollama-model') { model = args[++i]; if (model === undefined) throw new Error('--ollama-model requires a model tag'); }
57
69
  else if (args[i] === '--ollama-timeout-ms') {
58
70
  const value = args[++i];
@@ -69,9 +81,18 @@ export async function setup(args, {
69
81
  if (sessionLogDir === undefined || sessionLogDir.startsWith('--')) throw new Error('--session-log-dir requires a directory path');
70
82
  parseSessionLogDir(sessionLogDir, '--session-log-dir');
71
83
  }
84
+ else if (args[i] === '--session-log-mode') {
85
+ sessionLogMode = args[++i];
86
+ if (!['metadata', 'prompts'].includes(sessionLogMode)) throw new Error('--session-log-mode must be metadata or prompts');
87
+ }
88
+ else if (args[i] === '--secret-store') {
89
+ secretStore = args[++i];
90
+ if (!SECRET_STORES.includes(secretStore)) throw new Error('--secret-store must be file or keychain');
91
+ }
72
92
  else if (args[i] === '--pull') pull = true;
73
93
  else if (args[i] === '--force') overwrite = true;
74
- else throw new Error('Usage: claude-autorouter setup [--auth-mode subscription|api-key] [--client-profile compatible|native|auto] [--evaluator jev|ollama] [--ollama-model TAG] [--ollama-timeout-ms N] [--stop-hook-block-cap N] [--session-log-dir DIR] [--pull] [--force]');
94
+ else if (args[i] === '--replace') continue;
95
+ else throw new Error('Usage: claude-autorouter setup [--auth-mode subscription|api-key] [--client-profile compatible|native|auto] [--evaluator jev|ollama] [--ollama-model TAG] [--ollama-timeout-ms N] [--stop-hook-block-cap N] [--session-log-dir DIR] [--session-log-mode metadata|prompts] [--secret-store file|keychain] [--pull] [--force|--replace]');
75
96
  }
76
97
  if (!['subscription', 'api-key'].includes(authMode)) throw new Error('--auth-mode must be subscription or api-key');
77
98
  if (!CLIENT_PROFILES.includes(clientProfile)) throw new Error('--client-profile must be compatible, native or auto');
@@ -79,69 +100,151 @@ export async function setup(args, {
79
100
  if (evaluator !== 'ollama' && (model !== undefined || ollamaTimeoutMs !== undefined || pull)) throw new Error('Ollama model, deadline and download options require --evaluator ollama');
80
101
  if (stopHookBlockCap !== undefined) stopHookBlockCap = parseStopHookBlockCap(stopHookBlockCap);
81
102
  if (sessionLogDir !== undefined) sessionLogDir = parseSessionLogDir(sessionLogDir) ?? '';
82
- const path = getConfigPath(env);
83
- if (!overwrite && existsSync(path)) throw new Error('AutoRouter configuration already exists. Use setup --force to replace it.');
103
+ const path = loaded.path;
104
+ if (!overwrite && loaded.exists) throw new Error('AutoRouter configuration already exists. Use setup --force to update it while preserving unrelated settings.');
84
105
  write(evaluator === 'ollama'
85
106
  ? 'AutoRouter evaluates bounded prompt excerpts locally with Ollama. Complete requests still go to Anthropic.'
86
107
  : 'AutoRouter sends bounded prompt excerpts to TypeSafe Jev and complete requests to Anthropic.');
87
- const values = { AUTOROUTER_AUTH_MODE: authMode, AUTOROUTER_CLIENT_PROFILE: clientProfile, AUTOROUTER_EVALUATOR: evaluator };
108
+ const values = replace ? {} : { ...loaded.values };
109
+ const selectedBackendKeys = explicitEvaluator ? CONFIG_KEYS.filter(key => evaluator === 'ollama'
110
+ ? key.startsWith('AUTOROUTER_OLLAMA_') : key.startsWith('AUTOROUTER_JEV_') || key === 'AUTOROUTER_MIN_CONFIDENCE') : [];
111
+ for (const key of mergeExisting ? selectedBackendKeys : CONFIG_KEYS) {
112
+ if (!SECRET_CONFIG_KEYS.includes(key) && env[key] !== undefined) values[key] = env[key];
113
+ }
114
+ Object.assign(values, { AUTOROUTER_AUTH_MODE: authMode, AUTOROUTER_CLIENT_PROFILE: clientProfile, AUTOROUTER_EVALUATOR: evaluator });
88
115
  if (stopHookBlockCap !== undefined) values.CLAUDE_CODE_STOP_HOOK_BLOCK_CAP = String(stopHookBlockCap);
89
116
  if (sessionLogDir !== undefined) values.AUTOROUTER_SESSION_LOG_DIR = sessionLogDir;
117
+ if (sessionLogMode !== undefined) values.AUTOROUTER_SESSION_LOG_MODE = sessionLogMode;
118
+ if (secretStore !== undefined) values.AUTOROUTER_SECRET_STORE = secretStore;
119
+ // New and rebuilt configurations keep keys in the macOS Keychain. An existing
120
+ // configuration keeps its store until the user moves it deliberately.
121
+ const defaultedStore = values.AUTOROUTER_SECRET_STORE === undefined && platform === 'darwin' && (!loaded.exists || replace);
122
+ if (defaultedStore) values.AUTOROUTER_SECRET_STORE = 'keychain';
90
123
  if (evaluator === 'ollama') {
91
- values.AUTOROUTER_OLLAMA_MODEL = validateOllamaModel(model ?? env.AUTOROUTER_OLLAMA_MODEL ?? DEFAULT_OLLAMA_MODEL);
124
+ values.AUTOROUTER_OLLAMA_MODEL = validateOllamaModel(model
125
+ ?? (explicitEvaluator ? env.AUTOROUTER_OLLAMA_MODEL : undefined) ?? effectiveEnv.AUTOROUTER_OLLAMA_MODEL ?? DEFAULT_OLLAMA_MODEL);
92
126
  for (const key of ['AUTOROUTER_OLLAMA_URL', 'AUTOROUTER_OLLAMA_TIMEOUT_MS', 'AUTOROUTER_OLLAMA_KEEP_ALIVE']) {
93
- if (env[key] !== undefined) values[key] = env[key];
127
+ if ((!mergeExisting || explicitEvaluator) && env[key] !== undefined) values[key] = env[key];
94
128
  }
95
129
  if (ollamaTimeoutMs !== undefined) values.AUTOROUTER_OLLAMA_TIMEOUT_MS = ollamaTimeoutMs;
96
130
  }
97
131
  const keys = [...(evaluator === 'jev' ? ['TYPESAFE_API_KEY'] : []), ...(authMode === 'api-key' ? ['ANTHROPIC_API_KEY'] : [])];
132
+ // Reject invalid settings before inviting secret input or making local calls.
133
+ readConfig(values);
134
+ if (values.AUTOROUTER_SECRET_STORE === 'keychain' && platform !== 'darwin') {
135
+ throw new Error('--secret-store keychain is available only on macOS');
136
+ }
98
137
  for (const key of keys) {
99
- const value = (env[key]?.trim() || await prompt(key)).trim();
138
+ if (signal?.aborted) throw new Error('Setup cancelled');
139
+ const explicitlySelected = !mergeExisting || (key === 'TYPESAFE_API_KEY' ? explicitEvaluator : explicitAuthMode);
140
+ const value = ((explicitlySelected ? env[key]?.trim() : '') || effectiveEnv[key]?.trim() || env[key]?.trim() || await prompt(key)).trim();
141
+ if (signal?.aborted) throw new Error('Setup cancelled');
100
142
  if (!value || /[\r\n\0]/.test(value)) throw new Error(`${key} must be a nonempty, single-line key`);
101
143
  values[key] = value;
102
144
  }
103
145
  const config = readConfig(values);
104
146
  requireKeys(config);
147
+ if (signal?.aborted) throw new Error('Setup cancelled');
105
148
  if (config.clientProfile === 'auto') write('Auto-compatible profile: Sonnet/Opus task routing. Claude controls permission-mode availability and safety checks.');
106
149
  if (config.stopHookBlockCap !== undefined) write(stopHookCapText(config.stopHookBlockCap));
107
- if (config.sessionLogDir) write('Session decision logs enabled; files include up to 500 characters of user prompt text per decision.');
150
+ if (config.sessionLogDir) write(config.sessionLogMode === 'metadata' ? 'Session logs enabled with metadata only; prompt excerpts are omitted.'
151
+ : 'Session decision logs enabled; files include up to 500 characters of user prompt text per decision.');
108
152
  if (evaluator === 'ollama') {
109
153
  write(`Local evaluator: ${config.ollamaModel}; ${ollamaDeadlineText(config.ollamaTimeoutMs)}.`);
110
154
  const controller = new AbortController();
111
155
  const cancel = () => controller.abort();
112
156
  if (!signal) for (const name of ['SIGINT', 'SIGTERM']) process.once(name, cancel);
113
157
  try { await setupOllama(config, { pull, warm: true, write, fetchImpl, signal: signal ?? controller.signal }); }
158
+ catch (error) {
159
+ // Local evaluation is the default, not a choice the user made, so a
160
+ // missing or stopped Ollama should say how to pick another evaluator.
161
+ if (!explicitEvaluator && !mergeExisting && !signal?.aborted) {
162
+ error.message = `${error.message} Local Ollama is the default evaluator; add --pull to download its model, or use TypeSafe Jev instead: claude-autorouter setup --evaluator jev`;
163
+ }
164
+ throw error;
165
+ }
114
166
  finally { if (!signal) for (const name of ['SIGINT', 'SIGTERM']) process.removeListener(name, cancel); }
115
167
  }
116
168
  if (signal?.aborted) throw new Error('Setup cancelled');
117
- saveUserConfig(values, { env, overwrite });
169
+ let savedStore = values.AUTOROUTER_SECRET_STORE ?? 'file';
170
+ try {
171
+ saveUserConfig(values, { env, overwrite, expectedRevision: loaded.revision,
172
+ removeSecrets: keychainRemovals(loaded, savedStore, { replace, values }), ...store });
173
+ } catch (error) {
174
+ // The default must not make setup impossible where the Keychain cannot be
175
+ // used (locked, headless). An explicit request still fails.
176
+ if (!defaultedStore || !keys.length || error?.code !== 'AUTOROUTER_CONFIG_ERROR' || !/Keychain/.test(error.message)) throw error;
177
+ write('The macOS Keychain is unavailable; saving keys in the private configuration file instead. Move them later with: claude-autorouter config set AUTOROUTER_SECRET_STORE keychain');
178
+ delete values.AUTOROUTER_SECRET_STORE;
179
+ savedStore = 'file';
180
+ saveUserConfig(values, { env, overwrite, expectedRevision: loaded.revision,
181
+ removeSecrets: keychainRemovals(loaded, savedStore, { replace, values }), ...store });
182
+ }
118
183
  write(`Saved ${authMode} configuration to ${path}`);
119
- write(`${keys.length ? 'Keys and settings are' : 'Settings are'} stored locally in this file with owner-only permissions. Environment variables take precedence.`);
184
+ if (keys.length && savedStore === 'keychain') {
185
+ write('Keys are stored in the macOS Keychain; settings are stored in this file with owner-only permissions. Environment variables take precedence.');
186
+ } else {
187
+ write(`${keys.length ? 'Keys and settings are' : 'Settings are'} stored locally in this file with owner-only permissions. Environment variables take precedence.`);
188
+ if (keys.length && platform === 'darwin' && savedStore === 'file' && !defaultedStore) write('Keys are plaintext in this file. Move them into the macOS Keychain: claude-autorouter config set AUTOROUTER_SECRET_STORE keychain');
189
+ }
120
190
  write('Next: claude-autorouter doctor, then claude-autorouter claude from your project.');
121
191
  }
122
192
 
123
- export async function doctor({ env = process.env, write = console.log, run = execute, fetchImpl = fetch, signal } = {}) {
193
+ export async function doctor({ env = process.env, write = console.log, run = execute, fetchImpl = fetch, signal,
194
+ evaluateLocal = false, json = false, diagnosticRunner, keychain, platform = process.platform } = {}) {
195
+ if (evaluateLocal) {
196
+ try {
197
+ const config = readConfig(loadUserConfig(env).env);
198
+ if (config.evaluator !== 'ollama') throw new Error('Local evaluation requires AUTOROUTER_EVALUATOR=ollama. It does not call Jev or Anthropic.');
199
+ const diagnostic = await import('./local-diagnostic.mjs');
200
+ const onProgress = json ? () => {} : progress => {
201
+ if (progress.event === 'preflight') write('Checking the local evaluator; no Claude authentication or cloud requests are used.');
202
+ else if (progress.event === 'startup') write('Preparing the local model with a separate 60-second startup deadline…');
203
+ else if (progress.event === 'case_start') write('Checking a synthetic routing case…');
204
+ else if (progress.event === 'case_complete') write('Synthetic routing case finished.');
205
+ };
206
+ const result = await (diagnosticRunner ?? diagnostic.runLocalDiagnostic)(config, { fetchImpl, signal, onProgress });
207
+ if (json) write(JSON.stringify(result));
208
+ else for (const line of diagnostic.formatLocalDiagnostic(result)) write(line);
209
+ return result.passed;
210
+ } catch (error) {
211
+ if (signal?.aborted) throw error;
212
+ if (json) write(JSON.stringify({ schema_version: 1, type: 'local_evaluator_diagnostic', passed: false,
213
+ error: { code: 'configuration_error', message: error.message } }));
214
+ else write(`FAIL ${error.message}`);
215
+ return false;
216
+ }
217
+ }
124
218
  let healthy = true;
125
219
  const report = (ok, message) => { if (!ok) healthy = false; write(`${ok ? 'OK' : 'FAIL'} ${message}`); };
126
220
  report(Number(process.versions.node.split('.')[0]) >= 22, `Node.js ${process.versions.node} (requires 22+)`);
127
221
  let config;
128
222
  let effectiveEnv = env;
129
223
  try {
130
- const loaded = loadUserConfig(env);
224
+ const loaded = loadUserConfig(env, keychain ? { keychain } : {});
131
225
  effectiveEnv = loaded.env;
132
226
  write(`Config: ${loaded.path}${loaded.exists ? '' : ' (absent; using environment)'}`);
227
+ if (loaded.secretStore === 'keychain') write(`Saved secrets: macOS Keychain (${loaded.keychainSecrets.length} found).`);
228
+ else if (platform === 'darwin' && SECRET_CONFIG_KEYS.some(key => Object.hasOwn(loaded.values, key))) {
229
+ write('WARN Saved keys are plaintext in the configuration file. Move them into the macOS Keychain: claude-autorouter config set AUTOROUTER_SECRET_STORE keychain');
230
+ }
133
231
  config = readConfig(effectiveEnv);
134
232
  requireKeys(config);
135
233
  report(true, `Configuration and required keys present (${config.authMode})`);
136
234
  } catch (error) {
137
- report(false, `${error.message}. Run claude-autorouter setup.`);
235
+ report(false, `${error.message}. Use claude-autorouter config show --check-all to inspect settings, or setup for first-time configuration.`);
138
236
  }
139
237
  for (const key of conflictingProviders(effectiveEnv)) {
140
238
  report(false, `Unset ${key}; AutoRouter uses the Anthropic Messages API`);
141
239
  }
142
240
  if (config?.stopHookBlockCap !== undefined) write(stopHookCapText(config.stopHookBlockCap));
241
+ if (config) {
242
+ write(`Routing profile: ${config.clientProfile}; Haiku ${config.models.haiku}; Sonnet ${config.models.sonnet}; Opus ${config.models.opus}.`);
243
+ if (config.evaluator === 'jev') write(`Evaluator: Jev ${config.jevModel}; routing deadline ${config.jevTimeoutMs} ms; confidence floor ${config.minConfidence}.`);
244
+ }
143
245
  if (config?.clientProfile === 'auto') write('Auto-compatible profile: Sonnet/Opus task routing. Claude controls permission-mode availability and safety checks.');
144
- if (config?.sessionLogDir) write('Session decision logs enabled; files include up to 500 characters of user prompt text per decision.');
246
+ if (config?.sessionLogDir) write(config.sessionLogMode === 'metadata' ? 'Session logs enabled with metadata only; prompt excerpts are omitted.'
247
+ : 'Session decision logs enabled; files include up to 500 characters of user prompt text per decision.');
145
248
  if (config?.evaluator === 'ollama') {
146
249
  write(`Local evaluator: ${config.ollamaModel}; ${ollamaDeadlineText(config.ollamaTimeoutMs)}.`);
147
250
  write('Model availability is checked below; classification speed and accuracy are not tested.');
@@ -149,7 +252,7 @@ export async function doctor({ env = process.env, write = console.log, run = exe
149
252
  const result = await inspectOllama(config, { fetchImpl, signal });
150
253
  report(result.installed, result.installed
151
254
  ? `Local Ollama model available (${result.model})`
152
- : `Ollama model missing (${result.model}). Run ollama pull ${result.model}, or setup --evaluator ollama --pull --force.`);
255
+ : `Ollama model missing (${result.model}). Run claude-autorouter setup --evaluator ollama --ollama-model ${result.model} --pull --force.`);
153
256
  } catch (error) {
154
257
  report(false, error.message);
155
258
  }
@@ -168,6 +271,7 @@ export async function doctor({ env = process.env, write = console.log, run = exe
168
271
  const { stdout } = await run('claude', ['--version'], { env: childEnv, timeout: 10000, maxBuffer: 64 * 1024 });
169
272
  const version = /\b\d+\.\d+\.\d+\b/.exec(stdout)?.[0];
170
273
  report(Boolean(version), version ? `Claude Code ${version}` : 'Could not recognize Claude Code version');
274
+ if (version) write(`Historical integration observations cover Claude Code 2.1.284 and 2.1.285; finding an executable does not certify its full compatibility${['2.1.284', '2.1.285'].includes(version) ? '.' : ' (installed version differs).'}`);
171
275
  claudeAvailable = Boolean(version);
172
276
  } catch {
173
277
  report(false, 'Claude Code unavailable. Install claude and ensure it is on PATH.');
@@ -1,3 +1,5 @@
1
+ import { redactSensitive } from './redaction.mjs';
2
+
1
3
  // Claude Code prepends these as separate text blocks. Ignore only complete
2
4
  // wrapper blocks; a user's text that mentions a tag or mixes it with a task
3
5
  // must remain part of the classifier's input.
@@ -51,6 +53,16 @@ function excerpt(text, length) {
51
53
 
52
54
  const textCost = text => JSON.stringify(text).length - 2;
53
55
 
56
+ // Redact before excerpting so a value cannot be split into an unrecognizable
57
+ // fragment. Very long text keeps only end windows wider than any retained
58
+ // excerpt; the margin also keeps a value cut at a window edge out of the result.
59
+ const REDACTION_MARGIN = 4096;
60
+ function classifierText(text, limit) {
61
+ const span = limit + REDACTION_MARGIN;
62
+ if (text.length > 2 * span) text = `${text.slice(0, span)}\n[... omitted ...]\n${text.slice(-span)}`;
63
+ return redactSensitive(text);
64
+ }
65
+
54
66
  // Budget serialized characters, including JSON escaping, rather than just
55
67
  // raw text length. The two ends keep both an initial instruction and a final
56
68
  // question visible when one long text block must be shortened.
@@ -145,6 +157,9 @@ export function promptExcerpt(body, maxChars = 500) {
145
157
  const blocks = typeof content === 'string' ? [{ type: 'text', text: content }]
146
158
  : Array.isArray(content) ? content : [];
147
159
  const characters = [];
160
+ // Collect past the retained length: redaction must see a value that
161
+ // straddles the final boundary, and may shorten the text before the cut.
162
+ const window = maxChars + REDACTION_MARGIN;
148
163
  let nonText = false;
149
164
  for (const block of blocks) {
150
165
  if (block?.type !== 'text' || typeof block.text !== 'string') {
@@ -157,12 +172,12 @@ export function promptExcerpt(body, maxChars = 500) {
157
172
  if (!/\S/.test(value) || isReminderBlock(value)) continue;
158
173
  if (characters.length) characters.push('\n');
159
174
  for (const character of value) {
160
- if (characters.length >= maxChars) break;
175
+ if (characters.length >= window) break;
161
176
  characters.push(character);
162
177
  }
163
- if (characters.length >= maxChars) break;
178
+ if (characters.length >= window) break;
164
179
  }
165
- if (characters.length) return characters.join('').toWellFormed();
180
+ if (characters.length) return [...redactSensitive(characters.join('').toWellFormed())].slice(0, maxChars).join('').toWellFormed();
166
181
  // An image/document-only human turn is a new task with no safe excerpt;
167
182
  // do not incorrectly label it with the preceding human task's text.
168
183
  if (nonText) return '';
@@ -196,9 +211,9 @@ export function buildState(body, limit = 12000) {
196
211
  const remaining = () => Math.max(0, limit - JSON.stringify(state).length);
197
212
  // Reserve more than half of the budget for the actual latest human task
198
213
  // before considering reminders, original instructions, or tool results.
199
- state.current_task = fitText(currentTask, Math.min(remaining(), Math.floor(limit * 0.55)));
200
- state.original_task = fitText(firstTask, Math.min(2000, Math.floor(remaining() * 0.3)));
201
- state.system = fitText(contentText(body.system, true), Math.min(1000, Math.floor(remaining() * 0.3)));
214
+ state.current_task = fitText(classifierText(currentTask, limit), Math.min(remaining(), Math.floor(limit * 0.55)));
215
+ state.original_task = fitText(classifierText(firstTask, limit), Math.min(2000, Math.floor(remaining() * 0.3)));
216
+ state.system = fitText(classifierText(contentText(body.system, true), limit), Math.min(1000, Math.floor(remaining() * 0.3)));
202
217
 
203
218
  for (let index = messages.length - 1; index >= 0 && state.recent_messages.length < 8; index--) {
204
219
  // current_task already contains this message; leave room for actual
@@ -210,7 +225,7 @@ export function buildState(body, limit = 12000) {
210
225
  const overhead = JSON.stringify(entry).length + (state.recent_messages.length ? 1 : 0);
211
226
  const budget = Math.min(3000, remaining() - overhead);
212
227
  if (budget < 1) break;
213
- entry.content = fitText(text, budget);
228
+ entry.content = fitText(classifierText(text, limit), budget);
214
229
  state.recent_messages.unshift(entry);
215
230
  }
216
231
  return state;
@@ -0,0 +1,97 @@
1
+ // @ts-check
2
+ // Pattern-based redaction for evaluator excerpts. Classification needs the
3
+ // shape of a task, not credentials or personal identifiers, so recognizable
4
+ // values are replaced before any excerpt leaves the inference path. This is a
5
+ // best-effort filter: unrecognized formats can remain, and code that merely
6
+ // resembles an assignment can be over-redacted. Every quantifier is bounded or
7
+ // excludes its delimiter, so scanning stays linear in the input length.
8
+
9
+ const marker = kind => `[REDACTED:${kind}]`;
10
+
11
+ function ibanValid(value) {
12
+ const compact = value.replace(/ /g, '');
13
+ if (compact.length < 15 || compact.length > 34) return false;
14
+ const rearranged = compact.slice(4) + compact.slice(0, 4);
15
+ let remainder = 0;
16
+ for (const character of rearranged) {
17
+ const digits = /[0-9]/.test(character) ? character : String(character.charCodeAt(0) - 55);
18
+ for (const digit of digits) remainder = (remainder * 10 + Number(digit)) % 97;
19
+ }
20
+ return remainder === 1;
21
+ }
22
+
23
+ function luhnValid(value) {
24
+ const digits = value.replace(/[ -]/g, '');
25
+ if (digits.length < 13 || digits.length > 19) return false;
26
+ let sum = 0;
27
+ for (let i = 0; i < digits.length; i++) {
28
+ let digit = Number(digits[digits.length - 1 - i]);
29
+ if (i % 2 === 1) { digit *= 2; if (digit > 9) digit -= 9; }
30
+ sum += digit;
31
+ }
32
+ return sum % 10 === 0;
33
+ }
34
+
35
+ const base64 = code => (code >= 48 && code <= 57) || (code >= 65 && code <= 90) || (code >= 97 && code <= 122)
36
+ || code === 43 || code === 47 || code === 61;
37
+
38
+ // Key material whose BEGIN line was cut off before the excerpt: walk back from
39
+ // each END line over whole base64 lines. A regex line repetition would
40
+ // backtrack quadratically on a long base64 run that is not followed by END.
41
+ function redactKeyTails(text) {
42
+ const end = /-----END [A-Z0-9 ]{0,40}PRIVATE KEY(?: BLOCK)?-----/g;
43
+ let output = '', copied = 0;
44
+ for (let match; (match = end.exec(text));) {
45
+ let start = match.index;
46
+ while (start > copied && text[start - 1] === '\n') {
47
+ let lineEnd = start - 1;
48
+ if (lineEnd > copied && text[lineEnd - 1] === '\r') lineEnd--;
49
+ let lineStart = lineEnd;
50
+ while (lineStart > copied && base64(text.charCodeAt(lineStart - 1))) lineStart--;
51
+ if (lineEnd - lineStart < 16 || (lineStart > copied && text[lineStart - 1] !== '\n')) break;
52
+ start = lineStart;
53
+ }
54
+ if (start === match.index) continue;
55
+ output += text.slice(copied, start) + marker('private_key');
56
+ copied = end.lastIndex;
57
+ }
58
+ return copied ? output + text.slice(copied) : text;
59
+ }
60
+
61
+ /** @type {ReadonlyArray<readonly [RegExp, string | ((...match: string[]) => string)] | ((text: string) => string)>} */
62
+ const RULES = Object.freeze([
63
+ // A block cut off by excerpting is still redacted through the end of text.
64
+ [/-----BEGIN [A-Z0-9 ]{0,40}PRIVATE KEY(?: BLOCK)?-----[\s\S]*?(?:-----END [A-Z0-9 ]{0,40}PRIVATE KEY(?: BLOCK)?-----|$)/g, marker('private_key')],
65
+ redactKeyTails,
66
+ [/\b([a-z][a-z0-9+.-]{1,20}:\/\/)[^\s:@/]{1,256}:[^\s@/]{1,256}@/gi, (_, scheme) => `${scheme}${marker('credentials')}@`],
67
+ [/\bsk-[A-Za-z0-9_-]{20,}/g, marker('secret')],
68
+ [/\b[rs]k_(?:live|test)_[A-Za-z0-9]{16,}/g, marker('secret')],
69
+ [/\b(?:AKIA|ASIA|ABIA|ACCA)[A-Z0-9]{16}\b/g, marker('secret')],
70
+ [/\b(?:gh[pousr]_[A-Za-z0-9]{36,}|github_pat_[A-Za-z0-9_]{22,})/g, marker('secret')],
71
+ [/\bglpat-[A-Za-z0-9_-]{20,}/g, marker('secret')],
72
+ [/\bxox[abposr]-[A-Za-z0-9-]{10,}/g, marker('secret')],
73
+ [/https:\/\/hooks\.slack\.com\/services\/[A-Za-z0-9/]{8,}/g, marker('secret')],
74
+ [/\bAIza[0-9A-Za-z_-]{35}/g, marker('secret')],
75
+ [/\bnpm_[A-Za-z0-9]{36}/g, marker('secret')],
76
+ [/\beyJ[A-Za-z0-9_-]{8,}\.eyJ[A-Za-z0-9_-]{8,}\.[A-Za-z0-9_-]{8,}/g, marker('secret')],
77
+ [/\b((?:proxy-)?authorization["']?[ \t]{0,8}[:=][ \t]{0,8}["']?)(?:(Bearer|Basic|Token|Digest)[ \t]+)?[^\s"',;]{4,}/gi,
78
+ (_, prefix, scheme) => `${prefix}${scheme ? `${scheme} ` : ''}${marker('secret')}`],
79
+ [/\b(Bearer[ \t]+)[A-Za-z0-9._~+/-]{16,}=*/g, (_, prefix) => `${prefix}${marker('secret')}`],
80
+ // Keep the setting name: it tells the classifier what kind of work this is.
81
+ [/\b([A-Za-z0-9_.-]{0,40}(?:passw(?:or)?d|pwd|secret|token|api[_-]?key|access[_-]?key|private[_-]?key|credentials?)[A-Za-z0-9_.-]{0,40})(["']?[ \t]{0,8}[:=][ \t]{0,8})(["']?)[^\s"',;]{4,}/gi,
82
+ (_, name, separator, quote) => `${name}${separator}${quote}${marker('secret')}`],
83
+ [/\b[A-Za-z0-9._%+-]{1,64}@[A-Za-z0-9.-]{1,253}\.[A-Za-z]{2,24}\b/g, marker('email')],
84
+ [/\b[A-Z]{2}[0-9]{2}(?: ?[A-Z0-9]{4}){2,7}(?: ?[A-Z0-9]{1,4})?\b/g, match => ibanValid(match) ? marker('iban') : match],
85
+ // Major card networks only, so millisecond timestamps and IDs survive.
86
+ [/\b(?:4|5[1-5]|2[2-7]|3[47]|6)(?:[0-9][ -]?){11,17}[0-9]\b/g, match => luhnValid(match) ? marker('card') : match],
87
+ ]);
88
+
89
+ /** @param {string} text */
90
+ export function redactSensitive(text) {
91
+ let result = text;
92
+ for (const rule of RULES) {
93
+ // @ts-ignore String.replace accepts either replacement form.
94
+ result = typeof rule === 'function' ? rule(result) : result.replace(rule[0], rule[1]);
95
+ }
96
+ return result;
97
+ }