claude-autorouter 0.3.7 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. package/.env.example +6 -3
  2. package/CODE_OF_CONDUCT.md +9 -0
  3. package/CONTRIBUTING.md +57 -0
  4. package/README.md +47 -70
  5. package/SECURITY.md +23 -0
  6. package/SUPPORT.md +18 -0
  7. package/bin/autorouter.mjs +40 -57
  8. package/docs/development.md +48 -2
  9. package/docs/hardware-benchmark.md +29 -0
  10. package/docs/hardware-comparison.md +55 -0
  11. package/docs/hardware-results-16gb.json +4002 -0
  12. package/docs/hardware-results-16gb.md +26 -0
  13. package/docs/hardware-results-64gb.json +4020 -0
  14. package/docs/reference.md +92 -37
  15. package/docs/releasing.md +79 -37
  16. package/docs/router-performance.json +1697 -0
  17. package/docs/router-performance.md +50 -0
  18. package/docs/status-performance.json +363 -0
  19. package/docs/status-performance.md +44 -0
  20. package/docs/subscription-integration.md +27 -0
  21. package/package.json +66 -10
  22. package/src/auto-routing.mjs +184 -24
  23. package/src/bounded-json.mjs +57 -0
  24. package/src/cli-help.mjs +90 -0
  25. package/src/config-command.mjs +158 -0
  26. package/src/config.mjs +53 -28
  27. package/src/contracts.mjs +123 -0
  28. package/src/evaluation-report.mjs +114 -0
  29. package/src/keychain.mjs +58 -0
  30. package/src/local-diagnostic.mjs +191 -0
  31. package/src/model-catalog.mjs +96 -0
  32. package/src/model-request.mjs +6 -7
  33. package/src/ollama-evaluator.mjs +9 -27
  34. package/src/onboarding.mjs +130 -26
  35. package/src/prompt-state.mjs +22 -7
  36. package/src/redaction.mjs +97 -0
  37. package/src/request-validation.mjs +54 -0
  38. package/src/response-observer.mjs +126 -18
  39. package/src/router.mjs +151 -61
  40. package/src/savings.mjs +74 -16
  41. package/src/server.mjs +79 -12
  42. package/src/session-history.mjs +262 -0
  43. package/src/session-log.mjs +9 -58
  44. package/src/status-state.mjs +110 -62
  45. package/src/statusline.mjs +57 -27
  46. package/src/telemetry-event.mjs +200 -0
  47. package/src/token-counter.mjs +3 -1
  48. package/src/turn-state.mjs +132 -0
  49. package/src/user-config.mjs +81 -10
package/src/config.mjs CHANGED
@@ -1,15 +1,12 @@
1
+ // @ts-check
1
2
  import { DEFAULT_OLLAMA_MODEL, defaultOllamaTimeoutMs, validateOllamaEndpoint, validateOllamaModel } from './ollama-models.mjs';
2
3
  import { resolve } from 'node:path';
4
+ import { supportsAutoMode } from './model-catalog.mjs';
3
5
 
6
+ /** @type {import('./contracts.mjs').Tier[]} */
4
7
  export const TIERS = ['haiku', 'sonnet', 'opus'];
8
+ /** @type {import('./contracts.mjs').ClientProfile[]} */
5
9
  export const CLIENT_PROFILES = ['compatible', 'native', 'auto'];
6
- // Exact Anthropic models whose Auto permission-mode support is documented.
7
- // Custom aliases are not proof of the capabilities of their upstream model.
8
- const AUTO_MODE_MODELS = new Set([
9
- 'claude-sonnet-4-6', 'claude-sonnet-5', 'claude-sonnet-5-5',
10
- 'claude-opus-4-6', 'claude-opus-4-7', 'claude-opus-4-8', 'claude-opus-5', 'claude-opus-5-5',
11
- ]);
12
-
13
10
  export function parseStopHookBlockCap(value, name = 'CLAUDE_CODE_STOP_HOOK_BLOCK_CAP') {
14
11
  if (!['string', 'number'].includes(typeof value)
15
12
  || (typeof value === 'string' && !/^[0-9]+$/.test(value.trim()))
@@ -28,15 +25,19 @@ export function parseSessionLogDir(value, name = 'AUTOROUTER_SESSION_LOG_DIR') {
28
25
  }
29
26
 
30
27
  function number(env, key, fallback, min, max, integer = true) {
31
- const value = Number(env[key] ?? fallback);
32
- if (!Number.isFinite(value) || value < min || value > max || (integer && !Number.isInteger(value))) {
28
+ const raw = env[key] === undefined ? fallback : env[key];
29
+ const pattern = integer ? /^[0-9]+$/ : /^(?:[0-9]+(?:\.[0-9]+)?|\.[0-9]+)$/;
30
+ const value = Number(raw);
31
+ if (!['string', 'number'].includes(typeof raw) || (typeof raw === 'string' && !pattern.test(raw.trim()))
32
+ || !Number.isFinite(value) || value < min || value > max || (integer && !Number.isSafeInteger(value))) {
33
33
  throw new Error(`${key} must be ${integer ? 'an integer' : 'a number'} between ${min} and ${max}`);
34
34
  }
35
35
  return value;
36
36
  }
37
37
 
38
38
  function endpoint(value, name) {
39
- const url = new URL(value);
39
+ let url;
40
+ try { url = new URL(value); } catch { throw new Error(`${name} must be a valid URL`); }
40
41
  const local = ['127.0.0.1', '[::1]', 'localhost'].includes(url.hostname);
41
42
  if ((url.protocol !== 'https:' && !(local && url.protocol === 'http:')) || url.username || url.password || url.search || url.hash) {
42
43
  throw new Error(`${name} must use HTTPS (HTTP is allowed on loopback) with no credentials, query, or fragment`);
@@ -44,37 +45,59 @@ function endpoint(value, name) {
44
45
  return url.href.replace(/\/$/, '');
45
46
  }
46
47
 
47
- export function readConfig(env = process.env) {
48
- const evaluator = env.AUTOROUTER_EVALUATOR ?? 'jev';
49
- if (!['jev', 'ollama'].includes(evaluator)) throw new Error('AUTOROUTER_EVALUATOR must be jev or ollama');
50
- const ollamaModel = validateOllamaModel(env.AUTOROUTER_OLLAMA_MODEL ?? DEFAULT_OLLAMA_MODEL);
51
- const rawOllamaTimeout = env.AUTOROUTER_OLLAMA_TIMEOUT_MS;
48
+ function modelName(value, name) {
49
+ if (typeof value !== 'string' || !value.trim() || /[\u0000-\u001f\u007f-\u009f]/.test(value)) {
50
+ throw new Error(`${name} must be a nonempty model name without control characters`);
51
+ }
52
+ return value;
53
+ }
54
+
55
+ /**
56
+ * @param {Record<string,string|undefined>} [env]
57
+ * @param {{validateAll?:boolean}} [options]
58
+ * @returns {import('./contracts.mjs').RouterConfig}
59
+ */
60
+ export function readConfig(env = process.env, { validateAll = false } = {}) {
61
+ const sessionLogMode = env.AUTOROUTER_SESSION_LOG_MODE === undefined ? 'prompts' : env.AUTOROUTER_SESSION_LOG_MODE;
62
+ if (sessionLogMode !== 'metadata' && sessionLogMode !== 'prompts') throw new Error('AUTOROUTER_SESSION_LOG_MODE must be metadata or prompts');
63
+ for (const key of ['AUTOROUTER_STATUSLINE', 'AUTOROUTER_DEBUG']) {
64
+ if (env[key] !== undefined && !['0', '1'].includes(env[key])) throw new Error(`${key} must be 0 or 1`);
65
+ }
66
+ const evaluator = env.AUTOROUTER_EVALUATOR ?? 'ollama';
67
+ if (evaluator !== 'jev' && evaluator !== 'ollama') throw new Error('AUTOROUTER_EVALUATOR must be jev or ollama');
68
+ // A stale inactive backend must not stop the selected evaluator. Config
69
+ // inspection can request validation of both providers explicitly.
70
+ const ollamaEnv = evaluator === 'ollama' || validateAll ? env : {};
71
+ const jevEnv = evaluator === 'jev' || validateAll ? env : {};
72
+ const ollamaModel = validateOllamaModel(ollamaEnv.AUTOROUTER_OLLAMA_MODEL ?? DEFAULT_OLLAMA_MODEL);
73
+ const rawOllamaTimeout = ollamaEnv.AUTOROUTER_OLLAMA_TIMEOUT_MS;
52
74
  // Zero is an explicit opt-out. Reject blanks, coercible non-numbers, and
53
75
  // non-integer strings (including exponents that could underflow to zero).
54
76
  if (rawOllamaTimeout !== undefined && (!['string', 'number'].includes(typeof rawOllamaTimeout)
55
77
  || (typeof rawOllamaTimeout === 'string' && !/^[0-9]+$/.test(rawOllamaTimeout.trim())))) {
56
78
  throw new Error('AUTOROUTER_OLLAMA_TIMEOUT_MS must be an integer between 0 and 30000 (0 disables the runtime deadline)');
57
79
  }
58
- const ollamaKeepAlive = env.AUTOROUTER_OLLAMA_KEEP_ALIVE ?? '5m';
80
+ const ollamaKeepAlive = ollamaEnv.AUTOROUTER_OLLAMA_KEEP_ALIVE ?? '5m';
59
81
  if (!/^(?:0|[1-9]\d{0,3}(?:s|m|h))$/.test(ollamaKeepAlive)) {
60
82
  throw new Error('AUTOROUTER_OLLAMA_KEEP_ALIVE must be 0 or a positive duration such as 5m');
61
83
  }
62
84
  const authMode = env.AUTOROUTER_AUTH_MODE ?? 'api-key';
63
- if (!['api-key', 'subscription'].includes(authMode)) {
85
+ if (authMode !== 'api-key' && authMode !== 'subscription') {
64
86
  throw new Error('AUTOROUTER_AUTH_MODE must be api-key or subscription');
65
87
  }
66
88
  const clientProfile = env.AUTOROUTER_CLIENT_PROFILE ?? 'compatible';
67
- if (!CLIENT_PROFILES.includes(clientProfile)) {
89
+ if (clientProfile !== 'compatible' && clientProfile !== 'native' && clientProfile !== 'auto') {
68
90
  throw new Error('AUTOROUTER_CLIENT_PROFILE must be compatible, native or auto');
69
91
  }
70
92
  const models = {
71
- haiku: env.AUTOROUTER_HAIKU_MODEL ?? 'claude-haiku-4-5-20251001',
72
- sonnet: env.AUTOROUTER_SONNET_MODEL ?? (clientProfile === 'auto' ? 'claude-sonnet-5-5' : 'claude-sonnet-5'),
73
- opus: env.AUTOROUTER_OPUS_MODEL ?? 'claude-opus-5-5',
93
+ haiku: env.AUTOROUTER_HAIKU_MODEL === undefined ? 'claude-haiku-4-5-20251001' : env.AUTOROUTER_HAIKU_MODEL,
94
+ sonnet: env.AUTOROUTER_SONNET_MODEL === undefined ? (clientProfile === 'auto' ? 'claude-sonnet-5-5' : 'claude-sonnet-5') : env.AUTOROUTER_SONNET_MODEL,
95
+ opus: env.AUTOROUTER_OPUS_MODEL === undefined ? 'claude-opus-5-5' : env.AUTOROUTER_OPUS_MODEL,
74
96
  };
97
+ for (const tier of TIERS) modelName(models[tier], `AUTOROUTER_${tier.toUpperCase()}_MODEL`);
75
98
  if (clientProfile === 'auto') {
76
99
  for (const tier of ['sonnet', 'opus']) {
77
- if (!AUTO_MODE_MODELS.has(models[tier])) {
100
+ if (!supportsAutoMode(models[tier])) {
78
101
  throw new Error(`AUTOROUTER_${tier.toUpperCase()}_MODEL must be a known Auto-mode-capable Sonnet or Opus model for the auto profile`);
79
102
  }
80
103
  }
@@ -88,24 +111,25 @@ export function readConfig(env = process.env) {
88
111
  authMode,
89
112
  clientProfile,
90
113
  sessionLogDir: parseSessionLogDir(env.AUTOROUTER_SESSION_LOG_DIR),
114
+ sessionLogMode,
91
115
  stopHookBlockCap: env.CLAUDE_CODE_STOP_HOOK_BLOCK_CAP === undefined
92
116
  ? undefined : parseStopHookBlockCap(env.CLAUDE_CODE_STOP_HOOK_BLOCK_CAP),
93
117
  anthropicKey: authMode === 'api-key' ? env.ANTHROPIC_API_KEY : undefined,
94
118
  jevKey: env.TYPESAFE_API_KEY,
95
119
  localToken: env.AUTOROUTER_TOKEN,
96
120
  upstream,
97
- jevEndpoint: endpoint(env.AUTOROUTER_JEV_URL ?? 'https://api.typesafe.ai/v1/systemone', 'AUTOROUTER_JEV_URL'),
98
- jevModel: env.AUTOROUTER_JEV_MODEL ?? 'jev-latest',
99
- ollamaEndpoint: validateOllamaEndpoint(env.AUTOROUTER_OLLAMA_URL ?? 'http://127.0.0.1:11434'),
121
+ jevEndpoint: endpoint(jevEnv.AUTOROUTER_JEV_URL ?? 'https://api.typesafe.ai/v1/systemone', 'AUTOROUTER_JEV_URL'),
122
+ jevModel: modelName(jevEnv.AUTOROUTER_JEV_MODEL === undefined ? 'jev-latest' : jevEnv.AUTOROUTER_JEV_MODEL, 'AUTOROUTER_JEV_MODEL'),
123
+ ollamaEndpoint: validateOllamaEndpoint(ollamaEnv.AUTOROUTER_OLLAMA_URL ?? 'http://127.0.0.1:11434'),
100
124
  ollamaModel,
101
- ollamaTimeoutMs: number(env, 'AUTOROUTER_OLLAMA_TIMEOUT_MS', defaultOllamaTimeoutMs(ollamaModel), 0, 30000),
125
+ ollamaTimeoutMs: number(ollamaEnv, 'AUTOROUTER_OLLAMA_TIMEOUT_MS', defaultOllamaTimeoutMs(ollamaModel), 0, 30000),
102
126
  ollamaStateChars: 3000,
103
127
  ollamaKeepAlive,
104
128
  models,
105
129
  port: number(env, 'AUTOROUTER_PORT', 8787, 0, 65535),
106
- jevTimeoutMs: number(env, 'AUTOROUTER_JEV_TIMEOUT_MS', 1500, 1, 10000),
130
+ jevTimeoutMs: number(jevEnv, 'AUTOROUTER_JEV_TIMEOUT_MS', 1500, 1, 10000),
107
131
  tokenCountTimeoutMs: number(env, 'AUTOROUTER_TOKEN_COUNT_TIMEOUT_MS', 1500, 1, 10000),
108
- minConfidence: number(env, 'AUTOROUTER_MIN_CONFIDENCE', 0.75, 0, 1, false),
132
+ minConfidence: number(jevEnv, 'AUTOROUTER_MIN_CONFIDENCE', 0.75, 0, 1, false),
109
133
  stateChars: 12000,
110
134
  maxBodyBytes: 32 * 1024 * 1024,
111
135
  cacheEntries: 1000,
@@ -115,6 +139,7 @@ export function readConfig(env = process.env) {
115
139
  };
116
140
  }
117
141
 
142
+ /** @param {import('./contracts.mjs').RouterConfig} config */
118
143
  export function requireKeys(config) {
119
144
  const keys = config.evaluator === 'ollama' ? [] : [['TYPESAFE_API_KEY', config.jevKey]];
120
145
  if (config.authMode === 'api-key') keys.push(['ANTHROPIC_API_KEY', config.anthropicKey]);
@@ -0,0 +1,123 @@
1
+ // Development-time contracts only; JSDoc is erased by JavaScript engines.
2
+ /** @typedef {'haiku'|'sonnet'|'opus'} Tier */
3
+ /** @typedef {'jev'|'ollama'} Evaluator */
4
+ /** @typedef {'compatible'|'native'|'auto'} ClientProfile */
5
+ /** @typedef {'jev'|'ollama'|'cache'|'fallback'|'passthrough'} DecisionSource */
6
+ /** @typedef {'timeout'|'http_error'|'invalid_response'|'network_error'|'capacity_exhausted'} ClassifierError */
7
+ /** @typedef {'request_start'|'route'|'upstream_response'|'upstream_model'|'upstream_usage'|'upstream_error'|'request_complete'|'request_error'|'request_cancelled'} LifecycleName */
8
+ /** @typedef {'selected'|'confirmed'|'unknown'|'capacity_exhausted'} ContinuityState */
9
+ /**
10
+ * @typedef {object} RouterConfig
11
+ * @property {Evaluator} evaluator
12
+ * @property {'api-key'|'subscription'} authMode
13
+ * @property {ClientProfile} clientProfile
14
+ * @property {string|undefined} sessionLogDir
15
+ * @property {'metadata'|'prompts'} sessionLogMode
16
+ * @property {number|undefined} stopHookBlockCap
17
+ * @property {string|undefined} anthropicKey
18
+ * @property {string|undefined} jevKey
19
+ * @property {string|undefined} localToken
20
+ * @property {string} upstream
21
+ * @property {string} jevEndpoint
22
+ * @property {string} jevModel
23
+ * @property {string} ollamaEndpoint
24
+ * @property {string} ollamaModel
25
+ * @property {number} ollamaTimeoutMs
26
+ * @property {number} ollamaStateChars
27
+ * @property {string} ollamaKeepAlive
28
+ * @property {Record<Tier,string>} models
29
+ * @property {number} port
30
+ * @property {number} jevTimeoutMs
31
+ * @property {number} tokenCountTimeoutMs
32
+ * @property {number} [tokenCountCacheEntries]
33
+ * @property {number} [tokenCountCacheTtlMs]
34
+ * @property {number} minConfidence
35
+ * @property {number} stateChars
36
+ * @property {number} maxBodyBytes
37
+ * @property {number} cacheEntries
38
+ * @property {number} cacheTtlMs
39
+ * @property {number} turnTtlMs
40
+ * @property {number|undefined} [turnEntries]
41
+ * @property {number} upstreamTimeoutMs
42
+ */
43
+ /**
44
+ * @typedef {object} ClassifierDecision
45
+ * @property {Tier} tier
46
+ * @property {Tier} [classified_tier]
47
+ * @property {number} [confidence]
48
+ * @property {Evaluator} evaluator
49
+ * @property {DecisionSource} source
50
+ * @property {string} reason
51
+ * @property {ClassifierError} [classifier_error]
52
+ * @property {number} [classifier_status]
53
+ */
54
+ /**
55
+ * @typedef {object} RoutingDecision
56
+ * @property {string} model
57
+ * @property {DecisionSource} source
58
+ * @property {string} reason
59
+ * @property {number} latency_ms
60
+ * @property {number} evaluation_latency_ms
61
+ * @property {Tier} [tier]
62
+ * @property {Tier} [classified_tier]
63
+ * @property {number} [confidence]
64
+ * @property {Evaluator} [evaluator]
65
+ * @property {ClassifierError} [classifier_error]
66
+ * @property {number} [classifier_status]
67
+ * @property {string} [compatibility_reason]
68
+ * @property {ContinuityState} [continuity_state]
69
+ * @property {'within_budget'|'over_budget'|'count_unavailable'} [context_check]
70
+ * @property {number} [counted_input_tokens]
71
+ */
72
+ /**
73
+ * @typedef {object} TelemetryFields
74
+ * @property {number} [schema_version]
75
+ * @property {string} [timestamp]
76
+ * @property {string} [request_id]
77
+ * @property {string} [session_id]
78
+ * @property {string} [agent_id]
79
+ * @property {string} [prompt_id]
80
+ * @property {string} [request_class]
81
+ * @property {string} [requested_model]
82
+ * @property {string} [selected_model]
83
+ * @property {string} [confirmed_model]
84
+ * @property {string} [model]
85
+ * @property {string} [baseline_model]
86
+ * @property {string} [pricing_version]
87
+ * @property {string} [reason]
88
+ * @property {string} [compatibility_reason]
89
+ * @property {ContinuityState} [continuity_state]
90
+ * @property {DecisionSource} [source]
91
+ * @property {Evaluator} [evaluator]
92
+ * @property {Tier} [tier]
93
+ * @property {Tier} [classified_tier]
94
+ * @property {number} [latency_ms]
95
+ * @property {number} [evaluation_latency_ms]
96
+ * @property {number} [routing_latency_ms]
97
+ * @property {number} [decision_latency_ms]
98
+ * @property {number} [first_response_ms]
99
+ * @property {number} [upstream_latency_ms]
100
+ * @property {number} [total_latency_ms]
101
+ * @property {ClassifierError} [classifier_error]
102
+ * @property {number} [classifier_status]
103
+ * @property {string} [error_type]
104
+ * @property {number} [http_status]
105
+ * @property {number|string} [status]
106
+ * @property {'within_budget'|'over_budget'|'count_unavailable'} [context_check]
107
+ * @property {number} [counted_input_tokens]
108
+ * @property {string[]} [model_transitions]
109
+ * @property {boolean} [model_transitions_truncated]
110
+ * @property {object} [usage]
111
+ * @property {object} [pricing_context]
112
+ * @property {boolean} [usage_complete]
113
+ * @property {boolean} [pricing_eligible]
114
+ * @property {boolean} [completion_confirmed]
115
+ * @property {string} [unpriced_reason]
116
+ * @property {object} [savings]
117
+ * @property {object} [savings_coverage]
118
+ * @property {string} [prompt_excerpt]
119
+ * @property {boolean} [prompt_truncated]
120
+ */
121
+ /** @typedef {TelemetryFields & {event:LifecycleName,request_id:string}} LifecycleEvent */
122
+ /** @typedef {TelemetryFields & {event:'decision'|'outcome',request_id:string}} SessionRecord */
123
+ export {};
@@ -0,0 +1,114 @@
1
+ // Pure acceptance rules shared by the opt-in evaluation harnesses. Thresholds
2
+ // are validated before any provider calls; passing transport is not evidence
3
+ // of evaluator availability, routing correctness, or completed task quality.
4
+ const TIERS = ['haiku', 'sonnet', 'opus'];
5
+ const PROFILES = ['compatible', 'native', 'auto'];
6
+ const gate = (passed, details = {}) => ({ passed, ...details });
7
+ const allPassed = gates => Object.values(gates).every(value => value.passed !== false);
8
+
9
+ export function modelTier(model) {
10
+ return typeof model === 'string' ? /^claude-(haiku|sonnet|opus)(?:-|$)/.exec(model)?.[1] : undefined;
11
+ }
12
+
13
+ export function profileTier(tier, profile) {
14
+ return profile === 'auto' && tier === 'haiku' ? 'sonnet' : tier;
15
+ }
16
+
17
+ export function parseQualityThreshold(value) {
18
+ if (typeof value !== 'string' || !/^(?:\d+(?:\.\d+)?|\.\d+)$/.test(value)) throw new Error('Quality thresholds must be decimal numbers between 0 and 1');
19
+ const number = Number(value);
20
+ if (!Number.isFinite(number) || number < 0 || number > 1) throw new Error('Quality thresholds must be between 0 and 1');
21
+ return number;
22
+ }
23
+
24
+ export function createEvaluationPolicy({ profile = 'compatible', requiredTiers,
25
+ minAgreement = 1, maxUnderRouteRate = 0 } = {}) {
26
+ if (!PROFILES.includes(profile)) throw new Error('Evaluation profile must be compatible, native, or auto');
27
+ for (const [name, value] of [['minAgreement', minAgreement], ['maxUnderRouteRate', maxUnderRouteRate]]) {
28
+ if (typeof value !== 'number' || !Number.isFinite(value) || value < 0 || value > 1) throw new Error(`${name} must be between 0 and 1`);
29
+ }
30
+ const tiers = requiredTiers ?? (profile === 'native' ? [] : profile === 'auto' ? ['sonnet', 'opus'] : TIERS);
31
+ if (!Array.isArray(tiers) || tiers.some(tier => !TIERS.includes(tier)) || new Set(tiers).size !== tiers.length
32
+ || (profile === 'auto' && tiers.includes('haiku'))) throw new Error('Invalid required tier coverage for evaluation profile');
33
+ return Object.freeze({ profile, minAgreement, maxUnderRouteRate, requiredTiers: Object.freeze([...tiers]) });
34
+ }
35
+
36
+ function evaluatorGate(rows, evaluator, expectOutage) {
37
+ if (!['jev', 'ollama'].includes(evaluator)) throw new Error('Unknown evaluation backend');
38
+ const active = rows.filter(row => row.source !== 'passthrough');
39
+ const fresh = active.filter(row => row.source === evaluator).length;
40
+ const fallbacks = active.filter(row => row.source === 'fallback').length;
41
+ const passed = active.length > 0 && (expectOutage ? fallbacks === active.length
42
+ : fresh > 0 && active.every(row => row.source === evaluator || (row.source === 'cache' && row.evaluator === evaluator)));
43
+ return gate(passed, { expected: expectOutage ? 'fallback' : evaluator, evaluated_requests: active.length,
44
+ successful_evaluations: fresh, fallbacks });
45
+ }
46
+
47
+ export function tierCoverage(tiers, policy) {
48
+ const observed = [...new Set(tiers.filter(tier => TIERS.includes(tier)))].sort();
49
+ const missing = policy.requiredTiers.filter(tier => !observed.includes(tier));
50
+ return gate(missing.length === 0, { required: [...policy.requiredTiers], observed, missing });
51
+ }
52
+
53
+ export function evaluateRoutingReport(rows, { evaluator, policy = createEvaluationPolicy(), classifierOnly = false } = {}) {
54
+ policy = createEvaluationPolicy(policy);
55
+ if (!Array.isArray(rows) || rows.some(row => !row || !TIERS.includes(row.expected))) throw new Error('Invalid evaluation rows');
56
+ const valid = rows.filter(row => TIERS.includes(row.classified_tier)
57
+ && (row.source === evaluator || (row.source === 'cache' && row.evaluator === evaluator)));
58
+ const agreement = rows.length ? valid.filter(row => row.classified_tier === row.expected).length / rows.length : 0;
59
+ const underRoutes = valid.filter(row => TIERS.indexOf(row.classified_tier) < TIERS.indexOf(row.expected)).length;
60
+ const underRouteRate = rows.length ? underRoutes / rows.length : 0;
61
+ const eligibleTiers = valid.map(row => classifierOnly ? row.classified_tier : row.selected_tier ?? modelTier(row.selected_model));
62
+ const gates = {
63
+ transport: gate(null, { reason: 'Claude transport not exercised' }),
64
+ evaluator: evaluatorGate(rows, evaluator, false),
65
+ rubric: gate(rows.length > 0 && valid.length === rows.length && agreement >= policy.minAgreement
66
+ && underRouteRate <= policy.maxUnderRouteRate, { agreement, under_routes: underRoutes, under_route_rate: underRouteRate,
67
+ min_agreement: policy.minAgreement, max_under_route_rate: policy.maxUnderRouteRate }),
68
+ policy: gate(classifierOnly ? null : rows.length > 0 && rows.every(row => {
69
+ const selected = row.selected_tier ?? modelTier(row.selected_model);
70
+ return TIERS.includes(selected) && (policy.profile !== 'auto' || selected !== 'haiku')
71
+ && (row.expected_selected_tier === undefined || selected === row.expected_selected_tier)
72
+ && (row.expected_reason === undefined || row.reason === row.expected_reason);
73
+ }), classifierOnly ? { reason: 'Classifier-only benchmark; routing policy not exercised' } : {}),
74
+ coverage: tierCoverage(eligibleTiers, policy),
75
+ task: gate(null, { reason: 'No Claude task completion measured' }),
76
+ };
77
+ return { schema_version: 1, policy, gates, passed: allPassed(gates), requests: rows.length };
78
+ }
79
+
80
+ const TRANSPORT_CHECKS = new Set(['claude_success', 'requests_reached_router', 'upstream_model_evidence', 'no_upstream_api_errors', 'no_proxy_errors']);
81
+ const TASK_CHECKS = new Set(['expected_result', 'original_tests_preserved', 'independent_tests_pass', 'read_edit_bash_exercised']);
82
+ function checkGate(checks, names) {
83
+ const selected = Object.fromEntries(Object.entries(checks).filter(([name]) => names.has(name)));
84
+ return gate(Object.keys(selected).length ? Object.values(selected).every(value => value === true) : null, { checks: selected });
85
+ }
86
+
87
+ export function evaluateLiveCase({ checks, routes, evaluator, expectOutage = false, expectedClassifiedTier } = {}) {
88
+ const policyChecks = new Set(Object.keys(checks).filter(name => !TRANSPORT_CHECKS.has(name) && !TASK_CHECKS.has(name)));
89
+ const main = routes.filter(row => ['main', 'unspecified', undefined, ''].includes(row.request_class));
90
+ const labels = main.map(row => expectedClassifiedTier ?? row.expected_classified_tier);
91
+ const rubric = expectOutage ? gate(null, { reason: 'Explicit evaluator outage' })
92
+ : gate(main.length > 0 && labels.every(tier => TIERS.includes(tier))
93
+ && main.every((row, index) => row.classified_tier === labels[index]),
94
+ { expected: expectedClassifiedTier ?? labels, ...(labels.some(tier => !TIERS.includes(tier)) ? { reason: 'Missing declared live classifier label' } : {}) });
95
+ const gates = {
96
+ transport: checkGate(checks, TRANSPORT_CHECKS),
97
+ evaluator: evaluatorGate(routes, evaluator, expectOutage),
98
+ rubric,
99
+ policy: checkGate(checks, policyChecks),
100
+ task: checkGate(checks, TASK_CHECKS),
101
+ };
102
+ // Empty or incomplete harness evidence can never be a successful live case.
103
+ if (!routes.length || gates.transport.passed === null || gates.task.passed === null) gates.transport.passed = false;
104
+ return { gates, passed: allPassed(gates) };
105
+ }
106
+
107
+ export function evaluateLiveReport(cases, { policy = createEvaluationPolicy(), expectOutage = false } = {}) {
108
+ policy = createEvaluationPolicy(policy);
109
+ const tiers = cases.flatMap(item => (item.routes ?? []).filter(row =>
110
+ ['main', 'unspecified', undefined, ''].includes(row.request_class)
111
+ && ['jev', 'ollama', 'cache'].includes(row.source)).map(row => modelTier(row.model)));
112
+ const coverage = expectOutage ? gate(null, { reason: 'Explicit evaluator outage' }) : tierCoverage(tiers, policy);
113
+ return { policy, gates: { coverage }, passed: cases.length > 0 && cases.every(item => item.passed === true) && coverage.passed !== false };
114
+ }
@@ -0,0 +1,58 @@
1
+ import { spawnSync } from 'node:child_process';
2
+
3
+ // macOS Keychain access through the system `security` tool. Secret values are
4
+ // passed on stdin to its interactive mode, never as process arguments, and are
5
+ // read back from stdout. Tool output can describe items, so failures report
6
+ // only a generic category.
7
+ const SECURITY = '/usr/bin/security';
8
+ const SERVICE = 'claude-autorouter';
9
+ const NOT_FOUND = 44;
10
+ const PRINTABLE = /^[\x20-\x7e]+$/;
11
+
12
+ function keychainError(message) {
13
+ const error = new Error(message);
14
+ error.code = 'AUTOROUTER_CONFIG_ERROR';
15
+ return error;
16
+ }
17
+
18
+ // Quote for the `security -i` command parser, which accepts backslash escapes
19
+ // inside double quotes. Values are validated as printable single-line ASCII.
20
+ const quote = value => `"${value.replace(/[\\"]/g, '\\$&')}"`;
21
+
22
+ export function createKeychain({ run = spawnSync, platform = process.platform } = {}) {
23
+ const available = platform === 'darwin';
24
+ const call = (args, input) => {
25
+ if (!available) throw keychainError('The macOS Keychain secret store is available only on macOS.');
26
+ const result = run(SECURITY, args, {
27
+ input, encoding: 'utf8', timeout: 15000, maxBuffer: 64 * 1024, stdio: ['pipe', 'pipe', 'pipe'],
28
+ });
29
+ if (result.error) throw keychainError('Could not run the macOS Keychain tool.');
30
+ return result;
31
+ };
32
+ const read = account => {
33
+ const result = call(['find-generic-password', '-s', SERVICE, '-a', account, '-w']);
34
+ if (result.status === NOT_FOUND) return undefined;
35
+ if (result.status !== 0) {
36
+ throw keychainError('Could not read an AutoRouter secret from the macOS Keychain. Unlock the login keychain, or set the key in the environment.');
37
+ }
38
+ return String(result.stdout).replace(/\n$/, '');
39
+ };
40
+ return {
41
+ available,
42
+ read,
43
+ write(account, value, label) {
44
+ if (typeof value !== 'string' || !PRINTABLE.test(value)) {
45
+ throw keychainError('Keychain secrets must be printable single-line ASCII.');
46
+ }
47
+ call(['-i'], `add-generic-password -U -s ${quote(SERVICE)} -a ${quote(account)} -l ${quote(label)} -w ${quote(value)}\n`);
48
+ // Interactive mode exits successfully even when a command fails.
49
+ if (read(account) !== value) throw keychainError('Could not save an AutoRouter secret to the macOS Keychain.');
50
+ },
51
+ remove(account) {
52
+ const result = call(['delete-generic-password', '-s', SERVICE, '-a', account]);
53
+ if (result.status !== 0 && result.status !== NOT_FOUND) {
54
+ throw keychainError('Could not remove an AutoRouter secret from the macOS Keychain.');
55
+ }
56
+ },
57
+ };
58
+ }
@@ -0,0 +1,191 @@
1
+ import { createHash } from 'node:crypto';
2
+ import { Router } from './router.mjs';
3
+ import { buildOllamaState, OLLAMA_QUESTIONS } from './ollama-evaluator.mjs';
4
+ import { inspectOllama } from './ollama-setup.mjs';
5
+ import { validateOllamaEndpoint, validateOllamaModel } from './ollama-models.mjs';
6
+ import { createEvaluationPolicy, evaluateRoutingReport } from './evaluation-report.mjs';
7
+
8
+ // Packaged synthetic fixtures: no repository, transcript, or account data is
9
+ // read. Labels test this rubric on six examples, not general task quality.
10
+ export const LOCAL_DIAGNOSTIC_VERSION = 1;
11
+ export const LOCAL_DIAGNOSTIC_CASES = Object.freeze([
12
+ { id: 'literal', expected: 'haiku', prompt: 'Print the literal word READY exactly. Do not add any explanation.' },
13
+ { id: 'array-length', expected: 'haiku', prompt: 'What does the JavaScript expression `[].length` evaluate to? Reply with only the integer.' },
14
+ { id: 'bounded-feature', expected: 'sonnet', prompt: 'Add pagination to this REST endpoint using `page` and `pageSize`. Validate the parameters, preserve existing filtering, and add tests for empty results and out-of-range pages.' },
15
+ { id: 'distributed-fencing', expected: 'opus', prompt: "Review this distributed locking design: worker A's lease expires while paused. Worker B acquires a newer fencing token and writes successfully. A resumes and writes using its old token. The database checks only whether a token was ever issued. Explain the failure sequence and design the minimum atomic database check that prevents stale writes, including duplicate retries." },
16
+ { id: 'new-mechanical-task', expected: 'haiku', history: [
17
+ { role: 'user', content: 'Prove the safety of a distributed ledger during failover and concurrent retries.' },
18
+ { role: 'assistant', content: 'The proof and regression tests are complete.' },
19
+ ], prompt: "Separate task: replace the exact text 'recieve' with 'receive' in a label. Make no other changes." },
20
+ { id: 'new-difficult-task', expected: 'opus', history: [
21
+ { role: 'user', content: 'What is the value of [].length?' },
22
+ { role: 'assistant', content: '0' },
23
+ ], prompt: 'New task: diagnose nondeterministic deadlocks between several processes after a rolling deployment. Reconcile conflicting traces, identify the broken ordering invariant, and prove that the repair cannot introduce message loss.' },
24
+ ].map(item => Object.freeze({ ...item, ...(item.history ? { history: Object.freeze(item.history.map(Object.freeze)) } : {}) })));
25
+
26
+ const STARTUP_TIMEOUT_MS = 60000;
27
+ const METADATA_TIMEOUT_MS = 5000;
28
+ const MAX_METADATA_BYTES = 1024 * 1024;
29
+ const digest = value => createHash('sha256').update(JSON.stringify(value)).digest('hex');
30
+ const rounded = value => Math.round(value * 100) / 100;
31
+ const identity = model => {
32
+ const name = model.replace(/^registry\.ollama\.ai\//, '').replace(/^library\//, '');
33
+ return name.slice(name.lastIndexOf('/') + 1).includes(':') ? name : `${name}:latest`;
34
+ };
35
+
36
+ const MESSAGES = Object.freeze({
37
+ evaluator_required: 'Local evaluation requires AUTOROUTER_EVALUATOR=ollama; the diagnostic does not change your configuration.',
38
+ invalid_configuration: 'Check the local Ollama endpoint, model, and runtime deadline configuration.',
39
+ positive_keep_alive_required: 'The local diagnostic requires a positive keep-alive, such as AUTOROUTER_OLLAMA_KEEP_ALIVE=5m. Normal routing still supports 0.',
40
+ model_missing: 'The configured local model is not installed. Install it explicitly before rerunning this diagnostic.',
41
+ unrelated_models_resident: 'Other Ollama models are resident. Retry when only the selected model, or no model, is loaded; this diagnostic leaves them running.',
42
+ residency_unavailable: 'Could not safely inspect local Ollama model residency. No further evaluation was attempted.',
43
+ OLLAMA_VERSION: 'Local evaluation requires Ollama 0.35 or newer with /v1/systemone.',
44
+ OLLAMA_CLOUD: 'The selected Ollama model uses a remote service. Select an installed local model.',
45
+ OLLAMA_TIMEOUT: 'Local Ollama metadata inspection timed out.',
46
+ OLLAMA_HTTP: 'Local Ollama rejected metadata inspection. Check its version and selected model.',
47
+ OLLAMA_RESPONSE: 'Local Ollama returned invalid or oversized metadata.',
48
+ OLLAMA_UNAVAILABLE: 'Could not reach local Ollama. Start the existing service and retry.',
49
+ startup_failed: 'The initial synthetic evaluation failed; measured cases were not run.',
50
+ });
51
+ const failure = code => Object.assign(new Error(MESSAGES[code]), { code });
52
+
53
+ function requestBody(item) {
54
+ return { model: 'claude-haiku-4-5-20251001', max_tokens: 32000, tools: [],
55
+ system: [{ type: 'text', text: 'Synthetic coding assistant guidance: inspect relevant code and verify changes. '.repeat(110) }],
56
+ messages: [...structuredClone(item.history ?? []), { role: 'user', content: [
57
+ { type: 'text', text: '<system-reminder>SYNTHETIC_REMINDER_ONLY: unrelated environment metadata.</system-reminder>' },
58
+ { type: 'text', text: item.prompt },
59
+ ] }],
60
+ };
61
+ }
62
+
63
+ async function residency(config, fetchImpl, signal) {
64
+ const timeout = AbortSignal.timeout(METADATA_TIMEOUT_MS);
65
+ const combined = signal ? AbortSignal.any([signal, timeout]) : timeout;
66
+ const response = await fetchImpl(`${config.ollamaEndpoint}/api/ps`, { redirect: 'error', signal: combined });
67
+ if (!response.ok || response.redirected || !response.body) {
68
+ await response.body?.cancel();
69
+ throw failure('residency_unavailable');
70
+ }
71
+ const reader = response.body.getReader();
72
+ const chunks = [];
73
+ let size = 0;
74
+ const abort = () => { void reader.cancel(combined.reason).catch(() => {}); };
75
+ combined.addEventListener('abort', abort, { once: true });
76
+ try {
77
+ for (;;) {
78
+ combined.throwIfAborted();
79
+ const { value, done } = await reader.read();
80
+ if (done) break;
81
+ size += value.byteLength;
82
+ if (size > MAX_METADATA_BYTES) throw failure('residency_unavailable');
83
+ chunks.push(value);
84
+ }
85
+ combined.throwIfAborted();
86
+ const payload = JSON.parse(Buffer.concat(chunks).toString('utf8'));
87
+ if (!Array.isArray(payload?.models) || payload.models.some(item => typeof (item?.name ?? item?.model) !== 'string')) {
88
+ throw failure('residency_unavailable');
89
+ }
90
+ const models = payload.models.map(item => identity(item.name ?? item.model));
91
+ if (models.some(model => model !== identity(config.ollamaModel))) throw failure('unrelated_models_resident');
92
+ return models.length ? 'resident' : 'not_resident';
93
+ } catch (error) {
94
+ signal?.throwIfAborted();
95
+ throw error?.code === 'unrelated_models_resident' ? error : failure('residency_unavailable');
96
+ } finally {
97
+ combined.removeEventListener('abort', abort);
98
+ await reader.cancel().catch(() => {});
99
+ reader.releaseLock();
100
+ }
101
+ }
102
+
103
+ function finish(report) {
104
+ const evaluation = evaluateRoutingReport(report.rows, { evaluator: 'ollama', classifierOnly: true,
105
+ policy: createEvaluationPolicy({ profile: 'compatible' }) });
106
+ report.gates = { ...evaluation.gates,
107
+ preflight: { passed: report.preflight_passed },
108
+ startup: { passed: report.startup?.source === 'ollama' },
109
+ cases: { passed: report.rows.length === LOCAL_DIAGNOSTIC_CASES.length, expected: LOCAL_DIAGNOSTIC_CASES.length, completed: report.rows.length },
110
+ };
111
+ report.passed = !report.error && Object.values(report.gates).every(gate => gate.passed !== false);
112
+ return report;
113
+ }
114
+
115
+ // No warmup helper with pull/unload controls is reachable from this diagnostic.
116
+ // Caller cancellation is preserved, including when runtime timeout is zero.
117
+ export async function runLocalDiagnostic(config, { fetchImpl = fetch, signal, onProgress = () => {} } = {}) {
118
+ const report = { schema_version: 1, type: 'local_evaluator_diagnostic', fixture_version: LOCAL_DIAGNOSTIC_VERSION,
119
+ fixture_sha256: digest(LOCAL_DIAGNOSTIC_CASES), questions_sha256: digest(OLLAMA_QUESTIONS),
120
+ evaluator: 'ollama', startup_timeout_ms: STARTUP_TIMEOUT_MS, preflight_passed: false,
121
+ residency_before: 'unknown', startup: null, rows: [], paid_provider_calls: 0, downloads: 0,
122
+ models_unloaded: 0, configuration_changed: false };
123
+ const progress = event => { try { Promise.resolve(onProgress(event)).catch(() => {}); } catch {} };
124
+ try {
125
+ signal?.throwIfAborted();
126
+ if (config?.evaluator !== 'ollama') throw failure('evaluator_required');
127
+ try {
128
+ validateOllamaEndpoint(config.ollamaEndpoint);
129
+ validateOllamaModel(config.ollamaModel);
130
+ if (!Number.isSafeInteger(config.ollamaTimeoutMs) || config.ollamaTimeoutMs < 0 || config.ollamaTimeoutMs > 30000) throw new Error();
131
+ } catch { throw failure('invalid_configuration'); }
132
+ if (typeof config.ollamaKeepAlive !== 'string' || !/^[1-9]\d{0,3}[smh]$/.test(config.ollamaKeepAlive)) throw failure('positive_keep_alive_required');
133
+ report.evaluator_model = config.ollamaModel;
134
+ report.runtime_timeout_ms = config.ollamaTimeoutMs;
135
+ report.keep_alive = config.ollamaKeepAlive;
136
+ progress({ event: 'preflight' });
137
+ const inspection = await inspectOllama(config, { fetchImpl, signal });
138
+ if (!inspection.installed) throw failure('model_missing');
139
+ report.residency_before = await residency(config, fetchImpl, signal);
140
+ report.preflight_passed = true;
141
+ progress({ event: 'startup', residency_before: report.residency_before, timeout_ms: STARTUP_TIMEOUT_MS });
142
+ const initialStart = performance.now();
143
+ const startup = await new Router({ ...config, ollamaTimeoutMs: STARTUP_TIMEOUT_MS }, { fetchImpl })
144
+ .classify(requestBody({ prompt: 'Return the literal word ready.' }), signal);
145
+ report.startup = { latency_ms: rounded(performance.now() - initialStart), source: startup.source,
146
+ classified_tier: startup.classified_tier, classifier_error: startup.classifier_error, classifier_status: startup.classifier_status,
147
+ residency_before: report.residency_before, timeout_ms: STARTUP_TIMEOUT_MS };
148
+ progress({ event: 'startup_complete', ...report.startup });
149
+ if (startup.source !== 'ollama') throw failure('startup_failed');
150
+ for (const item of LOCAL_DIAGNOSTIC_CASES) {
151
+ signal?.throwIfAborted();
152
+ const resident = await residency(config, fetchImpl, signal);
153
+ progress({ event: 'case_start', case: item.id, residency_before: resident });
154
+ const body = requestBody(item);
155
+ const state = buildOllamaState(body, config.ollamaStateChars);
156
+ // A new classifier prevents cache hits from appearing as local inference.
157
+ const start = performance.now();
158
+ const decision = await new Router(config, { fetchImpl }).classify(body, signal);
159
+ const row = { case: item.id, expected: item.expected, classified_tier: decision.classified_tier,
160
+ tier: decision.tier, source: decision.source, evaluator: decision.evaluator, reason: decision.reason,
161
+ classifier_error: decision.classifier_error, classifier_status: decision.classifier_status,
162
+ latency_ms: rounded(performance.now() - start), residency_before: resident,
163
+ state_bytes: Buffer.byteLength(JSON.stringify(state)), current_task_matches: state.current_task === item.prompt };
164
+ report.rows.push(row);
165
+ progress({ event: 'case_complete', ...row });
166
+ }
167
+ } catch (error) {
168
+ signal?.throwIfAborted();
169
+ const code = Object.hasOwn(MESSAGES, error?.code) ? error.code : 'OLLAMA_UNAVAILABLE';
170
+ report.error = { code, message: MESSAGES[code] };
171
+ }
172
+ const result = finish(report);
173
+ // Correct labels alone cannot conceal broken task extraction.
174
+ if (report.rows.some(row => !row.current_task_matches)) {
175
+ result.gates.cases.passed = false;
176
+ result.passed = false;
177
+ }
178
+ return result;
179
+ }
180
+
181
+ export function formatLocalDiagnostic(report) {
182
+ const cause = row => row.classifier_error ? ` (${row.classifier_error}${row.classifier_status ? ` ${row.classifier_status}` : ''})` : '';
183
+ const lines = [`Local evaluator diagnostic: ${report.evaluator_model ?? 'not configured'}`];
184
+ if (report.runtime_timeout_ms !== undefined) lines.push(`Runtime deadline: ${report.runtime_timeout_ms === 0 ? 'disabled' : `${report.runtime_timeout_ms} ms`}; initial preparation bound: ${report.startup_timeout_ms} ms.`);
185
+ if (report.startup) lines.push(`Initial synthetic call: ${report.startup.latency_ms} ms; model ${report.startup.residency_before === 'resident' ? 'resident' : 'not resident'} before call; ${report.startup.source}${cause(report.startup)}.`);
186
+ for (const row of report.rows) lines.push(`${row.case}: expected ${row.expected}, got ${row.classified_tier ?? 'no verdict'}; ${row.source}${cause(row)}; ${row.latency_ms} ms; ${row.residency_before} before call.`);
187
+ if (report.error) lines.push(report.error.message);
188
+ lines.push(`${report.passed ? 'PASS' : 'FAIL'}: ${report.rows.filter(row => row.source === 'ollama' && row.classified_tier === row.expected).length}/${LOCAL_DIAGNOSTIC_CASES.length} synthetic cases; missing tiers: ${report.gates.coverage.missing.join(', ') || 'none'}.`);
189
+ lines.push('Residency is observed, not a guarantee of cold or warm caches. This checks six classifier labels, not Claude task completion or general accuracy.');
190
+ return lines;
191
+ }