claude-autorouter 0.3.7 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. package/.env.example +6 -3
  2. package/CODE_OF_CONDUCT.md +9 -0
  3. package/CONTRIBUTING.md +57 -0
  4. package/README.md +47 -70
  5. package/SECURITY.md +23 -0
  6. package/SUPPORT.md +18 -0
  7. package/bin/autorouter.mjs +40 -57
  8. package/docs/development.md +48 -2
  9. package/docs/hardware-benchmark.md +29 -0
  10. package/docs/hardware-comparison.md +55 -0
  11. package/docs/hardware-results-16gb.json +4002 -0
  12. package/docs/hardware-results-16gb.md +26 -0
  13. package/docs/hardware-results-64gb.json +4020 -0
  14. package/docs/reference.md +92 -37
  15. package/docs/releasing.md +79 -37
  16. package/docs/router-performance.json +1697 -0
  17. package/docs/router-performance.md +50 -0
  18. package/docs/status-performance.json +363 -0
  19. package/docs/status-performance.md +44 -0
  20. package/docs/subscription-integration.md +27 -0
  21. package/package.json +66 -10
  22. package/src/auto-routing.mjs +184 -24
  23. package/src/bounded-json.mjs +57 -0
  24. package/src/cli-help.mjs +90 -0
  25. package/src/config-command.mjs +158 -0
  26. package/src/config.mjs +53 -28
  27. package/src/contracts.mjs +123 -0
  28. package/src/evaluation-report.mjs +114 -0
  29. package/src/keychain.mjs +58 -0
  30. package/src/local-diagnostic.mjs +191 -0
  31. package/src/model-catalog.mjs +96 -0
  32. package/src/model-request.mjs +6 -7
  33. package/src/ollama-evaluator.mjs +9 -27
  34. package/src/onboarding.mjs +130 -26
  35. package/src/prompt-state.mjs +22 -7
  36. package/src/redaction.mjs +97 -0
  37. package/src/request-validation.mjs +54 -0
  38. package/src/response-observer.mjs +126 -18
  39. package/src/router.mjs +151 -61
  40. package/src/savings.mjs +74 -16
  41. package/src/server.mjs +79 -12
  42. package/src/session-history.mjs +262 -0
  43. package/src/session-log.mjs +9 -58
  44. package/src/status-state.mjs +110 -62
  45. package/src/statusline.mjs +57 -27
  46. package/src/telemetry-event.mjs +200 -0
  47. package/src/token-counter.mjs +3 -1
  48. package/src/turn-state.mjs +132 -0
  49. package/src/user-config.mjs +81 -10
@@ -1,54 +1,214 @@
1
+ import { modelCapabilities } from './model-catalog.mjs';
2
+ import { prepareRequest } from './model-request.mjs';
3
+
1
4
  // Shared execution capabilities, not a replacement for Claude's permission
2
5
  // classifier. Keep the complete safeguards contract and signed history on wire.
3
- const MODELS = new Set(['claude-sonnet-5', 'claude-sonnet-5-5', 'claude-opus-5', 'claude-opus-5-5']);
4
6
  const object = value => value !== null && typeof value === 'object' && !Array.isArray(value);
5
- const SHARED_TOOLS = new Set(['custom', 'tool_search_tool_regex_20251119', 'tool_search_tool_bm25_20251119',
6
- 'bash_20250124', 'text_editor_20250728']);
7
7
  const CONTEXT_EDITS = new Set(['clear_thinking_20251015', 'clear_tool_uses_20250919']);
8
- const sharedTool = tool => object(tool) && (tool.type === undefined || SHARED_TOOLS.has(tool.type));
8
+ // Reviewed 2026-10-05 against the Messages API/tool reference and context-edit
9
+ // contract. These are semantic envelopes; input schemas/examples and document
10
+ // data remain opaque and are never inspected or rewritten.
11
+ // https://platform.claude.com/docs/en/api/beta/messages/create
12
+ // https://platform.claude.com/docs/en/build-with-claude/mid-conversation-system-messages
13
+ // https://github.com/anthropics/anthropic-sdk-typescript/blob/main/src/resources/beta/messages/messages.ts
14
+ const CUSTOM_TOOL_FIELDS = new Set(['type', 'name', 'description', 'input_schema', 'cache_control',
15
+ 'defer_loading', 'strict', 'input_examples', 'allowed_callers', 'eager_input_streaming']);
16
+ const SEARCH_TOOL_FIELDS = new Set(['type', 'name', 'allowed_callers', 'cache_control', 'defer_loading', 'strict']);
17
+ const BASH_TOOL_FIELDS = new Set([...SEARCH_TOOL_FIELDS, 'input_examples']);
18
+ const EDITOR_TOOL_FIELDS = new Set([...BASH_TOOL_FIELDS, 'max_characters']);
19
+ const TOOL_FIELDS = new Map([
20
+ ['custom', CUSTOM_TOOL_FIELDS], ['bash_20250124', BASH_TOOL_FIELDS],
21
+ ['text_editor_20250728', EDITOR_TOOL_FIELDS],
22
+ ['tool_search_tool_regex_20251119', SEARCH_TOOL_FIELDS], ['tool_search_tool_bm25_20251119', SEARCH_TOOL_FIELDS],
23
+ ]);
24
+ const TOOL_CHOICE_FIELDS = new Map([
25
+ ['auto', new Set(['type', 'disable_parallel_tool_use'])], ['any', new Set(['type', 'disable_parallel_tool_use'])],
26
+ ['tool', new Set(['type', 'name', 'disable_parallel_tool_use'])], ['none', new Set(['type'])],
27
+ ]);
28
+ const MESSAGE_FIELDS = new Set(['role', 'content', 'output_config', 'clear_at']);
29
+ const THINKING_EDIT_FIELDS = new Set(['type', 'keep']);
30
+ const TOOL_EDIT_FIELDS = new Set(['type', 'clear_at_least', 'clear_tool_inputs', 'exclude_tools', 'keep', 'trigger']);
31
+ const EDIT_VALUE_FIELDS = new Set(['type', 'value']);
32
+ const knownFields = (value, fields) => Object.keys(value).every(key => fields.has(key));
33
+ const optionalEnvelope = (value, fields) => value == null || (object(value) && knownFields(value, fields));
34
+ const sharedTool = tool => object(tool) && TOOL_FIELDS.has(tool.type ?? 'custom')
35
+ && knownFields(tool, TOOL_FIELDS.get(tool.type ?? 'custom')) && optionalEnvelope(tool.cache_control, CACHE_FIELDS);
36
+ const sharedEdit = edit => object(edit) && CONTEXT_EDITS.has(edit.type)
37
+ && knownFields(edit, edit.type === 'clear_thinking_20251015' ? THINKING_EDIT_FIELDS : TOOL_EDIT_FIELDS)
38
+ && ['keep', 'trigger', 'clear_at_least'].every(key => !object(edit[key]) || knownFields(edit[key], EDIT_VALUE_FIELDS));
39
+ const REQUEST_FIELDS = new Set(['model', 'messages', 'system', 'max_tokens', 'metadata', 'stream',
40
+ 'stop_sequences', 'temperature', 'top_p', 'top_k', 'tools', 'tool_choice', 'thinking', 'output_config',
41
+ 'context_management', 'speed', 'container', 'mcp_servers', 'compaction', 'safeguards', 'service_tier',
42
+ 'inference_geo', 'cache_control']);
43
+ const CONTENT_TYPES = new Set(['text', 'image', 'document', 'tool_use', 'tool_result', 'thinking',
44
+ 'redacted_thinking', 'tool_reference', 'tool_search_tool_result', 'tool_search_tool_search_result',
45
+ 'tool_addition', 'tool_removal']);
46
+ // The beta API distinguishes semantic block envelopes from opaque payloads.
47
+ // See its BetaContentBlockParam union and the matching official SDK types:
48
+ // https://github.com/anthropics/anthropic-sdk-python/tree/main/src/anthropic/types/beta
49
+ const CONTENT_FIELDS = Object.fromEntries(Object.entries({
50
+ text: ['type', 'text', 'cache_control', 'citations'],
51
+ image: ['type', 'source', 'cache_control', 'transformations'],
52
+ document: ['type', 'source', 'cache_control', 'citations', 'context', 'title'],
53
+ tool_use: ['type', 'id', 'input', 'name', 'cache_control', 'caller', 'toolset_name'],
54
+ tool_result: ['type', 'tool_use_id', 'content', 'is_error', 'cache_control', 'toolset_name'],
55
+ thinking: ['type', 'signature', 'thinking'],
56
+ redacted_thinking: ['type', 'data'],
57
+ tool_reference: ['type', 'tool_name', 'cache_control'],
58
+ tool_search_tool_result: ['type', 'content', 'tool_use_id', 'cache_control'],
59
+ tool_search_tool_search_result: ['type', 'tool_references'],
60
+ tool_search_tool_result_error: ['type', 'error_code', 'error_message'],
61
+ tool_addition: ['type', 'tool', 'cache_control'],
62
+ tool_removal: ['type', 'tool', 'cache_control'],
63
+ }).map(([type, fields]) => [type, new Set(fields)]));
64
+ const CACHE_FIELDS = new Set(['type', 'ttl']);
65
+ const TRANSFORMATION_FIELDS = new Set(['oversized_image']);
66
+ const TOOL_REFERENCE_FIELDS = new Set(['type', 'name']);
67
+ const TOOL_DEFINITION_FIELDS = new Set(['type', 'definition']);
68
+ const OUTPUT_FIELDS = new Set(['effort', 'format', 'task_budget']);
69
+ const OUTPUT_FORMAT_FIELDS = new Set(['type', 'schema']);
70
+ const TASK_BUDGET_FIELDS = new Set(['type', 'total', 'remaining']);
71
+ const THINKING_BINDING_FIELDS = new Set(['prefix_mismatch_behavior']);
72
+ const THINKING_FIELDS = new Map([
73
+ ['enabled', new Set(['type', 'budget_tokens', 'display', 'block_binding'])],
74
+ ['adaptive', new Set(['type', 'display', 'block_binding'])],
75
+ ['disabled', new Set(['type'])], ['between_tools', new Set(['type'])],
76
+ ]);
77
+ const CALLER_FIELDS = new Map([
78
+ ['direct', new Set(['type'])], ['code_execution_20250825', new Set(['type', 'tool_id'])],
79
+ ['code_execution_20260120', new Set(['type', 'tool_id'])],
80
+ ]);
81
+ // Claude's versioned classifier context is opaque permission-review data;
82
+ // only its outer contract and version determine whether routing is supported.
83
+ const SAFEGUARD_FIELDS = new Set(['type', 'classifier_context']);
84
+ const compatible = Object.freeze({ compatible: true });
85
+ const incompatible = reason => ({ compatible: false, reason });
9
86
 
10
87
  export function hasRoutableSafeguards(body) {
11
- return MODELS.has(body.model) && Array.isArray(body.safeguards) && body.safeguards.length > 0
88
+ return modelCapabilities(body?.model)?.sharedAuto === true && Array.isArray(body.safeguards) && body.safeguards.length > 0
12
89
  && body.safeguards.every(entry => object(entry) && entry.type === 'dangerous_tool_use'
13
- && object(entry.classifier_context) && entry.classifier_context.v === 1);
90
+ && knownFields(entry, SAFEGUARD_FIELDS) && object(entry.classifier_context) && entry.classifier_context.v === 1);
14
91
  }
15
92
 
16
93
  export function canRouteAutoRequest(body, target) {
17
- if (!MODELS.has(body.model) || !MODELS.has(target)) return false;
18
- if (body.safeguards !== undefined && !hasRoutableSafeguards(body)) return false;
19
- if (body.max_tokens !== undefined && (!Number.isSafeInteger(body.max_tokens) || body.max_tokens < 1 || body.max_tokens > 128000)) return false;
94
+ return checkTarget(body, target, true).compatible;
95
+ }
96
+
97
+ /**
98
+ * Check a proposed model change after the same explicit thinking adaptation
99
+ * used by inference and token counting. Native same-model requests belong to
100
+ * the provider: this is a routing guard, not a replacement API validator.
101
+ * @returns {{compatible: boolean, reason?: string}}
102
+ */
103
+ export function targetCompatibility(body, target, { autoMode = false } = {}) {
104
+ if (typeof target === 'string' && body?.model === target) return compatible;
105
+ return checkTarget(body, target, autoMode);
106
+ }
107
+
108
+ function checkTarget(body, target, autoMode) {
109
+ const sourceFacts = modelCapabilities(body?.model);
110
+ const targetFacts = modelCapabilities(target);
111
+ if (!sourceFacts || !targetFacts) return incompatible('unknown_model');
112
+ if (!object(body) || !Array.isArray(body.messages)) return incompatible('invalid_request_shape');
113
+ if (autoMode && (!sourceFacts.sharedAuto || !targetFacts.sharedAuto)) return incompatible('auto_model');
114
+ if (Object.keys(body).some(key => !REQUEST_FIELDS.has(key))) return incompatible('request_extension');
115
+ if (!optionalEnvelope(body.cache_control, CACHE_FIELDS)) return incompatible('request_extension');
116
+ if (body.safeguards !== undefined && (!sourceFacts.sharedAuto || !targetFacts.sharedAuto || !hasRoutableSafeguards(body))) return incompatible('safeguards');
117
+ if (body.max_tokens !== undefined && (!Number.isSafeInteger(body.max_tokens) || body.max_tokens < 0
118
+ || body.max_tokens > targetFacts.maxOutputTokens)) return incompatible('output_limit');
20
119
  // Retain model-specific execution facilities whose contracts differ between
21
120
  // models. Ordinary Claude Code tools and native context editing are shared.
22
- if (body.speed !== undefined && body.speed !== 'standard') return false;
23
- if (body.container !== undefined || body.mcp_servers !== undefined || body.compaction !== undefined) return false;
24
- if (body.tools !== undefined && (!Array.isArray(body.tools) || !body.tools.every(sharedTool))) return false;
121
+ if (body.speed !== undefined && body.speed !== 'standard') return incompatible('speed');
122
+ if (body.container !== undefined || body.mcp_servers !== undefined || body.compaction !== undefined) return incompatible('execution_facility');
123
+ if (body.tools !== undefined && (!Array.isArray(body.tools) || !body.tools.every(sharedTool))) return incompatible('tool_type');
124
+ const contents = Array.isArray(body.system) ? [body.system] : [];
25
125
  for (const message of body.messages) {
126
+ if (!object(message)) return incompatible('invalid_request_shape');
127
+ if (!knownFields(message, MESSAGE_FIELDS)) return incompatible('content_extension');
128
+ if (message.role === 'system' && !targetFacts.midConversationSystem) return incompatible('system_message');
129
+ if (message.output_config != null && !targetFacts.perMessageEffort) return incompatible('message_effort');
130
+ if (message.output_config != null && (!object(message.output_config)
131
+ || Object.keys(message.output_config).some(key => key !== 'effort')
132
+ || (message.output_config.effort != null && !targetFacts.effortLevels.includes(message.output_config.effort)))) return incompatible('message_effort');
133
+ if (Array.isArray(message.content)) contents.push(message.content);
26
134
  if (message.role !== 'system' || !Array.isArray(message.content)) continue;
27
135
  for (const block of message.content) {
136
+ if (!object(block)) return incompatible('invalid_request_shape');
28
137
  if (!['tool_addition', 'tool_removal'].includes(block.type)) continue;
29
138
  const tool = block.tool;
30
139
  if (!object(tool) || (tool.type !== 'tool_reference'
31
- && !(block.type === 'tool_addition' && tool.type === 'tool_definition' && sharedTool(tool.definition)))) return false;
140
+ && !(block.type === 'tool_addition' && tool.type === 'tool_definition' && sharedTool(tool.definition)))
141
+ || !knownFields(tool, tool.type === 'tool_reference' ? TOOL_REFERENCE_FIELDS : TOOL_DEFINITION_FIELDS)) return incompatible('inline_tool');
32
142
  }
33
143
  }
34
- if (body.tool_choice !== undefined && (!object(body.tool_choice) || !['auto', 'none'].includes(body.tool_choice.type))) return false;
35
- if (body.context_management !== undefined) {
144
+ // Inspect known content containers only; tool input and document payloads
145
+ // are opaque. Unknown block types retain their source model, never stripped.
146
+ while (contents.length) for (const block of contents.pop()) {
147
+ if (!object(block) || !CONTENT_TYPES.has(block.type) || !knownFields(block, CONTENT_FIELDS[block.type])) return incompatible('content_extension');
148
+ if (!optionalEnvelope(block.cache_control, CACHE_FIELDS)) return incompatible('content_extension');
149
+ if (block.type === 'tool_use' && block.caller !== undefined && (!object(block.caller)
150
+ || !CALLER_FIELDS.has(block.caller.type) || !knownFields(block.caller, CALLER_FIELDS.get(block.caller.type)))) return incompatible('content_extension');
151
+ if (block.type === 'image' && object(block.transformations)
152
+ && !knownFields(block.transformations, TRANSFORMATION_FIELDS)) return incompatible('content_extension');
153
+ if (block.type === 'tool_result' && Array.isArray(block.content)) contents.push(block.content);
154
+ if (block.type === 'tool_search_tool_result') {
155
+ const result = block.content;
156
+ if (!object(result) || !['tool_search_tool_search_result', 'tool_search_tool_result_error'].includes(result.type)
157
+ || !knownFields(result, CONTENT_FIELDS[result.type])) return incompatible('content_extension');
158
+ if (result.type === 'tool_search_tool_search_result') contents.push([result]);
159
+ }
160
+ if (block.type === 'tool_search_tool_search_result') {
161
+ if (!Array.isArray(block.tool_references)) return incompatible('content_extension');
162
+ contents.push(block.tool_references);
163
+ }
164
+ }
165
+ if (!targetFacts.assistantPrefill && body.messages.at(-1)?.role === 'assistant') return incompatible('assistant_prefill');
166
+ if (body.tool_choice !== undefined) {
167
+ if (!object(body.tool_choice) || !TOOL_CHOICE_FIELDS.has(body.tool_choice.type)
168
+ || !knownFields(body.tool_choice, TOOL_CHOICE_FIELDS.get(body.tool_choice.type))) return incompatible('tool_choice');
169
+ if (['any', 'tool'].includes(body.tool_choice.type) && (autoMode || !targetFacts.forcedToolChoice
170
+ || body.thinking?.type === 'enabled')) return incompatible('forced_tool_choice');
171
+ }
172
+ if (body.context_management != null) {
36
173
  const context = body.context_management;
37
- if (!object(context) || Object.keys(context).some(key => key !== 'edits') || !Array.isArray(context.edits)
38
- || context.edits.some(edit => !object(edit) || !CONTEXT_EDITS.has(edit.type))) return false;
174
+ if (!sourceFacts.sharedAuto || !targetFacts.sharedAuto || !object(context)
175
+ || Object.keys(context).some(key => key !== 'edits') || !Array.isArray(context.edits)
176
+ || !context.edits.every(sharedEdit)) return incompatible('context_management');
39
177
  }
40
178
  if (body.thinking !== undefined) {
41
179
  const thinking = body.thinking;
42
- if (!object(thinking) || !['adaptive', 'disabled', 'between_tools'].includes(thinking.type)) return false;
180
+ if (!object(thinking) || (autoMode && !['adaptive', 'disabled', 'between_tools'].includes(thinking.type))) return incompatible('thinking_mode');
181
+ if (!THINKING_FIELDS.has(thinking.type) || !knownFields(thinking, THINKING_FIELDS.get(thinking.type))
182
+ || !optionalEnvelope(thinking.block_binding, THINKING_BINDING_FIELDS)) return incompatible('thinking_extension');
43
183
  // between_tools is Sonnet 5.5-specific. A routed Opus request uses
44
184
  // adaptive thinking; prepareRequest makes that explicit without touching
45
185
  // any prior thinking blocks or the conversation prefix they sign.
46
186
  if (thinking.type === 'between_tools' && (body.model !== 'claude-sonnet-5-5'
47
- || Object.keys(thinking).some(key => key !== 'type') || target === 'claude-sonnet-5')) return false;
187
+ || Object.keys(thinking).some(key => key !== 'type') || target === 'claude-sonnet-5')) return incompatible('thinking_mode');
188
+ const adapted = prepareRequest(body, target).request.thinking;
189
+ if (!targetFacts.thinkingTypes.includes(adapted.type)) return incompatible('thinking_mode');
190
+ if (adapted.type === 'between_tools' && (Object.keys(adapted).length !== 1
191
+ || ['xhigh', 'max'].includes(body.output_config?.effort)
192
+ || body.messages.some(message => message.output_config?.effort !== undefined
193
+ && message.output_config.effort !== (body.output_config?.effort ?? 'high')))) return incompatible('thinking_effort');
194
+ if (target === 'claude-opus-5' && adapted.type === 'disabled'
195
+ && ['xhigh', 'max'].includes(body.output_config?.effort)) return incompatible('thinking_effort');
196
+ }
197
+ if (body.output_config !== undefined) {
198
+ const output = body.output_config;
199
+ if (!object(output) || Object.keys(output).some(key => !OUTPUT_FIELDS.has(key))) return incompatible('output_extension');
200
+ if (!optionalEnvelope(output.format, OUTPUT_FORMAT_FIELDS)
201
+ || (output.format != null && output.format.type !== 'json_schema')) return incompatible('output_extension');
202
+ if (!optionalEnvelope(output.task_budget, TASK_BUDGET_FIELDS)
203
+ || (output.task_budget != null && output.task_budget.type !== 'tokens')) return incompatible('task_budget');
204
+ if (output.effort != null && !targetFacts.effortLevels.includes(output.effort)) return incompatible('effort');
205
+ if (output.task_budget != null && !targetFacts.taskBudget) return incompatible('task_budget');
48
206
  }
49
- // Sonnet 5 lacks mid-conversation system/tool/effort updates. Never flatten
50
- // those messages into the top-level prompt: that invalidates signed history.
51
- if (target === 'claude-sonnet-5' && (body.output_config?.task_budget !== undefined
52
- || body.messages.some(message => message.role === 'system' || message.output_config !== undefined))) return false;
53
- return true;
207
+ // Modern models reject non-default sampling. Shared modern source requests
208
+ // retain their own validation; upgrading an older model must not introduce
209
+ // this new restriction. Do not guess an undocumented numeric top_k default.
210
+ if (targetFacts.defaultSamplingOnly && !sourceFacts.defaultSamplingOnly
211
+ && ((body.temperature !== undefined && body.temperature !== 1)
212
+ || (body.top_p !== undefined && body.top_p !== 1) || body.top_k !== undefined)) return incompatible('sampling');
213
+ return compatible;
54
214
  }
@@ -0,0 +1,57 @@
1
+ export const DECISION_RESPONSE_LIMIT = 64 * 1024;
2
+ export const MODEL_METADATA_LIMIT = 1024 * 1024;
3
+
4
+ // Cancellation is best effort and must never await an uncooperative body's
5
+ // cancel promise. Fetch's signal remains responsible for its network request.
6
+ export function cancelResponseBody(response, reason) {
7
+ try { Promise.resolve(response.body?.cancel(reason)).catch(() => {}); } catch {}
8
+ }
9
+
10
+ /**
11
+ * Read at most limit bytes, observing cancellation throughout body delivery.
12
+ * @param {Response} response
13
+ * @param {{signal?:AbortSignal,limit?:number}} [options]
14
+ */
15
+ export async function readBoundedJson(response, { signal, limit = DECISION_RESPONSE_LIMIT } = {}) {
16
+ if (!Number.isSafeInteger(limit) || limit < 1) throw new Error('classifier_invalid_response');
17
+ let reader;
18
+ let bytes;
19
+ let total = 0;
20
+ const cancel = () => {
21
+ try { Promise.resolve(reader?.cancel(signal?.reason)).catch(() => {}); } catch {}
22
+ };
23
+ try {
24
+ signal?.throwIfAborted();
25
+ reader = response.body?.getReader();
26
+ if (!reader) throw new Error('classifier_invalid_response');
27
+ signal?.addEventListener('abort', cancel, { once: true });
28
+ // Header sizes are only an early rejection. The streamed byte count is
29
+ // authoritative for chunked, compressed, absent or inaccurate lengths.
30
+ const length = response.headers?.get('content-length');
31
+ if (length && /^\d+$/.test(length) && Number(length) > limit) throw new Error('classifier_invalid_response');
32
+ while (true) {
33
+ signal?.throwIfAborted();
34
+ const { value, done } = await reader.read();
35
+ signal?.throwIfAborted();
36
+ if (done) break;
37
+ if (!(value instanceof Uint8Array) || value.byteLength > limit - total) throw new Error('classifier_invalid_response');
38
+ const next = total + value.byteLength;
39
+ // A single growable buffer also bounds metadata: a malicious one-byte
40
+ // chunk stream must not retain tens of thousands of Buffer objects.
41
+ if (!bytes || next > bytes.length) {
42
+ const capacity = Math.min(limit, Math.max(next, bytes ? bytes.length * 2 : Math.min(1024, limit)));
43
+ const grown = Buffer.allocUnsafe(capacity);
44
+ bytes?.copy(grown, 0, 0, total);
45
+ bytes = grown;
46
+ }
47
+ bytes.set(value, total);
48
+ total = next;
49
+ }
50
+ try { return JSON.parse(bytes?.toString('utf8', 0, total) ?? ''); }
51
+ catch { throw new Error('classifier_invalid_response'); }
52
+ } finally {
53
+ signal?.removeEventListener('abort', cancel);
54
+ if (reader) { cancel(); try { reader.releaseLock(); } catch {} }
55
+ else cancelResponseBody(response, signal?.reason);
56
+ }
57
+ }
@@ -0,0 +1,90 @@
1
+ const commands = {
2
+ setup: `Usage: claude-autorouter setup [options]
3
+
4
+ Configure the local Ollama evaluator (default) or TypeSafe Jev.
5
+ --auth-mode subscription|api-key Default: subscription
6
+ --client-profile compatible|native|auto
7
+ --evaluator ollama|jev Default: ollama (local); jev sends excerpts to TypeSafe
8
+ --ollama-model TAG Select a /v1/systemone model
9
+ --ollama-timeout-ms N 0 disables the routing deadline
10
+ --pull Download the selected missing Ollama model
11
+ --stop-hook-block-cap N Optional Claude Stop-hook retry limit
12
+ --session-log-dir DIR Opt in to private logs with prompt excerpts
13
+ --session-log-mode metadata|prompts Choose whether excerpts are included
14
+ --secret-store file|keychain default on macOS for new setups; file is plaintext
15
+ --force Update an existing configuration
16
+ --replace Explicitly rebuild the saved configuration
17
+
18
+ First setup reads environment settings and keys, or prompts for missing keys.
19
+ --force retains saved defaults and applies explicit options; unrelated runtime
20
+ overrides stay temporary. Explicit evaluator/auth selection accepts its supplied key.
21
+ Examples: claude-autorouter setup --pull
22
+ claude-autorouter setup --evaluator jev`,
23
+ doctor: `Usage: claude-autorouter doctor [--evaluate-local] [--json]
24
+
25
+ Check configuration, installed Claude, and local model availability.
26
+ --evaluate-local explicitly tests synthetic routing cases on the existing
27
+ Ollama model. No downloads, Anthropic/Jev calls, or configuration changes.
28
+ --json returns the local evaluation report (requires --evaluate-local).
29
+
30
+ Example: claude-autorouter doctor --evaluate-local`,
31
+ config: `Usage: claude-autorouter config show [--json] [--check-all]
32
+ claude-autorouter config set KEY VALUE
33
+ claude-autorouter config set KEY --stdin
34
+ claude-autorouter config unset KEY
35
+
36
+ Show effective settings and their source; secrets are always redacted.
37
+ --check-all also validates settings for the inactive evaluator.
38
+ Set/unset changes only the named saved setting. Environment values still win.
39
+ Secret keys require --stdin or a hidden prompt, never a command-line value.
40
+ Setting AUTOROUTER_SECRET_STORE to keychain or file moves saved keys (macOS).
41
+
42
+ Example: claude-autorouter config set AUTOROUTER_OLLAMA_TIMEOUT_MS 0`,
43
+ serve: `Usage: claude-autorouter serve
44
+
45
+ Run a local Messages API gateway. Requires configured evaluator credentials,
46
+ authentication, and AUTOROUTER_TOKEN (at least 16 characters).
47
+ Use claude-autorouter claude for managed startup and cleanup.`,
48
+ sessions: `Usage: claude-autorouter sessions list [--json]
49
+ claude-autorouter sessions show ID [--json]
50
+
51
+ Read optional local session logs from AUTOROUTER_SESSION_LOG_DIR.
52
+ List prints the IDs used by show. History includes routing choices, observed
53
+ models, request outcomes, latency, and API-equivalent savings coverage.
54
+ Logging is off by default. To enable metadata-only history:
55
+ claude-autorouter config set AUTOROUTER_SESSION_LOG_MODE metadata
56
+ claude-autorouter config set AUTOROUTER_SESSION_LOG_DIR /path/to/private/logs`,
57
+ claude: `Usage: claude-autorouter claude [Claude Code arguments]
58
+
59
+ Launch Claude with automatic routing and an AutoRouter status line.
60
+ Example: claude-autorouter claude --permission-mode auto
61
+ Auto mode switches between Sonnet and Opus on new human tasks.
62
+ Claude owns permission checks and subscription authentication.
63
+ --help and --version pass directly to Claude without starting the router.`,
64
+ };
65
+
66
+ export function helpText(command = 'help') {
67
+ return commands[command] ?? `AutoRouter — automatic model routing for Claude Code
68
+
69
+ Usage: claude-autorouter <command> [options]
70
+
71
+ setup Configure Jev or local Ollama; prompt for keys privately
72
+ doctor Check configuration and local dependencies
73
+ config Inspect or change saved settings without replacing them
74
+ sessions Read optional local decision and outcome history
75
+ claude Launch Claude Code with automatic routing
76
+ serve Run the local gateway separately
77
+ --version Print the AutoRouter version
78
+
79
+ Start: claude-autorouter setup
80
+ claude-autorouter doctor
81
+ claude-autorouter claude --permission-mode auto
82
+
83
+ Run claude-autorouter help <command> for options and examples.
84
+ Local Ollama is the default evaluator and uses only the local /v1/systemone
85
+ endpoint. Jev is optional; it receives bounded, redacted prompt excerpts. Complete requests go to Anthropic.
86
+ Session logging is off unless AUTOROUTER_SESSION_LOG_DIR is configured.
87
+ Environment variables override ~/.config/claude-autorouter/config.json.
88
+ AUTOROUTER_CONFIG selects another file. Project .env files are not auto-loaded.
89
+ Reference: docs/reference.md`;
90
+ }
@@ -0,0 +1,158 @@
1
+ import { readConfig, parseSessionLogDir, parseStopHookBlockCap } from './config.mjs';
2
+ import { CONFIG_KEYS, SECRET_CONFIG_KEYS, SECRET_STORES, keychainRemovals, loadUserConfig, saveUserConfig } from './user-config.mjs';
3
+ import { modelCapabilities } from './model-catalog.mjs';
4
+ import { askSecret } from './onboarding.mjs';
5
+
6
+ const secretKey = key => SECRET_CONFIG_KEYS.includes(key);
7
+ const providerFor = key => key.startsWith('AUTOROUTER_OLLAMA_') ? 'ollama'
8
+ : key.startsWith('AUTOROUTER_JEV_') || key === 'AUTOROUTER_MIN_CONFIDENCE' || key === 'TYPESAFE_API_KEY' ? 'jev' : undefined;
9
+ const safeText = value => String(value).replace(/[\u0000-\u001f\u007f-\u009f\u202a-\u202e\u2066-\u2069]/g, '');
10
+
11
+ function effectiveValues(config, env) {
12
+ return {
13
+ AUTOROUTER_AUTH_MODE: config.authMode, AUTOROUTER_CLIENT_PROFILE: config.clientProfile,
14
+ AUTOROUTER_EVALUATOR: config.evaluator, AUTOROUTER_UPSTREAM_URL: config.upstream,
15
+ AUTOROUTER_HAIKU_MODEL: config.models.haiku, AUTOROUTER_SONNET_MODEL: config.models.sonnet,
16
+ AUTOROUTER_OPUS_MODEL: config.models.opus, AUTOROUTER_PORT: config.port,
17
+ AUTOROUTER_TOKEN_COUNT_TIMEOUT_MS: config.tokenCountTimeoutMs,
18
+ AUTOROUTER_JEV_URL: config.jevEndpoint, AUTOROUTER_JEV_MODEL: config.jevModel,
19
+ AUTOROUTER_JEV_TIMEOUT_MS: config.jevTimeoutMs, AUTOROUTER_MIN_CONFIDENCE: config.minConfidence,
20
+ AUTOROUTER_OLLAMA_URL: config.ollamaEndpoint, AUTOROUTER_OLLAMA_MODEL: config.ollamaModel,
21
+ AUTOROUTER_OLLAMA_TIMEOUT_MS: config.ollamaTimeoutMs, AUTOROUTER_OLLAMA_KEEP_ALIVE: config.ollamaKeepAlive,
22
+ AUTOROUTER_SESSION_LOG_DIR: config.sessionLogDir ?? null,
23
+ AUTOROUTER_SESSION_LOG_MODE: config.sessionLogMode,
24
+ CLAUDE_CODE_STOP_HOOK_BLOCK_CAP: config.stopHookBlockCap ?? null,
25
+ AUTOROUTER_STATUSLINE: env.AUTOROUTER_STATUSLINE !== '0',
26
+ AUTOROUTER_DEBUG: env.AUTOROUTER_DEBUG === '1', ENABLE_TOOL_SEARCH: env.ENABLE_TOOL_SEARCH ?? 'true',
27
+ };
28
+ }
29
+
30
+ /** Redacted effective configuration, independent of Claude launch arguments. */
31
+ export function configReport(loaded, env, { checkAll = false } = {}) {
32
+ const config = readConfig(loaded.env);
33
+ const values = { ...effectiveValues(config, loaded.env), AUTOROUTER_SECRET_STORE: loaded.secretStore ?? 'file' };
34
+ const settings = {};
35
+ for (const key of CONFIG_KEYS) {
36
+ const saved = loaded.keychainSecrets?.includes(key) ? 'keychain' : 'file';
37
+ // The saved store setting governs saved secrets; the environment cannot redirect it.
38
+ const source = env[key] !== undefined && key !== 'AUTOROUTER_SECRET_STORE' ? 'environment'
39
+ : Object.hasOwn(loaded.values, key) ? saved : 'default';
40
+ const provider = providerFor(key);
41
+ const active = (!provider || provider === config.evaluator)
42
+ && (key !== 'ANTHROPIC_API_KEY' || config.authMode === 'api-key');
43
+ const secret = secretKey(key);
44
+ settings[key] = { source, active,
45
+ ...(source === 'environment' && Object.hasOwn(loaded.values, key) ? { overrides_file: true } : {}),
46
+ ...(source === 'environment' && loaded.unavailableSecrets?.includes(key) ? { keychain_unavailable: true } : {}),
47
+ ...(secret ? { secret: true, present: typeof loaded.env[key] === 'string' && Boolean(loaded.env[key].trim()) }
48
+ : { value: active ? values[key] ?? null : null }),
49
+ };
50
+ }
51
+ const warnings = [];
52
+ if (config.clientProfile === 'auto' && [config.models.sonnet, config.models.opus].some(model => !modelCapabilities(model)?.sharedAuto)) {
53
+ warnings.push('Auto permission eligibility does not guarantee automatic switching. Older targets can retain the incoming model for shared execution features.');
54
+ }
55
+ let valid = true, error;
56
+ if (checkAll) {
57
+ try { readConfig(loaded.env, { validateAll: true }); }
58
+ catch (failure) { valid = false; error = failure.message; }
59
+ }
60
+ return { schema_version: 1, config_path: loaded.path, config_exists: loaded.exists, valid,
61
+ checked: checkAll ? 'all_evaluators' : 'active_evaluator', settings, warnings,
62
+ ...(error ? { error } : {}),
63
+ };
64
+ }
65
+
66
+ async function stdinSecret(input) {
67
+ if (input.isTTY) throw new Error('Pipe the secret to --stdin, or omit --stdin to use the hidden prompt.');
68
+ const chunks = [];
69
+ let bytes = 0;
70
+ try {
71
+ for await (const chunk of input) {
72
+ const buffer = Buffer.isBuffer(chunk) ? chunk : Buffer.from(chunk);
73
+ bytes += buffer.length;
74
+ if (bytes > 16384) throw new Error('Secret input exceeds the supported size.');
75
+ chunks.push(buffer);
76
+ }
77
+ } catch { throw new Error('Could not read a bounded secret from stdin.'); }
78
+ return Buffer.concat(chunks).toString('utf8').replace(/\r?\n$/, '');
79
+ }
80
+
81
+ function normalizedValue(key, value) {
82
+ if (typeof value !== 'string') throw new Error('Configuration values must be strings.');
83
+ if (secretKey(key)) {
84
+ const normalized = value.trim();
85
+ if (!normalized || /[\r\n\0]/.test(normalized)) throw new Error(`${key} must be a nonempty, single-line secret.`);
86
+ if (key === 'AUTOROUTER_TOKEN' && normalized.length < 16) throw new Error('AUTOROUTER_TOKEN must contain at least 16 characters.');
87
+ return normalized;
88
+ }
89
+ if (key === 'AUTOROUTER_SESSION_LOG_DIR') return parseSessionLogDir(value) ?? '';
90
+ if (key === 'CLAUDE_CODE_STOP_HOOK_BLOCK_CAP') return String(parseStopHookBlockCap(value));
91
+ if (key === 'AUTOROUTER_SECRET_STORE' && !SECRET_STORES.includes(value)) throw new Error('AUTOROUTER_SECRET_STORE must be file or keychain.');
92
+ if (/[\r\n\0\u001b]/.test(value)) throw new Error(`${key} must be a single-line setting.`);
93
+ return value;
94
+ }
95
+
96
+ export async function configCommand(args, {
97
+ env = process.env, write = console.log, input = process.stdin, promptSecret = askSecret, keychain,
98
+ } = {}) {
99
+ const store = keychain ? { keychain } : {};
100
+ const [operation, ...rest] = args;
101
+ if (operation === 'show') {
102
+ if (rest.some(arg => !['--json', '--check-all'].includes(arg))) throw new Error('Usage: claude-autorouter config show [--json] [--check-all]');
103
+ let loaded, report;
104
+ try {
105
+ loaded = loadUserConfig(env, { allowMissing: true, ...store });
106
+ report = configReport(loaded, env, { checkAll: rest.includes('--check-all') });
107
+ }
108
+ catch (error) {
109
+ report = { schema_version: 1, ...(loaded ? { config_path: loaded.path, config_exists: loaded.exists } : {}), valid: false, error: error.message };
110
+ }
111
+ if (rest.includes('--json')) write(JSON.stringify(report, null, 2));
112
+ else {
113
+ if (loaded) write(`Config: ${loaded.path}${loaded.exists ? '' : ' (not saved)'}`);
114
+ for (const [key, setting] of Object.entries(report.settings ?? {})) {
115
+ const value = setting.secret ? setting.present ? '[set; hidden]' : '[unset]'
116
+ : !setting.active ? '[inactive]' : setting.value === null ? '[unset]' : safeText(setting.value);
117
+ write(`${key}=${value} [${setting.source}${setting.overrides_file ? '; overrides file' : ''}${!setting.active ? '; inactive' : ''}]`);
118
+ }
119
+ for (const warning of report.warnings ?? []) write(warning);
120
+ if (report.error) write(`FAIL ${report.error}`);
121
+ }
122
+ return report.valid;
123
+ }
124
+ if (!['set', 'unset'].includes(operation)) throw new Error('Usage: claude-autorouter config show|set|unset');
125
+ const [key, value, ...extra] = rest;
126
+ // Never echo an unknown key: it may itself be a pasted credential.
127
+ if (!CONFIG_KEYS.includes(key)) throw new Error('Unsupported configuration key. Run claude-autorouter config show for supported settings.');
128
+ if (extra.length || (operation === 'unset' && value !== undefined)) throw new Error(`Usage: claude-autorouter config ${operation} KEY${operation === 'set' ? ' VALUE' : ''}`);
129
+ if (operation === 'set' && secretKey(key) && value !== undefined && value !== '--stdin') {
130
+ throw new Error('Secret values are not accepted as command arguments. Use --stdin or the hidden prompt.');
131
+ }
132
+ if (operation === 'set' && !secretKey(key) && (value === undefined || value === '--stdin')) throw new Error('Nonsecret settings require a value argument.');
133
+ const loaded = loadUserConfig(env, { allowMissing: true, ...store });
134
+ const next = { ...loaded.values };
135
+ if (operation === 'unset') delete next[key];
136
+ else next[key] = normalizedValue(key, secretKey(key)
137
+ ? value === '--stdin' ? await stdinSecret(input) : await promptSecret(key)
138
+ : value);
139
+ // Validate the persisted setting itself, not an environment value that could
140
+ // mask it. An explicitly edited inactive provider is checked too.
141
+ const provider = operation === 'set' ? providerFor(key) : undefined;
142
+ readConfig({ ...next, ...(provider ? { AUTOROUTER_EVALUATOR: provider } : {}) });
143
+ const nextStore = next.AUTOROUTER_SECRET_STORE ?? 'file';
144
+ // Changing the store moves saved secrets; unsetting a secret deletes its item.
145
+ const removeSecrets = key === 'AUTOROUTER_SECRET_STORE' ? keychainRemovals(loaded, nextStore)
146
+ : operation === 'unset' && secretKey(key) && loaded.secretStore === 'keychain' ? [key] : [];
147
+ saveUserConfig(next, { env, overwrite: loaded.exists, expectedRevision: loaded.revision, removeSecrets, ...store });
148
+ if (key === 'AUTOROUTER_SECRET_STORE') {
149
+ const moved = SECRET_CONFIG_KEYS.filter(name => Object.hasOwn(next, name)).length;
150
+ const where = nextStore === 'keychain' ? 'macOS Keychain' : 'configuration file';
151
+ write(nextStore === loaded.secretStore || !moved ? `Saved ${key}. Saved secrets are stored in the ${where}.`
152
+ : `Saved ${key}. Moved ${moved} saved secret${moved === 1 ? '' : 's'} to the ${where}.`);
153
+ return true;
154
+ }
155
+ write(`${operation === 'unset' ? 'Removed saved' : 'Saved'} ${key}${secretKey(key) && nextStore === 'keychain' ? ' in the macOS Keychain' : ''}. Other saved settings are unchanged.`);
156
+ if (env[key] !== undefined) write(`The current environment still overrides ${key}; unset that environment variable to use the saved/default value.`);
157
+ return true;
158
+ }