claude-autorouter 0.3.7 → 0.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +6 -3
- package/CODE_OF_CONDUCT.md +9 -0
- package/CONTRIBUTING.md +57 -0
- package/README.md +47 -70
- package/SECURITY.md +23 -0
- package/SUPPORT.md +18 -0
- package/bin/autorouter.mjs +40 -57
- package/docs/development.md +48 -2
- package/docs/hardware-benchmark.md +29 -0
- package/docs/hardware-comparison.md +55 -0
- package/docs/hardware-results-16gb.json +4002 -0
- package/docs/hardware-results-16gb.md +26 -0
- package/docs/hardware-results-64gb.json +4020 -0
- package/docs/reference.md +92 -37
- package/docs/releasing.md +79 -37
- package/docs/router-performance.json +1697 -0
- package/docs/router-performance.md +50 -0
- package/docs/status-performance.json +363 -0
- package/docs/status-performance.md +44 -0
- package/docs/subscription-integration.md +27 -0
- package/package.json +66 -10
- package/src/auto-routing.mjs +184 -24
- package/src/bounded-json.mjs +57 -0
- package/src/cli-help.mjs +90 -0
- package/src/config-command.mjs +158 -0
- package/src/config.mjs +53 -28
- package/src/contracts.mjs +123 -0
- package/src/evaluation-report.mjs +114 -0
- package/src/keychain.mjs +58 -0
- package/src/local-diagnostic.mjs +191 -0
- package/src/model-catalog.mjs +96 -0
- package/src/model-request.mjs +6 -7
- package/src/ollama-evaluator.mjs +9 -27
- package/src/onboarding.mjs +130 -26
- package/src/prompt-state.mjs +22 -7
- package/src/redaction.mjs +97 -0
- package/src/request-validation.mjs +54 -0
- package/src/response-observer.mjs +126 -18
- package/src/router.mjs +151 -61
- package/src/savings.mjs +74 -16
- package/src/server.mjs +79 -12
- package/src/session-history.mjs +262 -0
- package/src/session-log.mjs +9 -58
- package/src/status-state.mjs +110 -62
- package/src/statusline.mjs +57 -27
- package/src/telemetry-event.mjs +200 -0
- package/src/token-counter.mjs +3 -1
- package/src/turn-state.mjs +132 -0
- package/src/user-config.mjs +81 -10
package/src/auto-routing.mjs
CHANGED
|
@@ -1,54 +1,214 @@
|
|
|
1
|
+
import { modelCapabilities } from './model-catalog.mjs';
|
|
2
|
+
import { prepareRequest } from './model-request.mjs';
|
|
3
|
+
|
|
1
4
|
// Shared execution capabilities, not a replacement for Claude's permission
|
|
2
5
|
// classifier. Keep the complete safeguards contract and signed history on wire.
|
|
3
|
-
const MODELS = new Set(['claude-sonnet-5', 'claude-sonnet-5-5', 'claude-opus-5', 'claude-opus-5-5']);
|
|
4
6
|
const object = value => value !== null && typeof value === 'object' && !Array.isArray(value);
|
|
5
|
-
const SHARED_TOOLS = new Set(['custom', 'tool_search_tool_regex_20251119', 'tool_search_tool_bm25_20251119',
|
|
6
|
-
'bash_20250124', 'text_editor_20250728']);
|
|
7
7
|
const CONTEXT_EDITS = new Set(['clear_thinking_20251015', 'clear_tool_uses_20250919']);
|
|
8
|
-
|
|
8
|
+
// Reviewed 2026-10-05 against the Messages API/tool reference and context-edit
|
|
9
|
+
// contract. These are semantic envelopes; input schemas/examples and document
|
|
10
|
+
// data remain opaque and are never inspected or rewritten.
|
|
11
|
+
// https://platform.claude.com/docs/en/api/beta/messages/create
|
|
12
|
+
// https://platform.claude.com/docs/en/build-with-claude/mid-conversation-system-messages
|
|
13
|
+
// https://github.com/anthropics/anthropic-sdk-typescript/blob/main/src/resources/beta/messages/messages.ts
|
|
14
|
+
const CUSTOM_TOOL_FIELDS = new Set(['type', 'name', 'description', 'input_schema', 'cache_control',
|
|
15
|
+
'defer_loading', 'strict', 'input_examples', 'allowed_callers', 'eager_input_streaming']);
|
|
16
|
+
const SEARCH_TOOL_FIELDS = new Set(['type', 'name', 'allowed_callers', 'cache_control', 'defer_loading', 'strict']);
|
|
17
|
+
const BASH_TOOL_FIELDS = new Set([...SEARCH_TOOL_FIELDS, 'input_examples']);
|
|
18
|
+
const EDITOR_TOOL_FIELDS = new Set([...BASH_TOOL_FIELDS, 'max_characters']);
|
|
19
|
+
const TOOL_FIELDS = new Map([
|
|
20
|
+
['custom', CUSTOM_TOOL_FIELDS], ['bash_20250124', BASH_TOOL_FIELDS],
|
|
21
|
+
['text_editor_20250728', EDITOR_TOOL_FIELDS],
|
|
22
|
+
['tool_search_tool_regex_20251119', SEARCH_TOOL_FIELDS], ['tool_search_tool_bm25_20251119', SEARCH_TOOL_FIELDS],
|
|
23
|
+
]);
|
|
24
|
+
const TOOL_CHOICE_FIELDS = new Map([
|
|
25
|
+
['auto', new Set(['type', 'disable_parallel_tool_use'])], ['any', new Set(['type', 'disable_parallel_tool_use'])],
|
|
26
|
+
['tool', new Set(['type', 'name', 'disable_parallel_tool_use'])], ['none', new Set(['type'])],
|
|
27
|
+
]);
|
|
28
|
+
const MESSAGE_FIELDS = new Set(['role', 'content', 'output_config', 'clear_at']);
|
|
29
|
+
const THINKING_EDIT_FIELDS = new Set(['type', 'keep']);
|
|
30
|
+
const TOOL_EDIT_FIELDS = new Set(['type', 'clear_at_least', 'clear_tool_inputs', 'exclude_tools', 'keep', 'trigger']);
|
|
31
|
+
const EDIT_VALUE_FIELDS = new Set(['type', 'value']);
|
|
32
|
+
const knownFields = (value, fields) => Object.keys(value).every(key => fields.has(key));
|
|
33
|
+
const optionalEnvelope = (value, fields) => value == null || (object(value) && knownFields(value, fields));
|
|
34
|
+
const sharedTool = tool => object(tool) && TOOL_FIELDS.has(tool.type ?? 'custom')
|
|
35
|
+
&& knownFields(tool, TOOL_FIELDS.get(tool.type ?? 'custom')) && optionalEnvelope(tool.cache_control, CACHE_FIELDS);
|
|
36
|
+
const sharedEdit = edit => object(edit) && CONTEXT_EDITS.has(edit.type)
|
|
37
|
+
&& knownFields(edit, edit.type === 'clear_thinking_20251015' ? THINKING_EDIT_FIELDS : TOOL_EDIT_FIELDS)
|
|
38
|
+
&& ['keep', 'trigger', 'clear_at_least'].every(key => !object(edit[key]) || knownFields(edit[key], EDIT_VALUE_FIELDS));
|
|
39
|
+
const REQUEST_FIELDS = new Set(['model', 'messages', 'system', 'max_tokens', 'metadata', 'stream',
|
|
40
|
+
'stop_sequences', 'temperature', 'top_p', 'top_k', 'tools', 'tool_choice', 'thinking', 'output_config',
|
|
41
|
+
'context_management', 'speed', 'container', 'mcp_servers', 'compaction', 'safeguards', 'service_tier',
|
|
42
|
+
'inference_geo', 'cache_control']);
|
|
43
|
+
const CONTENT_TYPES = new Set(['text', 'image', 'document', 'tool_use', 'tool_result', 'thinking',
|
|
44
|
+
'redacted_thinking', 'tool_reference', 'tool_search_tool_result', 'tool_search_tool_search_result',
|
|
45
|
+
'tool_addition', 'tool_removal']);
|
|
46
|
+
// The beta API distinguishes semantic block envelopes from opaque payloads.
|
|
47
|
+
// See its BetaContentBlockParam union and the matching official SDK types:
|
|
48
|
+
// https://github.com/anthropics/anthropic-sdk-python/tree/main/src/anthropic/types/beta
|
|
49
|
+
const CONTENT_FIELDS = Object.fromEntries(Object.entries({
|
|
50
|
+
text: ['type', 'text', 'cache_control', 'citations'],
|
|
51
|
+
image: ['type', 'source', 'cache_control', 'transformations'],
|
|
52
|
+
document: ['type', 'source', 'cache_control', 'citations', 'context', 'title'],
|
|
53
|
+
tool_use: ['type', 'id', 'input', 'name', 'cache_control', 'caller', 'toolset_name'],
|
|
54
|
+
tool_result: ['type', 'tool_use_id', 'content', 'is_error', 'cache_control', 'toolset_name'],
|
|
55
|
+
thinking: ['type', 'signature', 'thinking'],
|
|
56
|
+
redacted_thinking: ['type', 'data'],
|
|
57
|
+
tool_reference: ['type', 'tool_name', 'cache_control'],
|
|
58
|
+
tool_search_tool_result: ['type', 'content', 'tool_use_id', 'cache_control'],
|
|
59
|
+
tool_search_tool_search_result: ['type', 'tool_references'],
|
|
60
|
+
tool_search_tool_result_error: ['type', 'error_code', 'error_message'],
|
|
61
|
+
tool_addition: ['type', 'tool', 'cache_control'],
|
|
62
|
+
tool_removal: ['type', 'tool', 'cache_control'],
|
|
63
|
+
}).map(([type, fields]) => [type, new Set(fields)]));
|
|
64
|
+
const CACHE_FIELDS = new Set(['type', 'ttl']);
|
|
65
|
+
const TRANSFORMATION_FIELDS = new Set(['oversized_image']);
|
|
66
|
+
const TOOL_REFERENCE_FIELDS = new Set(['type', 'name']);
|
|
67
|
+
const TOOL_DEFINITION_FIELDS = new Set(['type', 'definition']);
|
|
68
|
+
const OUTPUT_FIELDS = new Set(['effort', 'format', 'task_budget']);
|
|
69
|
+
const OUTPUT_FORMAT_FIELDS = new Set(['type', 'schema']);
|
|
70
|
+
const TASK_BUDGET_FIELDS = new Set(['type', 'total', 'remaining']);
|
|
71
|
+
const THINKING_BINDING_FIELDS = new Set(['prefix_mismatch_behavior']);
|
|
72
|
+
const THINKING_FIELDS = new Map([
|
|
73
|
+
['enabled', new Set(['type', 'budget_tokens', 'display', 'block_binding'])],
|
|
74
|
+
['adaptive', new Set(['type', 'display', 'block_binding'])],
|
|
75
|
+
['disabled', new Set(['type'])], ['between_tools', new Set(['type'])],
|
|
76
|
+
]);
|
|
77
|
+
const CALLER_FIELDS = new Map([
|
|
78
|
+
['direct', new Set(['type'])], ['code_execution_20250825', new Set(['type', 'tool_id'])],
|
|
79
|
+
['code_execution_20260120', new Set(['type', 'tool_id'])],
|
|
80
|
+
]);
|
|
81
|
+
// Claude's versioned classifier context is opaque permission-review data;
|
|
82
|
+
// only its outer contract and version determine whether routing is supported.
|
|
83
|
+
const SAFEGUARD_FIELDS = new Set(['type', 'classifier_context']);
|
|
84
|
+
const compatible = Object.freeze({ compatible: true });
|
|
85
|
+
const incompatible = reason => ({ compatible: false, reason });
|
|
9
86
|
|
|
10
87
|
export function hasRoutableSafeguards(body) {
|
|
11
|
-
return
|
|
88
|
+
return modelCapabilities(body?.model)?.sharedAuto === true && Array.isArray(body.safeguards) && body.safeguards.length > 0
|
|
12
89
|
&& body.safeguards.every(entry => object(entry) && entry.type === 'dangerous_tool_use'
|
|
13
|
-
&& object(entry.classifier_context) && entry.classifier_context.v === 1);
|
|
90
|
+
&& knownFields(entry, SAFEGUARD_FIELDS) && object(entry.classifier_context) && entry.classifier_context.v === 1);
|
|
14
91
|
}
|
|
15
92
|
|
|
16
93
|
export function canRouteAutoRequest(body, target) {
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
94
|
+
return checkTarget(body, target, true).compatible;
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
/**
|
|
98
|
+
* Check a proposed model change after the same explicit thinking adaptation
|
|
99
|
+
* used by inference and token counting. Native same-model requests belong to
|
|
100
|
+
* the provider: this is a routing guard, not a replacement API validator.
|
|
101
|
+
* @returns {{compatible: boolean, reason?: string}}
|
|
102
|
+
*/
|
|
103
|
+
export function targetCompatibility(body, target, { autoMode = false } = {}) {
|
|
104
|
+
if (typeof target === 'string' && body?.model === target) return compatible;
|
|
105
|
+
return checkTarget(body, target, autoMode);
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
function checkTarget(body, target, autoMode) {
|
|
109
|
+
const sourceFacts = modelCapabilities(body?.model);
|
|
110
|
+
const targetFacts = modelCapabilities(target);
|
|
111
|
+
if (!sourceFacts || !targetFacts) return incompatible('unknown_model');
|
|
112
|
+
if (!object(body) || !Array.isArray(body.messages)) return incompatible('invalid_request_shape');
|
|
113
|
+
if (autoMode && (!sourceFacts.sharedAuto || !targetFacts.sharedAuto)) return incompatible('auto_model');
|
|
114
|
+
if (Object.keys(body).some(key => !REQUEST_FIELDS.has(key))) return incompatible('request_extension');
|
|
115
|
+
if (!optionalEnvelope(body.cache_control, CACHE_FIELDS)) return incompatible('request_extension');
|
|
116
|
+
if (body.safeguards !== undefined && (!sourceFacts.sharedAuto || !targetFacts.sharedAuto || !hasRoutableSafeguards(body))) return incompatible('safeguards');
|
|
117
|
+
if (body.max_tokens !== undefined && (!Number.isSafeInteger(body.max_tokens) || body.max_tokens < 0
|
|
118
|
+
|| body.max_tokens > targetFacts.maxOutputTokens)) return incompatible('output_limit');
|
|
20
119
|
// Retain model-specific execution facilities whose contracts differ between
|
|
21
120
|
// models. Ordinary Claude Code tools and native context editing are shared.
|
|
22
|
-
if (body.speed !== undefined && body.speed !== 'standard') return
|
|
23
|
-
if (body.container !== undefined || body.mcp_servers !== undefined || body.compaction !== undefined) return
|
|
24
|
-
if (body.tools !== undefined && (!Array.isArray(body.tools) || !body.tools.every(sharedTool))) return
|
|
121
|
+
if (body.speed !== undefined && body.speed !== 'standard') return incompatible('speed');
|
|
122
|
+
if (body.container !== undefined || body.mcp_servers !== undefined || body.compaction !== undefined) return incompatible('execution_facility');
|
|
123
|
+
if (body.tools !== undefined && (!Array.isArray(body.tools) || !body.tools.every(sharedTool))) return incompatible('tool_type');
|
|
124
|
+
const contents = Array.isArray(body.system) ? [body.system] : [];
|
|
25
125
|
for (const message of body.messages) {
|
|
126
|
+
if (!object(message)) return incompatible('invalid_request_shape');
|
|
127
|
+
if (!knownFields(message, MESSAGE_FIELDS)) return incompatible('content_extension');
|
|
128
|
+
if (message.role === 'system' && !targetFacts.midConversationSystem) return incompatible('system_message');
|
|
129
|
+
if (message.output_config != null && !targetFacts.perMessageEffort) return incompatible('message_effort');
|
|
130
|
+
if (message.output_config != null && (!object(message.output_config)
|
|
131
|
+
|| Object.keys(message.output_config).some(key => key !== 'effort')
|
|
132
|
+
|| (message.output_config.effort != null && !targetFacts.effortLevels.includes(message.output_config.effort)))) return incompatible('message_effort');
|
|
133
|
+
if (Array.isArray(message.content)) contents.push(message.content);
|
|
26
134
|
if (message.role !== 'system' || !Array.isArray(message.content)) continue;
|
|
27
135
|
for (const block of message.content) {
|
|
136
|
+
if (!object(block)) return incompatible('invalid_request_shape');
|
|
28
137
|
if (!['tool_addition', 'tool_removal'].includes(block.type)) continue;
|
|
29
138
|
const tool = block.tool;
|
|
30
139
|
if (!object(tool) || (tool.type !== 'tool_reference'
|
|
31
|
-
&& !(block.type === 'tool_addition' && tool.type === 'tool_definition' && sharedTool(tool.definition)))
|
|
140
|
+
&& !(block.type === 'tool_addition' && tool.type === 'tool_definition' && sharedTool(tool.definition)))
|
|
141
|
+
|| !knownFields(tool, tool.type === 'tool_reference' ? TOOL_REFERENCE_FIELDS : TOOL_DEFINITION_FIELDS)) return incompatible('inline_tool');
|
|
32
142
|
}
|
|
33
143
|
}
|
|
34
|
-
|
|
35
|
-
|
|
144
|
+
// Inspect known content containers only; tool input and document payloads
|
|
145
|
+
// are opaque. Unknown block types retain their source model, never stripped.
|
|
146
|
+
while (contents.length) for (const block of contents.pop()) {
|
|
147
|
+
if (!object(block) || !CONTENT_TYPES.has(block.type) || !knownFields(block, CONTENT_FIELDS[block.type])) return incompatible('content_extension');
|
|
148
|
+
if (!optionalEnvelope(block.cache_control, CACHE_FIELDS)) return incompatible('content_extension');
|
|
149
|
+
if (block.type === 'tool_use' && block.caller !== undefined && (!object(block.caller)
|
|
150
|
+
|| !CALLER_FIELDS.has(block.caller.type) || !knownFields(block.caller, CALLER_FIELDS.get(block.caller.type)))) return incompatible('content_extension');
|
|
151
|
+
if (block.type === 'image' && object(block.transformations)
|
|
152
|
+
&& !knownFields(block.transformations, TRANSFORMATION_FIELDS)) return incompatible('content_extension');
|
|
153
|
+
if (block.type === 'tool_result' && Array.isArray(block.content)) contents.push(block.content);
|
|
154
|
+
if (block.type === 'tool_search_tool_result') {
|
|
155
|
+
const result = block.content;
|
|
156
|
+
if (!object(result) || !['tool_search_tool_search_result', 'tool_search_tool_result_error'].includes(result.type)
|
|
157
|
+
|| !knownFields(result, CONTENT_FIELDS[result.type])) return incompatible('content_extension');
|
|
158
|
+
if (result.type === 'tool_search_tool_search_result') contents.push([result]);
|
|
159
|
+
}
|
|
160
|
+
if (block.type === 'tool_search_tool_search_result') {
|
|
161
|
+
if (!Array.isArray(block.tool_references)) return incompatible('content_extension');
|
|
162
|
+
contents.push(block.tool_references);
|
|
163
|
+
}
|
|
164
|
+
}
|
|
165
|
+
if (!targetFacts.assistantPrefill && body.messages.at(-1)?.role === 'assistant') return incompatible('assistant_prefill');
|
|
166
|
+
if (body.tool_choice !== undefined) {
|
|
167
|
+
if (!object(body.tool_choice) || !TOOL_CHOICE_FIELDS.has(body.tool_choice.type)
|
|
168
|
+
|| !knownFields(body.tool_choice, TOOL_CHOICE_FIELDS.get(body.tool_choice.type))) return incompatible('tool_choice');
|
|
169
|
+
if (['any', 'tool'].includes(body.tool_choice.type) && (autoMode || !targetFacts.forcedToolChoice
|
|
170
|
+
|| body.thinking?.type === 'enabled')) return incompatible('forced_tool_choice');
|
|
171
|
+
}
|
|
172
|
+
if (body.context_management != null) {
|
|
36
173
|
const context = body.context_management;
|
|
37
|
-
if (!
|
|
38
|
-
|| context.
|
|
174
|
+
if (!sourceFacts.sharedAuto || !targetFacts.sharedAuto || !object(context)
|
|
175
|
+
|| Object.keys(context).some(key => key !== 'edits') || !Array.isArray(context.edits)
|
|
176
|
+
|| !context.edits.every(sharedEdit)) return incompatible('context_management');
|
|
39
177
|
}
|
|
40
178
|
if (body.thinking !== undefined) {
|
|
41
179
|
const thinking = body.thinking;
|
|
42
|
-
if (!object(thinking) || !['adaptive', 'disabled', 'between_tools'].includes(thinking.type)) return
|
|
180
|
+
if (!object(thinking) || (autoMode && !['adaptive', 'disabled', 'between_tools'].includes(thinking.type))) return incompatible('thinking_mode');
|
|
181
|
+
if (!THINKING_FIELDS.has(thinking.type) || !knownFields(thinking, THINKING_FIELDS.get(thinking.type))
|
|
182
|
+
|| !optionalEnvelope(thinking.block_binding, THINKING_BINDING_FIELDS)) return incompatible('thinking_extension');
|
|
43
183
|
// between_tools is Sonnet 5.5-specific. A routed Opus request uses
|
|
44
184
|
// adaptive thinking; prepareRequest makes that explicit without touching
|
|
45
185
|
// any prior thinking blocks or the conversation prefix they sign.
|
|
46
186
|
if (thinking.type === 'between_tools' && (body.model !== 'claude-sonnet-5-5'
|
|
47
|
-
|| Object.keys(thinking).some(key => key !== 'type') || target === 'claude-sonnet-5')) return
|
|
187
|
+
|| Object.keys(thinking).some(key => key !== 'type') || target === 'claude-sonnet-5')) return incompatible('thinking_mode');
|
|
188
|
+
const adapted = prepareRequest(body, target).request.thinking;
|
|
189
|
+
if (!targetFacts.thinkingTypes.includes(adapted.type)) return incompatible('thinking_mode');
|
|
190
|
+
if (adapted.type === 'between_tools' && (Object.keys(adapted).length !== 1
|
|
191
|
+
|| ['xhigh', 'max'].includes(body.output_config?.effort)
|
|
192
|
+
|| body.messages.some(message => message.output_config?.effort !== undefined
|
|
193
|
+
&& message.output_config.effort !== (body.output_config?.effort ?? 'high')))) return incompatible('thinking_effort');
|
|
194
|
+
if (target === 'claude-opus-5' && adapted.type === 'disabled'
|
|
195
|
+
&& ['xhigh', 'max'].includes(body.output_config?.effort)) return incompatible('thinking_effort');
|
|
196
|
+
}
|
|
197
|
+
if (body.output_config !== undefined) {
|
|
198
|
+
const output = body.output_config;
|
|
199
|
+
if (!object(output) || Object.keys(output).some(key => !OUTPUT_FIELDS.has(key))) return incompatible('output_extension');
|
|
200
|
+
if (!optionalEnvelope(output.format, OUTPUT_FORMAT_FIELDS)
|
|
201
|
+
|| (output.format != null && output.format.type !== 'json_schema')) return incompatible('output_extension');
|
|
202
|
+
if (!optionalEnvelope(output.task_budget, TASK_BUDGET_FIELDS)
|
|
203
|
+
|| (output.task_budget != null && output.task_budget.type !== 'tokens')) return incompatible('task_budget');
|
|
204
|
+
if (output.effort != null && !targetFacts.effortLevels.includes(output.effort)) return incompatible('effort');
|
|
205
|
+
if (output.task_budget != null && !targetFacts.taskBudget) return incompatible('task_budget');
|
|
48
206
|
}
|
|
49
|
-
//
|
|
50
|
-
//
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
207
|
+
// Modern models reject non-default sampling. Shared modern source requests
|
|
208
|
+
// retain their own validation; upgrading an older model must not introduce
|
|
209
|
+
// this new restriction. Do not guess an undocumented numeric top_k default.
|
|
210
|
+
if (targetFacts.defaultSamplingOnly && !sourceFacts.defaultSamplingOnly
|
|
211
|
+
&& ((body.temperature !== undefined && body.temperature !== 1)
|
|
212
|
+
|| (body.top_p !== undefined && body.top_p !== 1) || body.top_k !== undefined)) return incompatible('sampling');
|
|
213
|
+
return compatible;
|
|
54
214
|
}
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
export const DECISION_RESPONSE_LIMIT = 64 * 1024;
|
|
2
|
+
export const MODEL_METADATA_LIMIT = 1024 * 1024;
|
|
3
|
+
|
|
4
|
+
// Cancellation is best effort and must never await an uncooperative body's
|
|
5
|
+
// cancel promise. Fetch's signal remains responsible for its network request.
|
|
6
|
+
export function cancelResponseBody(response, reason) {
|
|
7
|
+
try { Promise.resolve(response.body?.cancel(reason)).catch(() => {}); } catch {}
|
|
8
|
+
}
|
|
9
|
+
|
|
10
|
+
/**
|
|
11
|
+
* Read at most limit bytes, observing cancellation throughout body delivery.
|
|
12
|
+
* @param {Response} response
|
|
13
|
+
* @param {{signal?:AbortSignal,limit?:number}} [options]
|
|
14
|
+
*/
|
|
15
|
+
export async function readBoundedJson(response, { signal, limit = DECISION_RESPONSE_LIMIT } = {}) {
|
|
16
|
+
if (!Number.isSafeInteger(limit) || limit < 1) throw new Error('classifier_invalid_response');
|
|
17
|
+
let reader;
|
|
18
|
+
let bytes;
|
|
19
|
+
let total = 0;
|
|
20
|
+
const cancel = () => {
|
|
21
|
+
try { Promise.resolve(reader?.cancel(signal?.reason)).catch(() => {}); } catch {}
|
|
22
|
+
};
|
|
23
|
+
try {
|
|
24
|
+
signal?.throwIfAborted();
|
|
25
|
+
reader = response.body?.getReader();
|
|
26
|
+
if (!reader) throw new Error('classifier_invalid_response');
|
|
27
|
+
signal?.addEventListener('abort', cancel, { once: true });
|
|
28
|
+
// Header sizes are only an early rejection. The streamed byte count is
|
|
29
|
+
// authoritative for chunked, compressed, absent or inaccurate lengths.
|
|
30
|
+
const length = response.headers?.get('content-length');
|
|
31
|
+
if (length && /^\d+$/.test(length) && Number(length) > limit) throw new Error('classifier_invalid_response');
|
|
32
|
+
while (true) {
|
|
33
|
+
signal?.throwIfAborted();
|
|
34
|
+
const { value, done } = await reader.read();
|
|
35
|
+
signal?.throwIfAborted();
|
|
36
|
+
if (done) break;
|
|
37
|
+
if (!(value instanceof Uint8Array) || value.byteLength > limit - total) throw new Error('classifier_invalid_response');
|
|
38
|
+
const next = total + value.byteLength;
|
|
39
|
+
// A single growable buffer also bounds metadata: a malicious one-byte
|
|
40
|
+
// chunk stream must not retain tens of thousands of Buffer objects.
|
|
41
|
+
if (!bytes || next > bytes.length) {
|
|
42
|
+
const capacity = Math.min(limit, Math.max(next, bytes ? bytes.length * 2 : Math.min(1024, limit)));
|
|
43
|
+
const grown = Buffer.allocUnsafe(capacity);
|
|
44
|
+
bytes?.copy(grown, 0, 0, total);
|
|
45
|
+
bytes = grown;
|
|
46
|
+
}
|
|
47
|
+
bytes.set(value, total);
|
|
48
|
+
total = next;
|
|
49
|
+
}
|
|
50
|
+
try { return JSON.parse(bytes?.toString('utf8', 0, total) ?? ''); }
|
|
51
|
+
catch { throw new Error('classifier_invalid_response'); }
|
|
52
|
+
} finally {
|
|
53
|
+
signal?.removeEventListener('abort', cancel);
|
|
54
|
+
if (reader) { cancel(); try { reader.releaseLock(); } catch {} }
|
|
55
|
+
else cancelResponseBody(response, signal?.reason);
|
|
56
|
+
}
|
|
57
|
+
}
|
package/src/cli-help.mjs
ADDED
|
@@ -0,0 +1,90 @@
|
|
|
1
|
+
const commands = {
|
|
2
|
+
setup: `Usage: claude-autorouter setup [options]
|
|
3
|
+
|
|
4
|
+
Configure the local Ollama evaluator (default) or TypeSafe Jev.
|
|
5
|
+
--auth-mode subscription|api-key Default: subscription
|
|
6
|
+
--client-profile compatible|native|auto
|
|
7
|
+
--evaluator ollama|jev Default: ollama (local); jev sends excerpts to TypeSafe
|
|
8
|
+
--ollama-model TAG Select a /v1/systemone model
|
|
9
|
+
--ollama-timeout-ms N 0 disables the routing deadline
|
|
10
|
+
--pull Download the selected missing Ollama model
|
|
11
|
+
--stop-hook-block-cap N Optional Claude Stop-hook retry limit
|
|
12
|
+
--session-log-dir DIR Opt in to private logs with prompt excerpts
|
|
13
|
+
--session-log-mode metadata|prompts Choose whether excerpts are included
|
|
14
|
+
--secret-store file|keychain default on macOS for new setups; file is plaintext
|
|
15
|
+
--force Update an existing configuration
|
|
16
|
+
--replace Explicitly rebuild the saved configuration
|
|
17
|
+
|
|
18
|
+
First setup reads environment settings and keys, or prompts for missing keys.
|
|
19
|
+
--force retains saved defaults and applies explicit options; unrelated runtime
|
|
20
|
+
overrides stay temporary. Explicit evaluator/auth selection accepts its supplied key.
|
|
21
|
+
Examples: claude-autorouter setup --pull
|
|
22
|
+
claude-autorouter setup --evaluator jev`,
|
|
23
|
+
doctor: `Usage: claude-autorouter doctor [--evaluate-local] [--json]
|
|
24
|
+
|
|
25
|
+
Check configuration, installed Claude, and local model availability.
|
|
26
|
+
--evaluate-local explicitly tests synthetic routing cases on the existing
|
|
27
|
+
Ollama model. No downloads, Anthropic/Jev calls, or configuration changes.
|
|
28
|
+
--json returns the local evaluation report (requires --evaluate-local).
|
|
29
|
+
|
|
30
|
+
Example: claude-autorouter doctor --evaluate-local`,
|
|
31
|
+
config: `Usage: claude-autorouter config show [--json] [--check-all]
|
|
32
|
+
claude-autorouter config set KEY VALUE
|
|
33
|
+
claude-autorouter config set KEY --stdin
|
|
34
|
+
claude-autorouter config unset KEY
|
|
35
|
+
|
|
36
|
+
Show effective settings and their source; secrets are always redacted.
|
|
37
|
+
--check-all also validates settings for the inactive evaluator.
|
|
38
|
+
Set/unset changes only the named saved setting. Environment values still win.
|
|
39
|
+
Secret keys require --stdin or a hidden prompt, never a command-line value.
|
|
40
|
+
Setting AUTOROUTER_SECRET_STORE to keychain or file moves saved keys (macOS).
|
|
41
|
+
|
|
42
|
+
Example: claude-autorouter config set AUTOROUTER_OLLAMA_TIMEOUT_MS 0`,
|
|
43
|
+
serve: `Usage: claude-autorouter serve
|
|
44
|
+
|
|
45
|
+
Run a local Messages API gateway. Requires configured evaluator credentials,
|
|
46
|
+
authentication, and AUTOROUTER_TOKEN (at least 16 characters).
|
|
47
|
+
Use claude-autorouter claude for managed startup and cleanup.`,
|
|
48
|
+
sessions: `Usage: claude-autorouter sessions list [--json]
|
|
49
|
+
claude-autorouter sessions show ID [--json]
|
|
50
|
+
|
|
51
|
+
Read optional local session logs from AUTOROUTER_SESSION_LOG_DIR.
|
|
52
|
+
List prints the IDs used by show. History includes routing choices, observed
|
|
53
|
+
models, request outcomes, latency, and API-equivalent savings coverage.
|
|
54
|
+
Logging is off by default. To enable metadata-only history:
|
|
55
|
+
claude-autorouter config set AUTOROUTER_SESSION_LOG_MODE metadata
|
|
56
|
+
claude-autorouter config set AUTOROUTER_SESSION_LOG_DIR /path/to/private/logs`,
|
|
57
|
+
claude: `Usage: claude-autorouter claude [Claude Code arguments]
|
|
58
|
+
|
|
59
|
+
Launch Claude with automatic routing and an AutoRouter status line.
|
|
60
|
+
Example: claude-autorouter claude --permission-mode auto
|
|
61
|
+
Auto mode switches between Sonnet and Opus on new human tasks.
|
|
62
|
+
Claude owns permission checks and subscription authentication.
|
|
63
|
+
--help and --version pass directly to Claude without starting the router.`,
|
|
64
|
+
};
|
|
65
|
+
|
|
66
|
+
export function helpText(command = 'help') {
|
|
67
|
+
return commands[command] ?? `AutoRouter — automatic model routing for Claude Code
|
|
68
|
+
|
|
69
|
+
Usage: claude-autorouter <command> [options]
|
|
70
|
+
|
|
71
|
+
setup Configure Jev or local Ollama; prompt for keys privately
|
|
72
|
+
doctor Check configuration and local dependencies
|
|
73
|
+
config Inspect or change saved settings without replacing them
|
|
74
|
+
sessions Read optional local decision and outcome history
|
|
75
|
+
claude Launch Claude Code with automatic routing
|
|
76
|
+
serve Run the local gateway separately
|
|
77
|
+
--version Print the AutoRouter version
|
|
78
|
+
|
|
79
|
+
Start: claude-autorouter setup
|
|
80
|
+
claude-autorouter doctor
|
|
81
|
+
claude-autorouter claude --permission-mode auto
|
|
82
|
+
|
|
83
|
+
Run claude-autorouter help <command> for options and examples.
|
|
84
|
+
Local Ollama is the default evaluator and uses only the local /v1/systemone
|
|
85
|
+
endpoint. Jev is optional; it receives bounded, redacted prompt excerpts. Complete requests go to Anthropic.
|
|
86
|
+
Session logging is off unless AUTOROUTER_SESSION_LOG_DIR is configured.
|
|
87
|
+
Environment variables override ~/.config/claude-autorouter/config.json.
|
|
88
|
+
AUTOROUTER_CONFIG selects another file. Project .env files are not auto-loaded.
|
|
89
|
+
Reference: docs/reference.md`;
|
|
90
|
+
}
|
|
@@ -0,0 +1,158 @@
|
|
|
1
|
+
import { readConfig, parseSessionLogDir, parseStopHookBlockCap } from './config.mjs';
|
|
2
|
+
import { CONFIG_KEYS, SECRET_CONFIG_KEYS, SECRET_STORES, keychainRemovals, loadUserConfig, saveUserConfig } from './user-config.mjs';
|
|
3
|
+
import { modelCapabilities } from './model-catalog.mjs';
|
|
4
|
+
import { askSecret } from './onboarding.mjs';
|
|
5
|
+
|
|
6
|
+
const secretKey = key => SECRET_CONFIG_KEYS.includes(key);
|
|
7
|
+
const providerFor = key => key.startsWith('AUTOROUTER_OLLAMA_') ? 'ollama'
|
|
8
|
+
: key.startsWith('AUTOROUTER_JEV_') || key === 'AUTOROUTER_MIN_CONFIDENCE' || key === 'TYPESAFE_API_KEY' ? 'jev' : undefined;
|
|
9
|
+
const safeText = value => String(value).replace(/[\u0000-\u001f\u007f-\u009f\u202a-\u202e\u2066-\u2069]/g, '');
|
|
10
|
+
|
|
11
|
+
function effectiveValues(config, env) {
|
|
12
|
+
return {
|
|
13
|
+
AUTOROUTER_AUTH_MODE: config.authMode, AUTOROUTER_CLIENT_PROFILE: config.clientProfile,
|
|
14
|
+
AUTOROUTER_EVALUATOR: config.evaluator, AUTOROUTER_UPSTREAM_URL: config.upstream,
|
|
15
|
+
AUTOROUTER_HAIKU_MODEL: config.models.haiku, AUTOROUTER_SONNET_MODEL: config.models.sonnet,
|
|
16
|
+
AUTOROUTER_OPUS_MODEL: config.models.opus, AUTOROUTER_PORT: config.port,
|
|
17
|
+
AUTOROUTER_TOKEN_COUNT_TIMEOUT_MS: config.tokenCountTimeoutMs,
|
|
18
|
+
AUTOROUTER_JEV_URL: config.jevEndpoint, AUTOROUTER_JEV_MODEL: config.jevModel,
|
|
19
|
+
AUTOROUTER_JEV_TIMEOUT_MS: config.jevTimeoutMs, AUTOROUTER_MIN_CONFIDENCE: config.minConfidence,
|
|
20
|
+
AUTOROUTER_OLLAMA_URL: config.ollamaEndpoint, AUTOROUTER_OLLAMA_MODEL: config.ollamaModel,
|
|
21
|
+
AUTOROUTER_OLLAMA_TIMEOUT_MS: config.ollamaTimeoutMs, AUTOROUTER_OLLAMA_KEEP_ALIVE: config.ollamaKeepAlive,
|
|
22
|
+
AUTOROUTER_SESSION_LOG_DIR: config.sessionLogDir ?? null,
|
|
23
|
+
AUTOROUTER_SESSION_LOG_MODE: config.sessionLogMode,
|
|
24
|
+
CLAUDE_CODE_STOP_HOOK_BLOCK_CAP: config.stopHookBlockCap ?? null,
|
|
25
|
+
AUTOROUTER_STATUSLINE: env.AUTOROUTER_STATUSLINE !== '0',
|
|
26
|
+
AUTOROUTER_DEBUG: env.AUTOROUTER_DEBUG === '1', ENABLE_TOOL_SEARCH: env.ENABLE_TOOL_SEARCH ?? 'true',
|
|
27
|
+
};
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
/** Redacted effective configuration, independent of Claude launch arguments. */
|
|
31
|
+
export function configReport(loaded, env, { checkAll = false } = {}) {
|
|
32
|
+
const config = readConfig(loaded.env);
|
|
33
|
+
const values = { ...effectiveValues(config, loaded.env), AUTOROUTER_SECRET_STORE: loaded.secretStore ?? 'file' };
|
|
34
|
+
const settings = {};
|
|
35
|
+
for (const key of CONFIG_KEYS) {
|
|
36
|
+
const saved = loaded.keychainSecrets?.includes(key) ? 'keychain' : 'file';
|
|
37
|
+
// The saved store setting governs saved secrets; the environment cannot redirect it.
|
|
38
|
+
const source = env[key] !== undefined && key !== 'AUTOROUTER_SECRET_STORE' ? 'environment'
|
|
39
|
+
: Object.hasOwn(loaded.values, key) ? saved : 'default';
|
|
40
|
+
const provider = providerFor(key);
|
|
41
|
+
const active = (!provider || provider === config.evaluator)
|
|
42
|
+
&& (key !== 'ANTHROPIC_API_KEY' || config.authMode === 'api-key');
|
|
43
|
+
const secret = secretKey(key);
|
|
44
|
+
settings[key] = { source, active,
|
|
45
|
+
...(source === 'environment' && Object.hasOwn(loaded.values, key) ? { overrides_file: true } : {}),
|
|
46
|
+
...(source === 'environment' && loaded.unavailableSecrets?.includes(key) ? { keychain_unavailable: true } : {}),
|
|
47
|
+
...(secret ? { secret: true, present: typeof loaded.env[key] === 'string' && Boolean(loaded.env[key].trim()) }
|
|
48
|
+
: { value: active ? values[key] ?? null : null }),
|
|
49
|
+
};
|
|
50
|
+
}
|
|
51
|
+
const warnings = [];
|
|
52
|
+
if (config.clientProfile === 'auto' && [config.models.sonnet, config.models.opus].some(model => !modelCapabilities(model)?.sharedAuto)) {
|
|
53
|
+
warnings.push('Auto permission eligibility does not guarantee automatic switching. Older targets can retain the incoming model for shared execution features.');
|
|
54
|
+
}
|
|
55
|
+
let valid = true, error;
|
|
56
|
+
if (checkAll) {
|
|
57
|
+
try { readConfig(loaded.env, { validateAll: true }); }
|
|
58
|
+
catch (failure) { valid = false; error = failure.message; }
|
|
59
|
+
}
|
|
60
|
+
return { schema_version: 1, config_path: loaded.path, config_exists: loaded.exists, valid,
|
|
61
|
+
checked: checkAll ? 'all_evaluators' : 'active_evaluator', settings, warnings,
|
|
62
|
+
...(error ? { error } : {}),
|
|
63
|
+
};
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
async function stdinSecret(input) {
|
|
67
|
+
if (input.isTTY) throw new Error('Pipe the secret to --stdin, or omit --stdin to use the hidden prompt.');
|
|
68
|
+
const chunks = [];
|
|
69
|
+
let bytes = 0;
|
|
70
|
+
try {
|
|
71
|
+
for await (const chunk of input) {
|
|
72
|
+
const buffer = Buffer.isBuffer(chunk) ? chunk : Buffer.from(chunk);
|
|
73
|
+
bytes += buffer.length;
|
|
74
|
+
if (bytes > 16384) throw new Error('Secret input exceeds the supported size.');
|
|
75
|
+
chunks.push(buffer);
|
|
76
|
+
}
|
|
77
|
+
} catch { throw new Error('Could not read a bounded secret from stdin.'); }
|
|
78
|
+
return Buffer.concat(chunks).toString('utf8').replace(/\r?\n$/, '');
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
function normalizedValue(key, value) {
|
|
82
|
+
if (typeof value !== 'string') throw new Error('Configuration values must be strings.');
|
|
83
|
+
if (secretKey(key)) {
|
|
84
|
+
const normalized = value.trim();
|
|
85
|
+
if (!normalized || /[\r\n\0]/.test(normalized)) throw new Error(`${key} must be a nonempty, single-line secret.`);
|
|
86
|
+
if (key === 'AUTOROUTER_TOKEN' && normalized.length < 16) throw new Error('AUTOROUTER_TOKEN must contain at least 16 characters.');
|
|
87
|
+
return normalized;
|
|
88
|
+
}
|
|
89
|
+
if (key === 'AUTOROUTER_SESSION_LOG_DIR') return parseSessionLogDir(value) ?? '';
|
|
90
|
+
if (key === 'CLAUDE_CODE_STOP_HOOK_BLOCK_CAP') return String(parseStopHookBlockCap(value));
|
|
91
|
+
if (key === 'AUTOROUTER_SECRET_STORE' && !SECRET_STORES.includes(value)) throw new Error('AUTOROUTER_SECRET_STORE must be file or keychain.');
|
|
92
|
+
if (/[\r\n\0\u001b]/.test(value)) throw new Error(`${key} must be a single-line setting.`);
|
|
93
|
+
return value;
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
export async function configCommand(args, {
|
|
97
|
+
env = process.env, write = console.log, input = process.stdin, promptSecret = askSecret, keychain,
|
|
98
|
+
} = {}) {
|
|
99
|
+
const store = keychain ? { keychain } : {};
|
|
100
|
+
const [operation, ...rest] = args;
|
|
101
|
+
if (operation === 'show') {
|
|
102
|
+
if (rest.some(arg => !['--json', '--check-all'].includes(arg))) throw new Error('Usage: claude-autorouter config show [--json] [--check-all]');
|
|
103
|
+
let loaded, report;
|
|
104
|
+
try {
|
|
105
|
+
loaded = loadUserConfig(env, { allowMissing: true, ...store });
|
|
106
|
+
report = configReport(loaded, env, { checkAll: rest.includes('--check-all') });
|
|
107
|
+
}
|
|
108
|
+
catch (error) {
|
|
109
|
+
report = { schema_version: 1, ...(loaded ? { config_path: loaded.path, config_exists: loaded.exists } : {}), valid: false, error: error.message };
|
|
110
|
+
}
|
|
111
|
+
if (rest.includes('--json')) write(JSON.stringify(report, null, 2));
|
|
112
|
+
else {
|
|
113
|
+
if (loaded) write(`Config: ${loaded.path}${loaded.exists ? '' : ' (not saved)'}`);
|
|
114
|
+
for (const [key, setting] of Object.entries(report.settings ?? {})) {
|
|
115
|
+
const value = setting.secret ? setting.present ? '[set; hidden]' : '[unset]'
|
|
116
|
+
: !setting.active ? '[inactive]' : setting.value === null ? '[unset]' : safeText(setting.value);
|
|
117
|
+
write(`${key}=${value} [${setting.source}${setting.overrides_file ? '; overrides file' : ''}${!setting.active ? '; inactive' : ''}]`);
|
|
118
|
+
}
|
|
119
|
+
for (const warning of report.warnings ?? []) write(warning);
|
|
120
|
+
if (report.error) write(`FAIL ${report.error}`);
|
|
121
|
+
}
|
|
122
|
+
return report.valid;
|
|
123
|
+
}
|
|
124
|
+
if (!['set', 'unset'].includes(operation)) throw new Error('Usage: claude-autorouter config show|set|unset');
|
|
125
|
+
const [key, value, ...extra] = rest;
|
|
126
|
+
// Never echo an unknown key: it may itself be a pasted credential.
|
|
127
|
+
if (!CONFIG_KEYS.includes(key)) throw new Error('Unsupported configuration key. Run claude-autorouter config show for supported settings.');
|
|
128
|
+
if (extra.length || (operation === 'unset' && value !== undefined)) throw new Error(`Usage: claude-autorouter config ${operation} KEY${operation === 'set' ? ' VALUE' : ''}`);
|
|
129
|
+
if (operation === 'set' && secretKey(key) && value !== undefined && value !== '--stdin') {
|
|
130
|
+
throw new Error('Secret values are not accepted as command arguments. Use --stdin or the hidden prompt.');
|
|
131
|
+
}
|
|
132
|
+
if (operation === 'set' && !secretKey(key) && (value === undefined || value === '--stdin')) throw new Error('Nonsecret settings require a value argument.');
|
|
133
|
+
const loaded = loadUserConfig(env, { allowMissing: true, ...store });
|
|
134
|
+
const next = { ...loaded.values };
|
|
135
|
+
if (operation === 'unset') delete next[key];
|
|
136
|
+
else next[key] = normalizedValue(key, secretKey(key)
|
|
137
|
+
? value === '--stdin' ? await stdinSecret(input) : await promptSecret(key)
|
|
138
|
+
: value);
|
|
139
|
+
// Validate the persisted setting itself, not an environment value that could
|
|
140
|
+
// mask it. An explicitly edited inactive provider is checked too.
|
|
141
|
+
const provider = operation === 'set' ? providerFor(key) : undefined;
|
|
142
|
+
readConfig({ ...next, ...(provider ? { AUTOROUTER_EVALUATOR: provider } : {}) });
|
|
143
|
+
const nextStore = next.AUTOROUTER_SECRET_STORE ?? 'file';
|
|
144
|
+
// Changing the store moves saved secrets; unsetting a secret deletes its item.
|
|
145
|
+
const removeSecrets = key === 'AUTOROUTER_SECRET_STORE' ? keychainRemovals(loaded, nextStore)
|
|
146
|
+
: operation === 'unset' && secretKey(key) && loaded.secretStore === 'keychain' ? [key] : [];
|
|
147
|
+
saveUserConfig(next, { env, overwrite: loaded.exists, expectedRevision: loaded.revision, removeSecrets, ...store });
|
|
148
|
+
if (key === 'AUTOROUTER_SECRET_STORE') {
|
|
149
|
+
const moved = SECRET_CONFIG_KEYS.filter(name => Object.hasOwn(next, name)).length;
|
|
150
|
+
const where = nextStore === 'keychain' ? 'macOS Keychain' : 'configuration file';
|
|
151
|
+
write(nextStore === loaded.secretStore || !moved ? `Saved ${key}. Saved secrets are stored in the ${where}.`
|
|
152
|
+
: `Saved ${key}. Moved ${moved} saved secret${moved === 1 ? '' : 's'} to the ${where}.`);
|
|
153
|
+
return true;
|
|
154
|
+
}
|
|
155
|
+
write(`${operation === 'unset' ? 'Removed saved' : 'Saved'} ${key}${secretKey(key) && nextStore === 'keychain' ? ' in the macOS Keychain' : ''}. Other saved settings are unchanged.`);
|
|
156
|
+
if (env[key] !== undefined) write(`The current environment still overrides ${key}; unset that environment variable to use the saved/default value.`);
|
|
157
|
+
return true;
|
|
158
|
+
}
|