claude-autorouter 0.3.6 → 0.3.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +8 -5
- package/README.md +3 -3
- package/bin/autorouter.mjs +1 -0
- package/docs/development.md +2 -0
- package/docs/reference.md +12 -8
- package/docs/releasing.md +2 -0
- package/package.json +1 -1
- package/src/auto-routing.mjs +54 -0
- package/src/config.mjs +1 -1
- package/src/model-request.mjs +5 -0
- package/src/router.mjs +27 -13
- package/src/statusline.mjs +1 -1
package/.env.example
CHANGED
|
@@ -6,7 +6,8 @@ AUTOROUTER_EVALUATOR=jev
|
|
|
6
6
|
# The launcher enables the router status line for this session. Set 0 to keep your own.
|
|
7
7
|
AUTOROUTER_STATUSLINE=1
|
|
8
8
|
# Claude's Auto permission mode needs a supported Sonnet or Opus client.
|
|
9
|
-
# This profile
|
|
9
|
+
# This profile switches between Sonnet 5.5 and Opus 5.5 on new human tasks.
|
|
10
|
+
# Haiku decisions become Sonnet; tool continuations keep the selected model.
|
|
10
11
|
# AutoRouter also selects it for: claude-autorouter claude --permission-mode auto
|
|
11
12
|
# AUTOROUTER_CLIENT_PROFILE=auto
|
|
12
13
|
# Optional metadata logs on stderr. Redirect stderr to a file when using the UI.
|
|
@@ -24,10 +25,12 @@ TYPESAFE_API_KEY=
|
|
|
24
25
|
# For API billing instead, set AUTOROUTER_AUTH_MODE=api-key and fill this in.
|
|
25
26
|
# ANTHROPIC_API_KEY=
|
|
26
27
|
|
|
27
|
-
#
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
28
|
+
# Optional overrides; leaving these unset uses each profile's defaults.
|
|
29
|
+
# Auto defaults to Sonnet 5.5; compatible/native default to Sonnet 5.
|
|
30
|
+
# An explicit Sonnet override also applies in Auto mode.
|
|
31
|
+
# AUTOROUTER_HAIKU_MODEL=claude-haiku-4-5-20251001
|
|
32
|
+
# AUTOROUTER_SONNET_MODEL=claude-sonnet-5-5
|
|
33
|
+
# AUTOROUTER_OPUS_MODEL=claude-opus-5-5
|
|
31
34
|
AUTOROUTER_JEV_MODEL=jev-latest
|
|
32
35
|
AUTOROUTER_JEV_TIMEOUT_MS=1500
|
|
33
36
|
# Only suspicious context sizes need counting; this overlaps the evaluator.
|
package/README.md
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# Claude AutoRouter
|
|
2
2
|
|
|
3
|
-
Use Haiku, Sonnet, and Opus in one Claude Code session. A local gateway classifies coding requests with the selected evaluator, applies compatibility and context checks, and streams the selected model's response back to Claude Code. Claude's internal classifiers
|
|
3
|
+
Use Haiku, Sonnet, and Opus in one Claude Code session. A local gateway classifies coding requests with the selected evaluator, applies compatibility and context checks, and streams the selected model's response back to Claude Code. Claude's internal permission classifiers retain their selected model; execution requests keep their server safety-review settings and verdicts when routed. [TypeSafe Jev](https://typesafe.ai/blog/introducing-system-one-models-and-jev) is the default; an experimental Ollama backend evaluates requests locally.
|
|
4
4
|
|
|
5
5
|
Requires Node.js 22+, macOS or Linux (including WSL), an installed `claude` command, and a Claude subscription login or Anthropic API key. The default evaluator also requires a [TypeSafe API key](https://console.typesafe.ai). There are no runtime package dependencies. Native Windows is not supported in this release.
|
|
6
6
|
|
|
@@ -35,13 +35,13 @@ claude-autorouter --version
|
|
|
35
35
|
|
|
36
36
|
For API billing, use `claude-autorouter setup --auth-mode api-key`. Use `--force` to replace an existing config. Automation can supply `TYPESAFE_API_KEY` and, in API-key mode, `ANTHROPIC_API_KEY` through the environment; keys are never command-line arguments. `doctor` checks local configuration and Claude installation/login state without paid requests. See the [configuration reference](docs/reference.md#configuration).
|
|
37
37
|
|
|
38
|
-
To use Claude's Auto permission mode (AutoRouter 0.3.
|
|
38
|
+
To use automatic Sonnet/Opus routing with Claude's Auto permission mode (AutoRouter 0.3.7+):
|
|
39
39
|
|
|
40
40
|
```sh
|
|
41
41
|
claude-autorouter claude --permission-mode auto
|
|
42
42
|
```
|
|
43
43
|
|
|
44
|
-
This selects the Auto
|
|
44
|
+
This selects the Auto profile, defaulting to Sonnet 5.5 and Opus 5.5. The evaluator can choose again for each new human task, while a task's tool calls and goal continuations retain its selected model. A Haiku verdict uses Sonnet. Native safety review stays enabled; organization policies still apply. Use `AUTOROUTER_CLIENT_PROFILE=auto` for sessions where you select Auto in Claude's UI or saved settings. Version 0.3.6 enabled Auto permissions but passed server-reviewed execution through without routing; version 0.3.7 removes that restriction for compatible requests. [Auto-mode support and limitations](docs/reference.md#auto-permission-mode).
|
|
45
45
|
|
|
46
46
|
## What you see
|
|
47
47
|
|
package/bin/autorouter.mjs
CHANGED
|
@@ -52,6 +52,7 @@ Without setup, AUTOROUTER_AUTH_MODE defaults to api-key and also requires ANTHRO
|
|
|
52
52
|
AUTOROUTER_CLIENT_PROFILE=compatible (default) enables all three routing tiers.
|
|
53
53
|
Use AUTOROUTER_CLIENT_PROFILE=native to retain Claude Code's own model/thinking settings.
|
|
54
54
|
Use AUTOROUTER_CLIENT_PROFILE=auto for Auto permission mode: Sonnet/Opus routing, native thinking.
|
|
55
|
+
Auto defaults to Sonnet 5.5 and Opus 5.5, switching on new human tasks and retaining tool turns.
|
|
55
56
|
An explicit claude --permission-mode auto selects the auto profile for that launch.
|
|
56
57
|
Claude's permission checks and organization policies still apply; Haiku does not support Auto mode.
|
|
57
58
|
Optional CLAUDE_CODE_STOP_HOOK_BLOCK_CAP=N limits consecutive tool-free Stop-hook continuations.
|
package/docs/development.md
CHANGED
|
@@ -12,6 +12,8 @@ npm run test:package
|
|
|
12
12
|
|
|
13
13
|
The test suite uses local mocks and fake credentials. It covers Jev and Ollama routing, bounded prompt extraction, confidence and timeout fallback, token checks, model continuity, authentication forwarding, streaming, cancellation, status state, savings, and launcher behavior. Tests that start HTTP services require loopback binding. Local tests make no paid provider calls or model downloads.
|
|
14
14
|
|
|
15
|
+
Auto-mode regressions exercise Sonnet → Opus → Opus tool continuation → Sonnet in one conversation, with and without gateway prompt IDs. They retain signed thinking, native context edits, mid-conversation system messages, safety-review settings, and streamed verdicts, including denied actions. Separate capability tests keep unknown review contracts and incompatible model features from being routed.
|
|
16
|
+
|
|
15
17
|
Package validation checks the distributable and installed command rather than relying on the source checkout's paths. Review the [release procedure](releasing.md) before distributing a tarball.
|
|
16
18
|
|
|
17
19
|
## Run from source
|
package/docs/reference.md
CHANGED
|
@@ -52,7 +52,7 @@ For an environment-only subscription launch, set `AUTOROUTER_AUTH_MODE=subscript
|
|
|
52
52
|
| `CLAUDE_CODE_STOP_HOOK_BLOCK_CAP` | unset; Claude currently uses `8` | Optional cap on consecutive Stop/SubagentStop continuations without tool use; `0` disables the cap |
|
|
53
53
|
| `ENABLE_TOOL_SEARCH` | `true` in launcher when unset | Load MCP tool definitions on demand; explicit values are preserved |
|
|
54
54
|
| `AUTOROUTER_HAIKU_MODEL` | `claude-haiku-4-5-20251001` | Routine tier |
|
|
55
|
-
| `AUTOROUTER_SONNET_MODEL` | `claude-sonnet-5` | Standard tier |
|
|
55
|
+
| `AUTOROUTER_SONNET_MODEL` | `claude-sonnet-5`; `claude-sonnet-5-5` in the `auto` profile | Standard tier |
|
|
56
56
|
| `AUTOROUTER_OPUS_MODEL` | `claude-opus-5-5` | Demanding tier and savings baseline |
|
|
57
57
|
| `AUTOROUTER_JEV_MODEL` | `jev-latest` | Classifier version |
|
|
58
58
|
| `AUTOROUTER_JEV_TIMEOUT_MS` | `1500` | Classifier deadline in milliseconds |
|
|
@@ -173,13 +173,13 @@ The following policy applies after classification:
|
|
|
173
173
|
- Jev confidence below 0.75 prevents a downgrade below Sonnet or the requested tier. Ollama returns a tier without calibrated confidence; its failure handling and compatibility guards still apply.
|
|
174
174
|
- Tool continuations retain the model chosen at the start of the human turn. Session, agent, and prompt headers identify turns; normalized conversation content provides a fallback. Text feedback from a Stop hook also retains the model when it serves the same gateway prompt ID and the client has not changed its requested model, subject to capability and context checks. Moving prompt-cache markers does not create a new turn.
|
|
175
175
|
- Claude's local `/goal` command can omit the prompt-ID header. For that path, an exact feedback label matching a preceding expanded `/goal` command keeps the original task and conversation anchor. This narrow text fallback also recognizes Claude's repeated-goal truncation format; arbitrary hook text is not treated as a goal. Feedback remains in the evaluator's recent conversation and the full API request. A new human message becomes the current task normally. The status line shows `prompt pinned` or `goal pinned` when either text-continuation rule applies.
|
|
176
|
-
- Thinking history, fixed-budget thinking, server tools, context management, and other recognized model-specific features preserve the current model. Adaptive thinking, effort, and output above 64K prevent a Haiku choice. Fields are never stripped to force a downgrade.
|
|
177
|
-
- Mid-conversation `system` messages preserve the requested model
|
|
178
|
-
- Auxiliary requests, including Claude's Auto permission classifier, pass through on their requested model without Jev/Ollama evaluation, token checks, or turn-state changes. Compaction retains its existing model and context-capacity policy.
|
|
176
|
+
- Thinking history, fixed-budget thinking, server tools, context management, and other recognized model-specific features preserve the current model except for the verified shared capabilities of the modern Auto-mode Sonnet/Opus pair described below. Adaptive thinking, effort, and output above 64K prevent a Haiku choice. Fields are never stripped to force a downgrade.
|
|
177
|
+
- Mid-conversation `system` messages preserve the requested model unless both Auto-mode models support them; they always pass through unchanged. They do not count as a tool continuation by themselves.
|
|
178
|
+
- Auxiliary requests, including Claude's Auto permission classifier, pass through on their requested model without Jev/Ollama evaluation, token checks, or turn-state changes. Compaction retains its existing model and context-capacity policy. Recognized server-reviewed execution requests can route between compatible Sonnet/Opus models while retaining `safeguards` and all verdicts unchanged. Unknown safeguards contracts pass through. Token counting and model discovery pass through without classification.
|
|
179
179
|
|
|
180
180
|
The default `compatible` profile starts Claude with Haiku-compatible requests and client-requested thinking disabled. AutoRouter uses adaptive thinking when upgrading these requests to Opus 5/5.5. Starting with 0.3.3, routing to exact `claude-sonnet-5-5` translates disabled thinking to `between_tools`, which skips up-front thinking but permits progress updates between tool calls. At `xhigh`/`max` effort, or when per-message effort differs from the top-level setting (default `high`), it uses adaptive thinking while preserving the effort settings. Token counting uses the same adaptation. Sonnet 5 still accepts disabled thinking and is unchanged. See [Sonnet 5.5 thinking requirements](https://platform.claude.com/docs/en/models/sonnet-5-5/migration-guide).
|
|
181
181
|
|
|
182
|
-
Explicit native `between_tools` and unknown thinking modes retain the incoming model on new human turns. Signed thinking blocks pass through unchanged and existing tool turns retain their model pin. `AUTOROUTER_CLIENT_PROFILE=native` preserves normal client settings, which can constrain routing. An explicit Claude `--model` argument overrides the starting model, but `/model` and `--model` are requested models, not locks on the routed result. Native same-model requests and unknown model aliases are not rewritten; clients must use settings supported by that model.
|
|
182
|
+
Explicit native `between_tools` and unknown thinking modes retain the incoming model on new human turns outside Auto routing. In Auto routing, a known Sonnet 5.5 `between_tools` request can upgrade to Opus with adaptive thinking. Signed thinking blocks pass through unchanged and existing tool turns retain their model pin. `AUTOROUTER_CLIENT_PROFILE=native` preserves normal client settings, which can constrain routing. An explicit Claude `--model` argument overrides the starting model, but `/model` and `--model` are requested models, not locks on the routed result. Native same-model requests and unknown model aliases are not rewritten; clients must use settings supported by that model.
|
|
183
183
|
|
|
184
184
|
The launcher enables `ENABLE_TOOL_SEARCH=true` when unset. Claude can otherwise disable on-demand MCP discovery when using a custom API address, loading connected-tool schemas into even a fresh conversation. Explicit values, including `false` or `auto:5`, are preserved. Managed settings and always-loaded tools can still affect deferral. See [Claude Code tool search](https://code.claude.com/docs/en/mcp#configure-tool-search).
|
|
185
185
|
|
|
@@ -187,7 +187,7 @@ The launcher enables `ENABLE_TOOL_SEARCH=true` when unset. Claude can otherwise
|
|
|
187
187
|
|
|
188
188
|
The default `compatible` profile starts Claude as Haiku to permit three-tier routing. Claude's Auto permission mode does not support Haiku, even if AutoRouter routes an API request to Sonnet. Eligibility is based on Claude's selected client model. Gateways themselves are supported. See [Claude's Auto-mode requirements](https://code.claude.com/docs/en/permission-modes#eliminate-permission-prompts-with-auto-mode).
|
|
189
189
|
|
|
190
|
-
AutoRouter 0.3.6
|
|
190
|
+
AutoRouter 0.3.6 introduced an `auto` client profile but bypassed evaluation for server-reviewed execution. Version 0.3.7 adds automatic switching on those requests. Launch with:
|
|
191
191
|
|
|
192
192
|
```sh
|
|
193
193
|
claude-autorouter claude --permission-mode auto
|
|
@@ -199,11 +199,15 @@ An explicit `--permission-mode auto` (or `--permission-mode=auto`) selects the p
|
|
|
199
199
|
env AUTOROUTER_CLIENT_PROFILE=auto claude-autorouter claude
|
|
200
200
|
```
|
|
201
201
|
|
|
202
|
-
The profile defaults
|
|
202
|
+
The profile defaults to Sonnet 5.5 and Opus 5.5, preserving explicit configured model IDs. The evaluator chooses Sonnet or Opus for each new human task; a routine Haiku verdict uses Sonnet and shows `Auto mode floor`. Tool and `/goal` continuations stay on the selected execution model. Claude's initial client model remains separate from the routed model. An explicit client `--model` or `ANTHROPIC_MODEL` can still make Auto unavailable if it selects Haiku or another unsupported model; choose a supported Sonnet or Opus instead.
|
|
203
203
|
|
|
204
204
|
Claude remains responsible for enabling the permission mode and enforcing organization settings, account availability, and tool rules. The profile does not enable Auto by itself or override `disableAutoMode`. AutoRouter does not reproduce Claude's settings precedence to infer a mode from settings files. For new saved configurations, `setup --client-profile auto` persists the profile; for an existing config, change only `AUTOROUTER_CLIENT_PROFILE` to `"auto"` to retain your other settings. `setup --force` replaces the config.
|
|
205
205
|
|
|
206
|
-
**
|
|
206
|
+
**Safety review:** Claude's permission-classifier requests retain their exact requested model and skip AutoRouter's evaluator. Ordinary execution requests with the known `dangerous_tool_use` version-1 review contract are evaluated and routed, retaining the complete `safeguards` object, beta headers, and streamed safety verdicts. This also detects server review when Auto was selected in Claude's UI rather than through the launch flag. Unknown or malformed review contracts and safeguarded compaction pass through with `Auto safety`; a target that cannot accept the request shows `Auto model guard`. AutoRouter never turns off server review or converts denied actions to approvals. See [server-side classifier review](https://code.claude.com/docs/en/permission-modes#server-side-classifier-review).
|
|
207
|
+
|
|
208
|
+
**Shared execution capabilities:** automatic Auto routing supports exact Sonnet 5/5.5 and Opus 5/5.5 IDs. The default 5.5 pair shares a native 1M context window, adaptive thinking, native context-editing strategies, and mid-conversation system updates. These fields and existing signed thinking no longer pin every future human task. Sonnet 5 cannot accept mid-conversation system messages, per-message effort, or task budgets; use Sonnet 5.5 for those sessions. Unknown context-editing strategies, specialized server tools, fixed thinking budgets, fast mode, older/custom targets, and other incompatible requests still retain a compatible model. A Sonnet 5.5 `between_tools` request uses adaptive thinking when upgraded to Opus; effort and conversation history remain unchanged.
|
|
209
|
+
|
|
210
|
+
Thinking blocks stay verbatim in the conversation. Anthropic may drop blocks the selected model cannot read, so switching models does not preserve access to every model's private reasoning on every turn. User text, tool results, and prior answers remain available. Returning to a model can make its preserved thinking readable again. Model changes can also miss the prior model's prompt cache. See [preserved thinking and model switching](https://platform.claude.com/docs/en/build-with-claude/preserved-thinking).
|
|
207
211
|
|
|
208
212
|
### Context capacity
|
|
209
213
|
|
package/docs/releasing.md
CHANGED
|
@@ -14,6 +14,8 @@ Version `0.3.5` adds opt-in saved configuration for Claude's native `CLAUDE_CODE
|
|
|
14
14
|
|
|
15
15
|
Version `0.3.6` fixes Auto permission-mode launches with an Auto-compatible Sonnet/Opus profile while preserving Claude's permission classifiers and server safety-review requests. It also adds optional per-session JSONL decision logs containing a bounded human prompt excerpt, selected model, and routing latency. Logging is disabled by default; set `AUTOROUTER_SESSION_LOG_DIR` or use `setup --session-log-dir DIR` to enable it. See [Auto permission mode](reference.md#auto-permission-mode) and [session decision logs](reference.md#session-decision-logs).
|
|
16
16
|
|
|
17
|
+
Version `0.3.7` enables automatic Sonnet/Opus switching for compatible Auto-mode execution requests, including requests carrying the known server safety-review contract. The Auto profile defaults to Sonnet 5.5 and Opus 5.5, floors Haiku decisions to Sonnet, and retains the selected model through tool and goal continuations. Shared native context edits, mid-conversation system messages, and signed thinking history no longer pin new human tasks. Permission-classifier requests and safety verdicts remain unchanged; unknown contracts and incompatible model features still preserve a compatible model. Explicit model overrides remain in effect. See [Auto permission mode](reference.md#auto-permission-mode).
|
|
18
|
+
|
|
17
19
|
The GitHub repository is private. Publishing to npm makes the tarball's runtime source, README, configuration example, license, and shipped documentation public. Model weights, user configuration, credentials, transcripts, session logs, local artifacts, and test fixtures are excluded. Review the archive before the first publication and whenever the package allowlist changes.
|
|
18
20
|
|
|
19
21
|
## What runs automatically
|
package/package.json
CHANGED
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
// Shared execution capabilities, not a replacement for Claude's permission
|
|
2
|
+
// classifier. Keep the complete safeguards contract and signed history on wire.
|
|
3
|
+
const MODELS = new Set(['claude-sonnet-5', 'claude-sonnet-5-5', 'claude-opus-5', 'claude-opus-5-5']);
|
|
4
|
+
const object = value => value !== null && typeof value === 'object' && !Array.isArray(value);
|
|
5
|
+
const SHARED_TOOLS = new Set(['custom', 'tool_search_tool_regex_20251119', 'tool_search_tool_bm25_20251119',
|
|
6
|
+
'bash_20250124', 'text_editor_20250728']);
|
|
7
|
+
const CONTEXT_EDITS = new Set(['clear_thinking_20251015', 'clear_tool_uses_20250919']);
|
|
8
|
+
const sharedTool = tool => object(tool) && (tool.type === undefined || SHARED_TOOLS.has(tool.type));
|
|
9
|
+
|
|
10
|
+
export function hasRoutableSafeguards(body) {
|
|
11
|
+
return MODELS.has(body.model) && Array.isArray(body.safeguards) && body.safeguards.length > 0
|
|
12
|
+
&& body.safeguards.every(entry => object(entry) && entry.type === 'dangerous_tool_use'
|
|
13
|
+
&& object(entry.classifier_context) && entry.classifier_context.v === 1);
|
|
14
|
+
}
|
|
15
|
+
|
|
16
|
+
export function canRouteAutoRequest(body, target) {
|
|
17
|
+
if (!MODELS.has(body.model) || !MODELS.has(target)) return false;
|
|
18
|
+
if (body.safeguards !== undefined && !hasRoutableSafeguards(body)) return false;
|
|
19
|
+
if (body.max_tokens !== undefined && (!Number.isSafeInteger(body.max_tokens) || body.max_tokens < 1 || body.max_tokens > 128000)) return false;
|
|
20
|
+
// Retain model-specific execution facilities whose contracts differ between
|
|
21
|
+
// models. Ordinary Claude Code tools and native context editing are shared.
|
|
22
|
+
if (body.speed !== undefined && body.speed !== 'standard') return false;
|
|
23
|
+
if (body.container !== undefined || body.mcp_servers !== undefined || body.compaction !== undefined) return false;
|
|
24
|
+
if (body.tools !== undefined && (!Array.isArray(body.tools) || !body.tools.every(sharedTool))) return false;
|
|
25
|
+
for (const message of body.messages) {
|
|
26
|
+
if (message.role !== 'system' || !Array.isArray(message.content)) continue;
|
|
27
|
+
for (const block of message.content) {
|
|
28
|
+
if (!['tool_addition', 'tool_removal'].includes(block.type)) continue;
|
|
29
|
+
const tool = block.tool;
|
|
30
|
+
if (!object(tool) || (tool.type !== 'tool_reference'
|
|
31
|
+
&& !(block.type === 'tool_addition' && tool.type === 'tool_definition' && sharedTool(tool.definition)))) return false;
|
|
32
|
+
}
|
|
33
|
+
}
|
|
34
|
+
if (body.tool_choice !== undefined && (!object(body.tool_choice) || !['auto', 'none'].includes(body.tool_choice.type))) return false;
|
|
35
|
+
if (body.context_management !== undefined) {
|
|
36
|
+
const context = body.context_management;
|
|
37
|
+
if (!object(context) || Object.keys(context).some(key => key !== 'edits') || !Array.isArray(context.edits)
|
|
38
|
+
|| context.edits.some(edit => !object(edit) || !CONTEXT_EDITS.has(edit.type))) return false;
|
|
39
|
+
}
|
|
40
|
+
if (body.thinking !== undefined) {
|
|
41
|
+
const thinking = body.thinking;
|
|
42
|
+
if (!object(thinking) || !['adaptive', 'disabled', 'between_tools'].includes(thinking.type)) return false;
|
|
43
|
+
// between_tools is Sonnet 5.5-specific. A routed Opus request uses
|
|
44
|
+
// adaptive thinking; prepareRequest makes that explicit without touching
|
|
45
|
+
// any prior thinking blocks or the conversation prefix they sign.
|
|
46
|
+
if (thinking.type === 'between_tools' && (body.model !== 'claude-sonnet-5-5'
|
|
47
|
+
|| Object.keys(thinking).some(key => key !== 'type') || target === 'claude-sonnet-5')) return false;
|
|
48
|
+
}
|
|
49
|
+
// Sonnet 5 lacks mid-conversation system/tool/effort updates. Never flatten
|
|
50
|
+
// those messages into the top-level prompt: that invalidates signed history.
|
|
51
|
+
if (target === 'claude-sonnet-5' && (body.output_config?.task_budget !== undefined
|
|
52
|
+
|| body.messages.some(message => message.role === 'system' || message.output_config !== undefined))) return false;
|
|
53
|
+
return true;
|
|
54
|
+
}
|
package/src/config.mjs
CHANGED
|
@@ -69,7 +69,7 @@ export function readConfig(env = process.env) {
|
|
|
69
69
|
}
|
|
70
70
|
const models = {
|
|
71
71
|
haiku: env.AUTOROUTER_HAIKU_MODEL ?? 'claude-haiku-4-5-20251001',
|
|
72
|
-
sonnet: env.AUTOROUTER_SONNET_MODEL ?? 'claude-sonnet-5',
|
|
72
|
+
sonnet: env.AUTOROUTER_SONNET_MODEL ?? (clientProfile === 'auto' ? 'claude-sonnet-5-5' : 'claude-sonnet-5'),
|
|
73
73
|
opus: env.AUTOROUTER_OPUS_MODEL ?? 'claude-opus-5-5',
|
|
74
74
|
};
|
|
75
75
|
if (clientProfile === 'auto') {
|
package/src/model-request.mjs
CHANGED
|
@@ -14,6 +14,11 @@ function sonnetNeedsAdaptive(body) {
|
|
|
14
14
|
export function prepareRequest(body, model) {
|
|
15
15
|
const request = { ...body, model };
|
|
16
16
|
const adjustments = [];
|
|
17
|
+
if (model !== body.model && body.model === 'claude-sonnet-5-5' && body.thinking?.type === 'between_tools'
|
|
18
|
+
&& Object.keys(body.thinking).length === 1 && ADAPTIVE_TARGETS.has(model)) {
|
|
19
|
+
request.thinking = { type: 'adaptive' };
|
|
20
|
+
adjustments.push('adaptive_thinking_required');
|
|
21
|
+
}
|
|
17
22
|
if (model !== body.model && body.thinking?.type === 'disabled') {
|
|
18
23
|
if (model === 'claude-sonnet-5-5') {
|
|
19
24
|
const type = sonnetNeedsAdaptive(body) ? 'adaptive' : 'between_tools';
|
package/src/router.mjs
CHANGED
|
@@ -2,6 +2,7 @@ import { createHash } from 'node:crypto';
|
|
|
2
2
|
import { TIERS } from './config.mjs';
|
|
3
3
|
import { buildState, goalFeedbackIndexes } from './prompt-state.mjs';
|
|
4
4
|
import { buildOllamaState, evaluateOllama } from './ollama-evaluator.mjs';
|
|
5
|
+
import { canRouteAutoRequest, hasRoutableSafeguards } from './auto-routing.mjs';
|
|
5
6
|
export { buildState } from './prompt-state.mjs';
|
|
6
7
|
|
|
7
8
|
const hash = value => createHash('sha256').update(JSON.stringify(value)).digest('hex');
|
|
@@ -226,10 +227,13 @@ export class Router {
|
|
|
226
227
|
async route(body, { scope = '', signal, requestClass = '', promptId = '', countTokens } = {}) {
|
|
227
228
|
const start = performance.now();
|
|
228
229
|
const c = this.config;
|
|
229
|
-
|
|
230
|
-
//
|
|
231
|
-
//
|
|
232
|
-
|
|
230
|
+
const autoMode = c.clientProfile === 'auto' || hasRoutableSafeguards(body);
|
|
231
|
+
// Auxiliary permission classifiers keep their model and verdicts. Main
|
|
232
|
+
// execution requests can switch between compatible Sonnet/Opus models
|
|
233
|
+
// while retaining the server review contract verbatim. Unknown contracts
|
|
234
|
+
// still pass through, including any future safeguards version.
|
|
235
|
+
if (requestClass === 'auxiliary' || (body.safeguards !== undefined
|
|
236
|
+
&& (!hasRoutableSafeguards(body) || requestClass === 'compaction'))) {
|
|
233
237
|
// A safeguarded main request still produces the next tool turn. Replace
|
|
234
238
|
// any older routing pin with the actual preserved model so a later
|
|
235
239
|
// request that omits safeguards cannot restore that stale model. Side
|
|
@@ -263,7 +267,7 @@ export class Router {
|
|
|
263
267
|
// Check suspicious input in parallel with Jev. Byte size only triggers a
|
|
264
268
|
// check: common tool catalogs can be 200KB yet occupy far less than 200K
|
|
265
269
|
// tokens. Tiny requests keep the one-call fast path.
|
|
266
|
-
const earlyCount =
|
|
270
|
+
const earlyCount = !autoMode && !capacityLocked && countTokens && CAPACITY_UPGRADE_MODELS.has(c.models.haiku)
|
|
267
271
|
&& (contextSizeBytes(body, c.models.haiku) > 150000 || hasAttachments)
|
|
268
272
|
? safelyCount(c.models.haiku) : undefined;
|
|
269
273
|
const decision = await this.classify(body, signal);
|
|
@@ -272,7 +276,7 @@ export class Router {
|
|
|
272
276
|
// Auto permission mode requires a supported execution model. Retain the
|
|
273
277
|
// evaluator's verdict for observability; stronger compatibility and turn
|
|
274
278
|
// constraints below still decide whether this ordinary choice can apply.
|
|
275
|
-
if (
|
|
279
|
+
if (autoMode && decision.tier === 'haiku') {
|
|
276
280
|
model = c.models.sonnet;
|
|
277
281
|
reason = 'auto_mode_floor';
|
|
278
282
|
}
|
|
@@ -282,6 +286,9 @@ export class Router {
|
|
|
282
286
|
let previous = turnPin?.model;
|
|
283
287
|
const textTurn = !turn.continuation || turn.goalFeedback;
|
|
284
288
|
const textPin = promptPin ?? (!promptId && turn.goalFeedback ? turnPin : undefined);
|
|
289
|
+
const pinnedTarget = (turn.continuation || (textTurn && textPin?.requestedModel === body.model))
|
|
290
|
+
? previous : undefined;
|
|
291
|
+
const sharedAutoRequest = autoMode && canRouteAutoRequest(body, pinnedTarget ?? model);
|
|
285
292
|
// A new human prompt can still carry signed thinking from the preceding
|
|
286
293
|
// turn. Recover that turn's actual routed model when it is known.
|
|
287
294
|
if (!turn.continuation && body.messages.length > 1 && !previous) {
|
|
@@ -300,19 +307,20 @@ export class Router {
|
|
|
300
307
|
// Mid-conversation system messages are only supported by certain models.
|
|
301
308
|
// Keep the client's capable model and all message fields (including
|
|
302
309
|
// clear_at, tool changes, and output_config) instead of down-routing.
|
|
303
|
-
else if (hasSystemMessage) preserve(body.model, 'mid_conversation_system');
|
|
310
|
+
else if (hasSystemMessage && !sharedAutoRequest) preserve(body.model, 'mid_conversation_system');
|
|
304
311
|
else if (unknownModel) preserve(body.model, 'unknown_model');
|
|
305
312
|
// A new native request can explicitly select a model-specific thinking
|
|
306
313
|
// mode, including between_tools. An earlier turn's model is not evidence
|
|
307
314
|
// that it accepts that mode. Existing tool turns retain their pin below.
|
|
308
|
-
else if (modelSpecificThinking && body.thinking.type !== 'enabled' && textTurn) preserve(body.model, 'model_specific_features');
|
|
315
|
+
else if (modelSpecificThinking && body.thinking.type !== 'enabled' && textTurn && !sharedAutoRequest) preserve(body.model, 'model_specific_features');
|
|
309
316
|
// Stop hooks (including /goal) return feedback as user-role text, even
|
|
310
317
|
// though it still serves the same human prompt. Trust the scoped gateway
|
|
311
318
|
// identity instead of treating that text as a new task. A client model
|
|
312
319
|
// change can be an explicit fallback after a failure; do not undo it.
|
|
313
320
|
// Local /goal commands can omit the gateway prompt ID. Exact feedback for
|
|
314
321
|
// a known goal then uses the original conversation anchor as a fallback.
|
|
315
|
-
else if (textTurn && textPin?.requestedModel === body.model && !modelSpecificFeatures
|
|
322
|
+
else if (textTurn && textPin?.requestedModel === body.model && (!modelSpecificFeatures
|
|
323
|
+
|| (autoMode && canRouteAutoRequest(body, textPin.model)))) {
|
|
316
324
|
const needsSonnet = body.thinking?.type === 'adaptive' || body.output_config?.effort || body.max_tokens > 64000;
|
|
317
325
|
if (needsSonnet && (textPin.model === c.models.haiku || rank(textPin.model) === 0)) {
|
|
318
326
|
preserve(c.models.sonnet, 'requires_sonnet_capabilities');
|
|
@@ -322,10 +330,10 @@ export class Router {
|
|
|
322
330
|
else if (turn.continuation && !turn.goalFeedback) keep(previous ? 'tool_turn_pinned' : 'unknown_continuation');
|
|
323
331
|
// Unknown or model-specific features are preserved, never silently removed.
|
|
324
332
|
else if (decision.source === 'fallback' && rank(body.model) >= 1) keep('classifier_unavailable');
|
|
325
|
-
else if (modelSpecificFeatures) keep('model_specific_features');
|
|
326
|
-
else if (thinkingHistory) keep('thinking_history');
|
|
333
|
+
else if (modelSpecificFeatures && !sharedAutoRequest) keep('model_specific_features');
|
|
334
|
+
else if (thinkingHistory && !sharedAutoRequest) keep('thinking_history');
|
|
327
335
|
else if (body.thinking?.type === 'adaptive' || body.output_config?.effort || body.max_tokens > 64000) {
|
|
328
|
-
if (decision.tier === 'haiku') { model = c.models.sonnet; reason = 'requires_sonnet_capabilities'; }
|
|
336
|
+
if (decision.tier === 'haiku') { model = c.models.sonnet; if (!autoMode) reason = 'requires_sonnet_capabilities'; }
|
|
329
337
|
}
|
|
330
338
|
// Account for all context, including system instructions and loaded tool
|
|
331
339
|
// schemas that are intentionally omitted from Jev's bounded excerpt.
|
|
@@ -344,7 +352,10 @@ export class Router {
|
|
|
344
352
|
const configured = TIERS.findIndex(tier => c.models[tier] === value);
|
|
345
353
|
return configured >= 0 ? configured : rank(value);
|
|
346
354
|
};
|
|
347
|
-
|
|
355
|
+
// The verified modern Auto pair shares a native 1M input window. A large
|
|
356
|
+
// prompt is not a reason to pin Opus forever after the task becomes easy.
|
|
357
|
+
// This does not assert that the prompt fits the upstream context limit.
|
|
358
|
+
if (!preserved && largeContext && !sharedAutoRequest) {
|
|
348
359
|
const baseline = previous ?? body.model;
|
|
349
360
|
// Prevent a downgrade; a compatible Haiku client must still be able to
|
|
350
361
|
// upgrade a demanding request to a larger-context, stronger model.
|
|
@@ -362,6 +373,9 @@ export class Router {
|
|
|
362
373
|
capacityUpgraded = true;
|
|
363
374
|
}
|
|
364
375
|
}
|
|
376
|
+
if (autoMode && model !== body.model && !canRouteAutoRequest(body, model)) {
|
|
377
|
+
preserve(body.model, 'auto_mode_incompatible');
|
|
378
|
+
}
|
|
365
379
|
const identifiableUpgrade = capacityUpgraded && (turn.index >= 0 || promptId);
|
|
366
380
|
if ((!turn.continuation || previous || identifiableUpgrade || reason === 'mid_conversation_system') && requestClass !== 'compaction') {
|
|
367
381
|
const pin = { model, requestedModel: body.model };
|
package/src/statusline.mjs
CHANGED
|
@@ -6,7 +6,7 @@ const REASONS = {
|
|
|
6
6
|
model_specific_features: 'model features', large_or_multimodal_request: 'large request',
|
|
7
7
|
context_capacity: 'large context',
|
|
8
8
|
internal_request: 'internal request', unknown_model: 'custom model', low_confidence: 'low confidence',
|
|
9
|
-
auto_mode_floor: 'Auto mode floor', auto_mode_safeguards: 'Auto safety',
|
|
9
|
+
auto_mode_floor: 'Auto mode floor', auto_mode_safeguards: 'Auto safety', auto_mode_incompatible: 'Auto model guard',
|
|
10
10
|
};
|
|
11
11
|
const CLASSIFIER_ERRORS = {
|
|
12
12
|
timeout: 'timeout', http_error: 'HTTP error', invalid_response: 'invalid response', network_error: 'network error',
|