claude-autorouter 0.3.5 → 0.3.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +15 -4
- package/README.md +18 -1
- package/bin/autorouter.mjs +25 -2
- package/docs/development.md +2 -0
- package/docs/reference.md +69 -6
- package/docs/releasing.md +5 -1
- package/package.json +1 -1
- package/src/auth.mjs +16 -0
- package/src/auto-routing.mjs +54 -0
- package/src/config.mjs +32 -7
- package/src/model-request.mjs +5 -0
- package/src/onboarding.mjs +18 -3
- package/src/prompt-state.mjs +48 -0
- package/src/router.mjs +51 -10
- package/src/server.mjs +17 -1
- package/src/session-log.mjs +171 -0
- package/src/status-state.mjs +1 -1
- package/src/statusline.mjs +3 -2
- package/src/user-config.mjs +1 -0
package/.env.example
CHANGED
|
@@ -5,8 +5,17 @@ AUTOROUTER_CLIENT_PROFILE=compatible
|
|
|
5
5
|
AUTOROUTER_EVALUATOR=jev
|
|
6
6
|
# The launcher enables the router status line for this session. Set 0 to keep your own.
|
|
7
7
|
AUTOROUTER_STATUSLINE=1
|
|
8
|
+
# Claude's Auto permission mode needs a supported Sonnet or Opus client.
|
|
9
|
+
# This profile switches between Sonnet 5.5 and Opus 5.5 on new human tasks.
|
|
10
|
+
# Haiku decisions become Sonnet; tool continuations keep the selected model.
|
|
11
|
+
# AutoRouter also selects it for: claude-autorouter claude --permission-mode auto
|
|
12
|
+
# AUTOROUTER_CLIENT_PROFILE=auto
|
|
8
13
|
# Optional metadata logs on stderr. Redirect stderr to a file when using the UI.
|
|
9
14
|
# AUTOROUTER_DEBUG=1
|
|
15
|
+
# Optional persistent JSONL decision logs, one file per session per launch.
|
|
16
|
+
# Includes up to 500 characters of user prompt text; keep the directory local.
|
|
17
|
+
# Unset or empty disables logging. This does not print prompts in the terminal.
|
|
18
|
+
# AUTOROUTER_SESSION_LOG_DIR=/absolute/path/to/autorouter-sessions
|
|
10
19
|
# Optional: allow two tool-free Stop-hook continuations, then end the turn on
|
|
11
20
|
# the third block. Applies to /goal and all Stop/SubagentStop hooks.
|
|
12
21
|
# Unset keeps Claude's default (currently 8); 0 DISABLES the cap.
|
|
@@ -16,10 +25,12 @@ TYPESAFE_API_KEY=
|
|
|
16
25
|
# For API billing instead, set AUTOROUTER_AUTH_MODE=api-key and fill this in.
|
|
17
26
|
# ANTHROPIC_API_KEY=
|
|
18
27
|
|
|
19
|
-
#
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
28
|
+
# Optional overrides; leaving these unset uses each profile's defaults.
|
|
29
|
+
# Auto defaults to Sonnet 5.5; compatible/native default to Sonnet 5.
|
|
30
|
+
# An explicit Sonnet override also applies in Auto mode.
|
|
31
|
+
# AUTOROUTER_HAIKU_MODEL=claude-haiku-4-5-20251001
|
|
32
|
+
# AUTOROUTER_SONNET_MODEL=claude-sonnet-5-5
|
|
33
|
+
# AUTOROUTER_OPUS_MODEL=claude-opus-5-5
|
|
23
34
|
AUTOROUTER_JEV_MODEL=jev-latest
|
|
24
35
|
AUTOROUTER_JEV_TIMEOUT_MS=1500
|
|
25
36
|
# Only suspicious context sizes need counting; this overlaps the evaluator.
|
package/README.md
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# Claude AutoRouter
|
|
2
2
|
|
|
3
|
-
Use Haiku, Sonnet, and Opus in one Claude Code session. A local gateway classifies
|
|
3
|
+
Use Haiku, Sonnet, and Opus in one Claude Code session. A local gateway classifies coding requests with the selected evaluator, applies compatibility and context checks, and streams the selected model's response back to Claude Code. Claude's internal permission classifiers retain their selected model; execution requests keep their server safety-review settings and verdicts when routed. [TypeSafe Jev](https://typesafe.ai/blog/introducing-system-one-models-and-jev) is the default; an experimental Ollama backend evaluates requests locally.
|
|
4
4
|
|
|
5
5
|
Requires Node.js 22+, macOS or Linux (including WSL), an installed `claude` command, and a Claude subscription login or Anthropic API key. The default evaluator also requires a [TypeSafe API key](https://console.typesafe.ai). There are no runtime package dependencies. Native Windows is not supported in this release.
|
|
6
6
|
|
|
@@ -35,6 +35,14 @@ claude-autorouter --version
|
|
|
35
35
|
|
|
36
36
|
For API billing, use `claude-autorouter setup --auth-mode api-key`. Use `--force` to replace an existing config. Automation can supply `TYPESAFE_API_KEY` and, in API-key mode, `ANTHROPIC_API_KEY` through the environment; keys are never command-line arguments. `doctor` checks local configuration and Claude installation/login state without paid requests. See the [configuration reference](docs/reference.md#configuration).
|
|
37
37
|
|
|
38
|
+
To use automatic Sonnet/Opus routing with Claude's Auto permission mode (AutoRouter 0.3.7+):
|
|
39
|
+
|
|
40
|
+
```sh
|
|
41
|
+
claude-autorouter claude --permission-mode auto
|
|
42
|
+
```
|
|
43
|
+
|
|
44
|
+
This selects the Auto profile, defaulting to Sonnet 5.5 and Opus 5.5. The evaluator can choose again for each new human task, while a task's tool calls and goal continuations retain its selected model. A Haiku verdict uses Sonnet. Native safety review stays enabled; organization policies still apply. Use `AUTOROUTER_CLIENT_PROFILE=auto` for sessions where you select Auto in Claude's UI or saved settings. Version 0.3.6 enabled Auto permissions but passed server-reviewed execution through without routing; version 0.3.7 removes that restriction for compatible requests. [Auto-mode support and limitations](docs/reference.md#auto-permission-mode).
|
|
45
|
+
|
|
38
46
|
## What you see
|
|
39
47
|
|
|
40
48
|
The launcher adds a temporary status line and leaves saved Claude Code settings unchanged:
|
|
@@ -46,6 +54,15 @@ The launcher adds a temporary status line and leaves saved Claude Code settings
|
|
|
46
54
|
|
|
47
55
|
The confirmed model comes from Anthropic's response. Claude's own model label can still show its Haiku starting model. `API ctx` measures input against the actual model's known window; a different client limit remains visible as `CLI ctx`.
|
|
48
56
|
|
|
57
|
+
For a separate decision log per session (optional, disabled by default; AutoRouter 0.3.6+):
|
|
58
|
+
|
|
59
|
+
```sh
|
|
60
|
+
env AUTOROUTER_SESSION_LOG_DIR="$HOME/.local/state/claude-autorouter/sessions" \
|
|
61
|
+
claude-autorouter claude
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
Each JSONL record includes a bounded prompt excerpt, selected model, decision latency, and routing reason. Files persist after Claude exits; terminal output stays quiet. [Session logs](docs/reference.md#session-decision-logs).
|
|
65
|
+
|
|
49
66
|
Savings are an **API-equivalent estimate for the same token counts**, using Opus as the baseline. They do not measure subscription bill reductions or quota credits and exclude Jev and local compute costs. [Status line and savings details](docs/reference.md#status-line-and-savings).
|
|
50
67
|
|
|
51
68
|
## Experimental local evaluator
|
package/bin/autorouter.mjs
CHANGED
|
@@ -4,10 +4,11 @@ import { spawn } from 'node:child_process';
|
|
|
4
4
|
import { readFileSync } from 'node:fs';
|
|
5
5
|
import { readConfig, requireKeys } from '../src/config.mjs';
|
|
6
6
|
import { createRouterServer, listen } from '../src/server.mjs';
|
|
7
|
-
import { buildClaudeEnv, conflictingProviders } from '../src/auth.mjs';
|
|
7
|
+
import { buildClaudeEnv, clientProfileForLaunch, conflictingProviders } from '../src/auth.mjs';
|
|
8
8
|
import { dirname } from 'node:path';
|
|
9
9
|
import { createStatusState } from '../src/status-state.mjs';
|
|
10
10
|
import { addStatusLineSettings } from '../src/status-settings.mjs';
|
|
11
|
+
import { createSessionLog } from '../src/session-log.mjs';
|
|
11
12
|
import { loadUserConfig } from '../src/user-config.mjs';
|
|
12
13
|
import { setup, doctor, ollamaDeadlineText } from '../src/onboarding.mjs';
|
|
13
14
|
import { setupOllama } from '../src/ollama-setup.mjs';
|
|
@@ -21,9 +22,11 @@ if (['--version', '-v', 'version'].includes(command)) {
|
|
|
21
22
|
|
|
22
23
|
Usage:
|
|
23
24
|
claude-autorouter setup [--auth-mode subscription|api-key] [--force]
|
|
25
|
+
[--client-profile compatible|native|auto]
|
|
24
26
|
[--evaluator jev|ollama]
|
|
25
27
|
[--ollama-model MODEL] [--ollama-timeout-ms N] [--pull]
|
|
26
28
|
[--stop-hook-block-cap N]
|
|
29
|
+
[--session-log-dir DIR]
|
|
27
30
|
claude-autorouter doctor
|
|
28
31
|
claude-autorouter claude [Claude Code arguments]
|
|
29
32
|
claude-autorouter serve
|
|
@@ -48,6 +51,10 @@ AUTOROUTER_AUTH_MODE=subscription uses your saved Claude Code login.
|
|
|
48
51
|
Without setup, AUTOROUTER_AUTH_MODE defaults to api-key and also requires ANTHROPIC_API_KEY.
|
|
49
52
|
AUTOROUTER_CLIENT_PROFILE=compatible (default) enables all three routing tiers.
|
|
50
53
|
Use AUTOROUTER_CLIENT_PROFILE=native to retain Claude Code's own model/thinking settings.
|
|
54
|
+
Use AUTOROUTER_CLIENT_PROFILE=auto for Auto permission mode: Sonnet/Opus routing, native thinking.
|
|
55
|
+
Auto defaults to Sonnet 5.5 and Opus 5.5, switching on new human tasks and retaining tool turns.
|
|
56
|
+
An explicit claude --permission-mode auto selects the auto profile for that launch.
|
|
57
|
+
Claude's permission checks and organization policies still apply; Haiku does not support Auto mode.
|
|
51
58
|
Optional CLAUDE_CODE_STOP_HOOK_BLOCK_CAP=N limits consecutive tool-free Stop-hook continuations.
|
|
52
59
|
Use 2 to stop on the third block; applies to /goal and all Stop/SubagentStop hooks.
|
|
53
60
|
Unset preserves Claude's default; 0 disables the cap. Setup --stop-hook-block-cap N saves it.
|
|
@@ -55,6 +62,8 @@ Standalone serve also requires AUTOROUTER_TOKEN (at least 16 characters).
|
|
|
55
62
|
The claude launcher creates a temporary credential and an ephemeral port.
|
|
56
63
|
It enables an AutoRouter status line for this session (AUTOROUTER_STATUSLINE=0 to opt out).
|
|
57
64
|
Launcher logs are quiet by default; AUTOROUTER_DEBUG=1 enables diagnostic logs on stderr.
|
|
65
|
+
AUTOROUTER_SESSION_LOG_DIR writes private per-session JSONL decision logs with prompt excerpts.
|
|
66
|
+
Unset or empty disables session logs. Setup --session-log-dir DIR saves the directory.
|
|
58
67
|
Jev sends prompt excerpts to TypeSafe; Ollama keeps classification on this machine.
|
|
59
68
|
Complete inference requests still go to Anthropic. See README.md.`);
|
|
60
69
|
} else if (command === 'setup' || command === 'doctor') {
|
|
@@ -71,14 +80,24 @@ Complete inference requests still go to Anthropic. See README.md.`);
|
|
|
71
80
|
} else {
|
|
72
81
|
let server;
|
|
73
82
|
let status;
|
|
83
|
+
let sessionLog;
|
|
84
|
+
let stopping;
|
|
74
85
|
const stop = () => {
|
|
86
|
+
if (stopping) return stopping;
|
|
75
87
|
if (server) { server.close(); server.closeAllConnections(); }
|
|
76
88
|
status?.close();
|
|
89
|
+
// Drain accepted decision records before normal process exit. Pending
|
|
90
|
+
// filesystem writes keep Node alive; no timer or fire-and-forget buffer.
|
|
91
|
+
stopping = Promise.resolve().then(() => sessionLog?.close()).catch(() => {});
|
|
92
|
+
return stopping;
|
|
77
93
|
};
|
|
78
94
|
try {
|
|
79
95
|
if (command === 'serve' && args.length) throw new Error('Usage: claude-autorouter serve');
|
|
80
96
|
const runtimeEnv = loadUserConfig().env;
|
|
81
|
-
const config = readConfig(
|
|
97
|
+
const config = readConfig(command === 'claude' ? {
|
|
98
|
+
...runtimeEnv,
|
|
99
|
+
AUTOROUTER_CLIENT_PROFILE: clientProfileForLaunch(runtimeEnv.AUTOROUTER_CLIENT_PROFILE ?? 'compatible', args),
|
|
100
|
+
} : runtimeEnv);
|
|
82
101
|
requireKeys(config);
|
|
83
102
|
const diagnosticLogs = command === 'serve' || runtimeEnv.AUTOROUTER_DEBUG === '1';
|
|
84
103
|
if (command === 'claude') {
|
|
@@ -109,11 +128,15 @@ Complete inference requests still go to Anthropic. See README.md.`);
|
|
|
109
128
|
}
|
|
110
129
|
} else console.error('AutoRouter status line unavailable: could not create local status storage.');
|
|
111
130
|
}
|
|
131
|
+
if (config.sessionLogDir) sessionLog = await createSessionLog(config.sessionLogDir, {
|
|
132
|
+
warn: message => console.error(message),
|
|
133
|
+
});
|
|
112
134
|
// Claude owns the terminal while its UI is running. Status updates use the
|
|
113
135
|
// local snapshot independently; proxy JSON must not write over the UI.
|
|
114
136
|
server = createRouterServer(config, {
|
|
115
137
|
log: diagnosticLogs ? undefined : () => {},
|
|
116
138
|
onStatus: event => status?.update(event),
|
|
139
|
+
onDecision: sessionLog ? entry => sessionLog.record(entry) : undefined,
|
|
117
140
|
});
|
|
118
141
|
const address = await listen(server, command === 'claude' ? 0 : config.port);
|
|
119
142
|
const baseUrl = `http://127.0.0.1:${address.port}`;
|
package/docs/development.md
CHANGED
|
@@ -12,6 +12,8 @@ npm run test:package
|
|
|
12
12
|
|
|
13
13
|
The test suite uses local mocks and fake credentials. It covers Jev and Ollama routing, bounded prompt extraction, confidence and timeout fallback, token checks, model continuity, authentication forwarding, streaming, cancellation, status state, savings, and launcher behavior. Tests that start HTTP services require loopback binding. Local tests make no paid provider calls or model downloads.
|
|
14
14
|
|
|
15
|
+
Auto-mode regressions exercise Sonnet → Opus → Opus tool continuation → Sonnet in one conversation, with and without gateway prompt IDs. They retain signed thinking, native context edits, mid-conversation system messages, safety-review settings, and streamed verdicts, including denied actions. Separate capability tests keep unknown review contracts and incompatible model features from being routed.
|
|
16
|
+
|
|
15
17
|
Package validation checks the distributable and installed command rather than relying on the source checkout's paths. Review the [release procedure](releasing.md) before distributing a tarball.
|
|
16
18
|
|
|
17
19
|
## Run from source
|
package/docs/reference.md
CHANGED
|
@@ -6,7 +6,9 @@
|
|
|
6
6
|
| --- | --- |
|
|
7
7
|
| `claude-autorouter setup` | Save subscription-mode configuration and a Jev key |
|
|
8
8
|
| `claude-autorouter setup --auth-mode api-key` | Configure Jev and Anthropic API-key billing |
|
|
9
|
+
| `claude-autorouter setup --client-profile auto` | Save a Sonnet/Opus profile compatible with Claude's Auto permission mode |
|
|
9
10
|
| `claude-autorouter setup --stop-hook-block-cap 2` | Opt into a shorter native Stop-hook continuation cap during setup |
|
|
11
|
+
| `claude-autorouter setup --session-log-dir DIR` | Save an opt-in directory for per-session JSONL decision logs |
|
|
10
12
|
| `claude-autorouter setup --evaluator ollama --pull` | Configure the native local evaluator and download its selected model if missing |
|
|
11
13
|
| `claude-autorouter setup --evaluator ollama --ollama-timeout-ms 0 --force` | Save a disabled runtime evaluator deadline |
|
|
12
14
|
| `claude-autorouter setup --force` | Replace an existing user config |
|
|
@@ -43,13 +45,14 @@ For an environment-only subscription launch, set `AUTOROUTER_AUTH_MODE=subscript
|
|
|
43
45
|
| `AUTOROUTER_CONFIG` | see path order above | Explicit user config path |
|
|
44
46
|
| `AUTOROUTER_AUTH_MODE` | `api-key` without saved config; setup selects `subscription` | Authentication mode |
|
|
45
47
|
| `AUTOROUTER_EVALUATOR` | `jev` | `jev` or local `ollama` classification |
|
|
46
|
-
| `AUTOROUTER_CLIENT_PROFILE` | `compatible` | `native` retains
|
|
48
|
+
| `AUTOROUTER_CLIENT_PROFILE` | `compatible` | `native` retains client model/thinking settings; `auto` starts with Sonnet when no explicit model is set and excludes Haiku from task routing |
|
|
47
49
|
| `AUTOROUTER_STATUSLINE` | enabled | `0` retains your existing status line |
|
|
48
50
|
| `AUTOROUTER_DEBUG` | off | `1` enables launcher metadata logs on stderr |
|
|
51
|
+
| `AUTOROUTER_SESSION_LOG_DIR` | off | Write per-session JSONL decision logs with prompt excerpts into this directory; unset or empty disables it |
|
|
49
52
|
| `CLAUDE_CODE_STOP_HOOK_BLOCK_CAP` | unset; Claude currently uses `8` | Optional cap on consecutive Stop/SubagentStop continuations without tool use; `0` disables the cap |
|
|
50
53
|
| `ENABLE_TOOL_SEARCH` | `true` in launcher when unset | Load MCP tool definitions on demand; explicit values are preserved |
|
|
51
54
|
| `AUTOROUTER_HAIKU_MODEL` | `claude-haiku-4-5-20251001` | Routine tier |
|
|
52
|
-
| `AUTOROUTER_SONNET_MODEL` | `claude-sonnet-5` | Standard tier |
|
|
55
|
+
| `AUTOROUTER_SONNET_MODEL` | `claude-sonnet-5`; `claude-sonnet-5-5` in the `auto` profile | Standard tier |
|
|
53
56
|
| `AUTOROUTER_OPUS_MODEL` | `claude-opus-5-5` | Demanding tier and savings baseline |
|
|
54
57
|
| `AUTOROUTER_JEV_MODEL` | `jev-latest` | Classifier version |
|
|
55
58
|
| `AUTOROUTER_JEV_TIMEOUT_MS` | `1500` | Classifier deadline in milliseconds |
|
|
@@ -170,16 +173,42 @@ The following policy applies after classification:
|
|
|
170
173
|
- Jev confidence below 0.75 prevents a downgrade below Sonnet or the requested tier. Ollama returns a tier without calibrated confidence; its failure handling and compatibility guards still apply.
|
|
171
174
|
- Tool continuations retain the model chosen at the start of the human turn. Session, agent, and prompt headers identify turns; normalized conversation content provides a fallback. Text feedback from a Stop hook also retains the model when it serves the same gateway prompt ID and the client has not changed its requested model, subject to capability and context checks. Moving prompt-cache markers does not create a new turn.
|
|
172
175
|
- Claude's local `/goal` command can omit the prompt-ID header. For that path, an exact feedback label matching a preceding expanded `/goal` command keeps the original task and conversation anchor. This narrow text fallback also recognizes Claude's repeated-goal truncation format; arbitrary hook text is not treated as a goal. Feedback remains in the evaluator's recent conversation and the full API request. A new human message becomes the current task normally. The status line shows `prompt pinned` or `goal pinned` when either text-continuation rule applies.
|
|
173
|
-
- Thinking history, fixed-budget thinking, server tools, context management, and other recognized model-specific features preserve the current model. Adaptive thinking, effort, and output above 64K prevent a Haiku choice. Fields are never stripped to force a downgrade.
|
|
174
|
-
- Mid-conversation `system` messages preserve the requested model
|
|
175
|
-
-
|
|
176
|
+
- Thinking history, fixed-budget thinking, server tools, context management, and other recognized model-specific features preserve the current model except for the verified shared capabilities of the modern Auto-mode Sonnet/Opus pair described below. Adaptive thinking, effort, and output above 64K prevent a Haiku choice. Fields are never stripped to force a downgrade.
|
|
177
|
+
- Mid-conversation `system` messages preserve the requested model unless both Auto-mode models support them; they always pass through unchanged. They do not count as a tool continuation by themselves.
|
|
178
|
+
- Auxiliary requests, including Claude's Auto permission classifier, pass through on their requested model without Jev/Ollama evaluation, token checks, or turn-state changes. Compaction retains its existing model and context-capacity policy. Recognized server-reviewed execution requests can route between compatible Sonnet/Opus models while retaining `safeguards` and all verdicts unchanged. Unknown safeguards contracts pass through. Token counting and model discovery pass through without classification.
|
|
176
179
|
|
|
177
180
|
The default `compatible` profile starts Claude with Haiku-compatible requests and client-requested thinking disabled. AutoRouter uses adaptive thinking when upgrading these requests to Opus 5/5.5. Starting with 0.3.3, routing to exact `claude-sonnet-5-5` translates disabled thinking to `between_tools`, which skips up-front thinking but permits progress updates between tool calls. At `xhigh`/`max` effort, or when per-message effort differs from the top-level setting (default `high`), it uses adaptive thinking while preserving the effort settings. Token counting uses the same adaptation. Sonnet 5 still accepts disabled thinking and is unchanged. See [Sonnet 5.5 thinking requirements](https://platform.claude.com/docs/en/models/sonnet-5-5/migration-guide).
|
|
178
181
|
|
|
179
|
-
Explicit native `between_tools` and unknown thinking modes retain the incoming model on new human turns. Signed thinking blocks pass through unchanged and existing tool turns retain their model pin. `AUTOROUTER_CLIENT_PROFILE=native` preserves normal client settings, which can constrain routing. An explicit Claude `--model` argument overrides the starting model, but `/model` and `--model` are requested models, not locks on the routed result. Native same-model requests and unknown model aliases are not rewritten; clients must use settings supported by that model.
|
|
182
|
+
Explicit native `between_tools` and unknown thinking modes retain the incoming model on new human turns outside Auto routing. In Auto routing, a known Sonnet 5.5 `between_tools` request can upgrade to Opus with adaptive thinking. Signed thinking blocks pass through unchanged and existing tool turns retain their model pin. `AUTOROUTER_CLIENT_PROFILE=native` preserves normal client settings, which can constrain routing. An explicit Claude `--model` argument overrides the starting model, but `/model` and `--model` are requested models, not locks on the routed result. Native same-model requests and unknown model aliases are not rewritten; clients must use settings supported by that model.
|
|
180
183
|
|
|
181
184
|
The launcher enables `ENABLE_TOOL_SEARCH=true` when unset. Claude can otherwise disable on-demand MCP discovery when using a custom API address, loading connected-tool schemas into even a fresh conversation. Explicit values, including `false` or `auto:5`, are preserved. Managed settings and always-loaded tools can still affect deferral. See [Claude Code tool search](https://code.claude.com/docs/en/mcp#configure-tool-search).
|
|
182
185
|
|
|
186
|
+
### Auto permission mode
|
|
187
|
+
|
|
188
|
+
The default `compatible` profile starts Claude as Haiku to permit three-tier routing. Claude's Auto permission mode does not support Haiku, even if AutoRouter routes an API request to Sonnet. Eligibility is based on Claude's selected client model. Gateways themselves are supported. See [Claude's Auto-mode requirements](https://code.claude.com/docs/en/permission-modes#eliminate-permission-prompts-with-auto-mode).
|
|
189
|
+
|
|
190
|
+
AutoRouter 0.3.6 introduced an `auto` client profile but bypassed evaluation for server-reviewed execution. Version 0.3.7 adds automatic switching on those requests. Launch with:
|
|
191
|
+
|
|
192
|
+
```sh
|
|
193
|
+
claude-autorouter claude --permission-mode auto
|
|
194
|
+
```
|
|
195
|
+
|
|
196
|
+
An explicit `--permission-mode auto` (or `--permission-mode=auto`) selects the profile for that launch, including when your saved profile is `native`. All arguments still go to Claude unchanged. For Auto chosen from Claude's UI or existing settings instead, use:
|
|
197
|
+
|
|
198
|
+
```sh
|
|
199
|
+
env AUTOROUTER_CLIENT_PROFILE=auto claude-autorouter claude
|
|
200
|
+
```
|
|
201
|
+
|
|
202
|
+
The profile defaults to Sonnet 5.5 and Opus 5.5, preserving explicit configured model IDs. The evaluator chooses Sonnet or Opus for each new human task; a routine Haiku verdict uses Sonnet and shows `Auto mode floor`. Tool and `/goal` continuations stay on the selected execution model. Claude's initial client model remains separate from the routed model. An explicit client `--model` or `ANTHROPIC_MODEL` can still make Auto unavailable if it selects Haiku or another unsupported model; choose a supported Sonnet or Opus instead.
|
|
203
|
+
|
|
204
|
+
Claude remains responsible for enabling the permission mode and enforcing organization settings, account availability, and tool rules. The profile does not enable Auto by itself or override `disableAutoMode`. AutoRouter does not reproduce Claude's settings precedence to infer a mode from settings files. For new saved configurations, `setup --client-profile auto` persists the profile; for an existing config, change only `AUTOROUTER_CLIENT_PROFILE` to `"auto"` to retain your other settings. `setup --force` replaces the config.
|
|
205
|
+
|
|
206
|
+
**Safety review:** Claude's permission-classifier requests retain their exact requested model and skip AutoRouter's evaluator. Ordinary execution requests with the known `dangerous_tool_use` version-1 review contract are evaluated and routed, retaining the complete `safeguards` object, beta headers, and streamed safety verdicts. This also detects server review when Auto was selected in Claude's UI rather than through the launch flag. Unknown or malformed review contracts and safeguarded compaction pass through with `Auto safety`; a target that cannot accept the request shows `Auto model guard`. AutoRouter never turns off server review or converts denied actions to approvals. See [server-side classifier review](https://code.claude.com/docs/en/permission-modes#server-side-classifier-review).
|
|
207
|
+
|
|
208
|
+
**Shared execution capabilities:** automatic Auto routing supports exact Sonnet 5/5.5 and Opus 5/5.5 IDs. The default 5.5 pair shares a native 1M context window, adaptive thinking, native context-editing strategies, and mid-conversation system updates. These fields and existing signed thinking no longer pin every future human task. Sonnet 5 cannot accept mid-conversation system messages, per-message effort, or task budgets; use Sonnet 5.5 for those sessions. Unknown context-editing strategies, specialized server tools, fixed thinking budgets, fast mode, older/custom targets, and other incompatible requests still retain a compatible model. A Sonnet 5.5 `between_tools` request uses adaptive thinking when upgraded to Opus; effort and conversation history remain unchanged.
|
|
209
|
+
|
|
210
|
+
Thinking blocks stay verbatim in the conversation. Anthropic may drop blocks the selected model cannot read, so switching models does not preserve access to every model's private reasoning on every turn. User text, tool results, and prior answers remain available. Returning to a model can make its preserved thinking readable again. Model changes can also miss the prior model's prompt cache. See [preserved thinking and model switching](https://platform.claude.com/docs/en/build-with-claude/preserved-thinking).
|
|
211
|
+
|
|
183
212
|
### Context capacity
|
|
184
213
|
|
|
185
214
|
Context-relevant JSON above 150KB, or image/document blocks including those inside tool results, trigger a token check for otherwise compatible small models. This byte threshold is a trigger, not a token estimate. The check includes system instructions and active tool schemas. Unused deferred schemas are excluded from the trigger; discovered references and historical tool calls add them back. Unknown shapes are counted conservatively.
|
|
@@ -218,6 +247,40 @@ This estimate does not measure subscription bill savings or quota credits. It ex
|
|
|
218
247
|
|
|
219
248
|
Set `AUTOROUTER_STATUSLINE=0` to retain an existing status line. Other `--settings` values are retained in the temporary overlay; source-relative Read/Edit rules keep their anchors. Ambiguous relative sandbox paths cause the launcher to skip the overlay and pass original settings through with a notice. Safe mode disables custom status lines; print mode has no status-line UI. Standalone `serve` does not install one.
|
|
220
249
|
|
|
250
|
+
## Session decision logs
|
|
251
|
+
|
|
252
|
+
AutoRouter 0.3.6 adds optional persistent logs, separate from stderr and the temporary status-line snapshot. Logging is disabled by default. Enable it for one launch:
|
|
253
|
+
|
|
254
|
+
```sh
|
|
255
|
+
env AUTOROUTER_SESSION_LOG_DIR="$HOME/.local/state/claude-autorouter/sessions" \
|
|
256
|
+
claude-autorouter claude
|
|
257
|
+
```
|
|
258
|
+
|
|
259
|
+
The setting also works with `serve` and the Auto-compatible profile. For a new saved configuration, add `--session-log-dir DIR` to `setup`. For an existing config, add `AUTOROUTER_SESSION_LOG_DIR` with an absolute directory path to preserve your other settings. Setup resolves relative paths at setup time; an environment-only relative path resolves from the launch directory. Environment values override saved values; `AUTOROUTER_SESSION_LOG_DIR=''` disables a saved preference for one launch. `doctor` reports the setting without creating log files.
|
|
260
|
+
|
|
261
|
+
Files are named `autorouter-session-*.jsonl`: one file per observed Claude session within a router launch, with a timestamp, random launch identifier, and hashed session identifier in the name. A resumed session in a new launch creates a new file. Requests without a session header share an anonymous file for that launch. Subagents with the same session ID share its file and retain their agent ID. Files are created only when a decision is recorded, and remain after the session ends.
|
|
262
|
+
|
|
263
|
+
Every line is a standalone JSON object. The key fields look like this (additional IDs and routing metadata are included):
|
|
264
|
+
|
|
265
|
+
```json
|
|
266
|
+
{"schema_version":1,"event":"decision","timestamp":"2026-09-30T12:00:00.000Z","session_id":"example-session","prompt_excerpt":"Fix the typo in README.md","prompt_truncated":false,"requested_model":"claude-haiku-4-5-20251001","selected_model":"claude-sonnet-5","decision_latency_ms":214.37,"source":"jev","reason":"classified"}
|
|
267
|
+
```
|
|
268
|
+
|
|
269
|
+
- `prompt_excerpt`: up to 500 Unicode characters of the current human task for main requests or requests without a class header. Tool continuations and recognized `/goal` feedback keep the originating human task. System instructions, standalone reminder blocks, tool results, images, documents, and thinking are omitted. Auxiliary classifiers, compaction, subagents, and workflows have empty excerpts; a new attachment-only task also has an empty excerpt. `prompt_truncated` indicates that text exceeded the limit.
|
|
270
|
+
- `selected_model`: AutoRouter's final selected model after compatibility checks, before the upstream response. It does not confirm which model successfully answered.
|
|
271
|
+
- `decision_latency_ms`: time spent making the routing decision, including evaluator waiting, cache lookup, and any context checks. It excludes Claude generation time and log writing. A `passthrough` entry can be near zero because no evaluator was called.
|
|
272
|
+
- `source` and `reason`: distinguish evaluator choices, cache hits, fallbacks, turn/model constraints, and native safety pass-through. `classified_tier` is included when an evaluator returned a tier, which may differ from the final selected model.
|
|
273
|
+
|
|
274
|
+
Inspect a file with:
|
|
275
|
+
|
|
276
|
+
```sh
|
|
277
|
+
jq -c '{prompt_excerpt, selected_model, decision_latency_ms, source, reason}' /path/to/autorouter-session-EXAMPLE.jsonl
|
|
278
|
+
```
|
|
279
|
+
|
|
280
|
+
There is one record per completed routing decision, including requests whose upstream call later fails. Requests rejected before routing or cancelled before a decision are not recorded. Logs contain user text and are local plaintext: the feature is disabled by default, new directories use `0700`, and files use `0600`. Existing directory permissions are left unchanged. Authentication headers, provider replies, full transcripts, and tool payloads are not logged; text you put directly in a prompt can appear in its excerpt.
|
|
281
|
+
|
|
282
|
+
Writes run asynchronously through a bounded 1 MiB queue and support up to 128 session files per router process. Normal shutdown drains accepted records. Filesystem failure or a queue/session limit disables further logging with one generic warning while routing continues. An abrupt process kill or storage failure can lose unwritten records. Logs are retained without automatic rotation or deletion; manage them in your chosen directory. Log filenames are ignored by this repository and excluded from the npm package.
|
|
283
|
+
|
|
221
284
|
## Troubleshooting
|
|
222
285
|
|
|
223
286
|
Run `claude-autorouter doctor` first. It performs local checks without Claude generations or Jev calls, including local HTTP checks for the selected Ollama model. It cannot establish Anthropic/TypeSafe availability, current quota, or whether a key will be accepted remotely.
|
package/docs/releasing.md
CHANGED
|
@@ -12,7 +12,11 @@ Version `0.3.4` preserves the selected model across Stop-hook feedback for the s
|
|
|
12
12
|
|
|
13
13
|
Version `0.3.5` adds opt-in saved configuration for Claude's native `CLAUDE_CODE_STOP_HOOK_BLOCK_CAP`, with `setup --stop-hook-block-cap N`, validation, and `doctor` reporting. A value of `2` permits two consecutive Stop-hook continuations without tool use and ends the turn on the third blocking verdict, leaving an unmet goal set. This affects all Stop/SubagentStop hooks, and tool activity resets the counter. Defaults and completion verdicts are unchanged; `0` disables the guard. See [shorter Stop-hook loops](reference.md#shorter-stop-hook-loops-opt-in).
|
|
14
14
|
|
|
15
|
-
|
|
15
|
+
Version `0.3.6` fixes Auto permission-mode launches with an Auto-compatible Sonnet/Opus profile while preserving Claude's permission classifiers and server safety-review requests. It also adds optional per-session JSONL decision logs containing a bounded human prompt excerpt, selected model, and routing latency. Logging is disabled by default; set `AUTOROUTER_SESSION_LOG_DIR` or use `setup --session-log-dir DIR` to enable it. See [Auto permission mode](reference.md#auto-permission-mode) and [session decision logs](reference.md#session-decision-logs).
|
|
16
|
+
|
|
17
|
+
Version `0.3.7` enables automatic Sonnet/Opus switching for compatible Auto-mode execution requests, including requests carrying the known server safety-review contract. The Auto profile defaults to Sonnet 5.5 and Opus 5.5, floors Haiku decisions to Sonnet, and retains the selected model through tool and goal continuations. Shared native context edits, mid-conversation system messages, and signed thinking history no longer pin new human tasks. Permission-classifier requests and safety verdicts remain unchanged; unknown contracts and incompatible model features still preserve a compatible model. Explicit model overrides remain in effect. See [Auto permission mode](reference.md#auto-permission-mode).
|
|
18
|
+
|
|
19
|
+
The GitHub repository is private. Publishing to npm makes the tarball's runtime source, README, configuration example, license, and shipped documentation public. Model weights, user configuration, credentials, transcripts, session logs, local artifacts, and test fixtures are excluded. Review the archive before the first publication and whenever the package allowlist changes.
|
|
16
20
|
|
|
17
21
|
## What runs automatically
|
|
18
22
|
|
package/package.json
CHANGED
package/src/auth.mjs
CHANGED
|
@@ -1,5 +1,17 @@
|
|
|
1
1
|
export const LOCAL_AUTH_HEADER = 'x-autorouter-token';
|
|
2
2
|
|
|
3
|
+
// Leave Claude's permission selection and policy enforcement to Claude. This
|
|
4
|
+
// only chooses a compatible routing profile for an explicit Auto-mode launch.
|
|
5
|
+
export function clientProfileForLaunch(profile, args) {
|
|
6
|
+
let permissionMode;
|
|
7
|
+
for (let i = 0; i < args.length; i++) {
|
|
8
|
+
if (args[i] === '--') break;
|
|
9
|
+
if (args[i] === '--permission-mode') permissionMode = args[++i];
|
|
10
|
+
else if (args[i].startsWith('--permission-mode=')) permissionMode = args[i].slice('--permission-mode='.length);
|
|
11
|
+
}
|
|
12
|
+
return permissionMode === 'auto' ? 'auto' : profile;
|
|
13
|
+
}
|
|
14
|
+
|
|
3
15
|
export function conflictingProviders(env = process.env) {
|
|
4
16
|
return ['CLAUDE_CODE_USE_BEDROCK', 'CLAUDE_CODE_USE_VERTEX', 'CLAUDE_CODE_USE_FOUNDRY',
|
|
5
17
|
'CLAUDE_CODE_USE_MANTLE', 'CLAUDE_CODE_USE_ANTHROPIC_AWS']
|
|
@@ -30,6 +42,10 @@ export function buildClaudeEnv(config, baseUrl, parent = process.env) {
|
|
|
30
42
|
if (config.clientProfile === 'compatible') {
|
|
31
43
|
env.ANTHROPIC_MODEL = config.models.haiku;
|
|
32
44
|
env.MAX_THINKING_TOKENS = '0';
|
|
45
|
+
} else if (config.clientProfile === 'auto') {
|
|
46
|
+
// Haiku cannot run Auto permission mode. Leave explicit client choices
|
|
47
|
+
// and thinking settings to Claude; it still enforces account/admin gates.
|
|
48
|
+
env.ANTHROPIC_MODEL ??= config.models.sonnet;
|
|
33
49
|
}
|
|
34
50
|
const subscription = config.authMode === 'subscription';
|
|
35
51
|
// Keep unrelated custom headers. Remove stale router credentials and, in
|
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
// Shared execution capabilities, not a replacement for Claude's permission
|
|
2
|
+
// classifier. Keep the complete safeguards contract and signed history on wire.
|
|
3
|
+
const MODELS = new Set(['claude-sonnet-5', 'claude-sonnet-5-5', 'claude-opus-5', 'claude-opus-5-5']);
|
|
4
|
+
const object = value => value !== null && typeof value === 'object' && !Array.isArray(value);
|
|
5
|
+
const SHARED_TOOLS = new Set(['custom', 'tool_search_tool_regex_20251119', 'tool_search_tool_bm25_20251119',
|
|
6
|
+
'bash_20250124', 'text_editor_20250728']);
|
|
7
|
+
const CONTEXT_EDITS = new Set(['clear_thinking_20251015', 'clear_tool_uses_20250919']);
|
|
8
|
+
const sharedTool = tool => object(tool) && (tool.type === undefined || SHARED_TOOLS.has(tool.type));
|
|
9
|
+
|
|
10
|
+
export function hasRoutableSafeguards(body) {
|
|
11
|
+
return MODELS.has(body.model) && Array.isArray(body.safeguards) && body.safeguards.length > 0
|
|
12
|
+
&& body.safeguards.every(entry => object(entry) && entry.type === 'dangerous_tool_use'
|
|
13
|
+
&& object(entry.classifier_context) && entry.classifier_context.v === 1);
|
|
14
|
+
}
|
|
15
|
+
|
|
16
|
+
export function canRouteAutoRequest(body, target) {
|
|
17
|
+
if (!MODELS.has(body.model) || !MODELS.has(target)) return false;
|
|
18
|
+
if (body.safeguards !== undefined && !hasRoutableSafeguards(body)) return false;
|
|
19
|
+
if (body.max_tokens !== undefined && (!Number.isSafeInteger(body.max_tokens) || body.max_tokens < 1 || body.max_tokens > 128000)) return false;
|
|
20
|
+
// Retain model-specific execution facilities whose contracts differ between
|
|
21
|
+
// models. Ordinary Claude Code tools and native context editing are shared.
|
|
22
|
+
if (body.speed !== undefined && body.speed !== 'standard') return false;
|
|
23
|
+
if (body.container !== undefined || body.mcp_servers !== undefined || body.compaction !== undefined) return false;
|
|
24
|
+
if (body.tools !== undefined && (!Array.isArray(body.tools) || !body.tools.every(sharedTool))) return false;
|
|
25
|
+
for (const message of body.messages) {
|
|
26
|
+
if (message.role !== 'system' || !Array.isArray(message.content)) continue;
|
|
27
|
+
for (const block of message.content) {
|
|
28
|
+
if (!['tool_addition', 'tool_removal'].includes(block.type)) continue;
|
|
29
|
+
const tool = block.tool;
|
|
30
|
+
if (!object(tool) || (tool.type !== 'tool_reference'
|
|
31
|
+
&& !(block.type === 'tool_addition' && tool.type === 'tool_definition' && sharedTool(tool.definition)))) return false;
|
|
32
|
+
}
|
|
33
|
+
}
|
|
34
|
+
if (body.tool_choice !== undefined && (!object(body.tool_choice) || !['auto', 'none'].includes(body.tool_choice.type))) return false;
|
|
35
|
+
if (body.context_management !== undefined) {
|
|
36
|
+
const context = body.context_management;
|
|
37
|
+
if (!object(context) || Object.keys(context).some(key => key !== 'edits') || !Array.isArray(context.edits)
|
|
38
|
+
|| context.edits.some(edit => !object(edit) || !CONTEXT_EDITS.has(edit.type))) return false;
|
|
39
|
+
}
|
|
40
|
+
if (body.thinking !== undefined) {
|
|
41
|
+
const thinking = body.thinking;
|
|
42
|
+
if (!object(thinking) || !['adaptive', 'disabled', 'between_tools'].includes(thinking.type)) return false;
|
|
43
|
+
// between_tools is Sonnet 5.5-specific. A routed Opus request uses
|
|
44
|
+
// adaptive thinking; prepareRequest makes that explicit without touching
|
|
45
|
+
// any prior thinking blocks or the conversation prefix they sign.
|
|
46
|
+
if (thinking.type === 'between_tools' && (body.model !== 'claude-sonnet-5-5'
|
|
47
|
+
|| Object.keys(thinking).some(key => key !== 'type') || target === 'claude-sonnet-5')) return false;
|
|
48
|
+
}
|
|
49
|
+
// Sonnet 5 lacks mid-conversation system/tool/effort updates. Never flatten
|
|
50
|
+
// those messages into the top-level prompt: that invalidates signed history.
|
|
51
|
+
if (target === 'claude-sonnet-5' && (body.output_config?.task_budget !== undefined
|
|
52
|
+
|| body.messages.some(message => message.role === 'system' || message.output_config !== undefined))) return false;
|
|
53
|
+
return true;
|
|
54
|
+
}
|
package/src/config.mjs
CHANGED
|
@@ -1,6 +1,14 @@
|
|
|
1
1
|
import { DEFAULT_OLLAMA_MODEL, defaultOllamaTimeoutMs, validateOllamaEndpoint, validateOllamaModel } from './ollama-models.mjs';
|
|
2
|
+
import { resolve } from 'node:path';
|
|
2
3
|
|
|
3
4
|
export const TIERS = ['haiku', 'sonnet', 'opus'];
|
|
5
|
+
export const CLIENT_PROFILES = ['compatible', 'native', 'auto'];
|
|
6
|
+
// Exact Anthropic models whose Auto permission-mode support is documented.
|
|
7
|
+
// Custom aliases are not proof of the capabilities of their upstream model.
|
|
8
|
+
const AUTO_MODE_MODELS = new Set([
|
|
9
|
+
'claude-sonnet-4-6', 'claude-sonnet-5', 'claude-sonnet-5-5',
|
|
10
|
+
'claude-opus-4-6', 'claude-opus-4-7', 'claude-opus-4-8', 'claude-opus-5', 'claude-opus-5-5',
|
|
11
|
+
]);
|
|
4
12
|
|
|
5
13
|
export function parseStopHookBlockCap(value, name = 'CLAUDE_CODE_STOP_HOOK_BLOCK_CAP') {
|
|
6
14
|
if (!['string', 'number'].includes(typeof value)
|
|
@@ -11,6 +19,14 @@ export function parseStopHookBlockCap(value, name = 'CLAUDE_CODE_STOP_HOOK_BLOCK
|
|
|
11
19
|
return Number(value);
|
|
12
20
|
}
|
|
13
21
|
|
|
22
|
+
export function parseSessionLogDir(value, name = 'AUTOROUTER_SESSION_LOG_DIR') {
|
|
23
|
+
if (value === undefined || value === '') return undefined;
|
|
24
|
+
if (typeof value !== 'string' || !value.trim() || /[\u0000-\u001f\u007f]/.test(value)) {
|
|
25
|
+
throw new Error(`${name} must be a directory path, or an empty string to disable session logging`);
|
|
26
|
+
}
|
|
27
|
+
return resolve(value);
|
|
28
|
+
}
|
|
29
|
+
|
|
14
30
|
function number(env, key, fallback, min, max, integer = true) {
|
|
15
31
|
const value = Number(env[key] ?? fallback);
|
|
16
32
|
if (!Number.isFinite(value) || value < min || value > max || (integer && !Number.isInteger(value))) {
|
|
@@ -48,8 +64,20 @@ export function readConfig(env = process.env) {
|
|
|
48
64
|
throw new Error('AUTOROUTER_AUTH_MODE must be api-key or subscription');
|
|
49
65
|
}
|
|
50
66
|
const clientProfile = env.AUTOROUTER_CLIENT_PROFILE ?? 'compatible';
|
|
51
|
-
if (!
|
|
52
|
-
throw new Error('AUTOROUTER_CLIENT_PROFILE must be compatible or
|
|
67
|
+
if (!CLIENT_PROFILES.includes(clientProfile)) {
|
|
68
|
+
throw new Error('AUTOROUTER_CLIENT_PROFILE must be compatible, native or auto');
|
|
69
|
+
}
|
|
70
|
+
const models = {
|
|
71
|
+
haiku: env.AUTOROUTER_HAIKU_MODEL ?? 'claude-haiku-4-5-20251001',
|
|
72
|
+
sonnet: env.AUTOROUTER_SONNET_MODEL ?? (clientProfile === 'auto' ? 'claude-sonnet-5-5' : 'claude-sonnet-5'),
|
|
73
|
+
opus: env.AUTOROUTER_OPUS_MODEL ?? 'claude-opus-5-5',
|
|
74
|
+
};
|
|
75
|
+
if (clientProfile === 'auto') {
|
|
76
|
+
for (const tier of ['sonnet', 'opus']) {
|
|
77
|
+
if (!AUTO_MODE_MODELS.has(models[tier])) {
|
|
78
|
+
throw new Error(`AUTOROUTER_${tier.toUpperCase()}_MODEL must be a known Auto-mode-capable Sonnet or Opus model for the auto profile`);
|
|
79
|
+
}
|
|
80
|
+
}
|
|
53
81
|
}
|
|
54
82
|
const upstream = endpoint(env.AUTOROUTER_UPSTREAM_URL ?? 'https://api.anthropic.com', 'AUTOROUTER_UPSTREAM_URL');
|
|
55
83
|
if (authMode === 'subscription' && upstream !== 'https://api.anthropic.com') {
|
|
@@ -59,6 +87,7 @@ export function readConfig(env = process.env) {
|
|
|
59
87
|
evaluator,
|
|
60
88
|
authMode,
|
|
61
89
|
clientProfile,
|
|
90
|
+
sessionLogDir: parseSessionLogDir(env.AUTOROUTER_SESSION_LOG_DIR),
|
|
62
91
|
stopHookBlockCap: env.CLAUDE_CODE_STOP_HOOK_BLOCK_CAP === undefined
|
|
63
92
|
? undefined : parseStopHookBlockCap(env.CLAUDE_CODE_STOP_HOOK_BLOCK_CAP),
|
|
64
93
|
anthropicKey: authMode === 'api-key' ? env.ANTHROPIC_API_KEY : undefined,
|
|
@@ -72,11 +101,7 @@ export function readConfig(env = process.env) {
|
|
|
72
101
|
ollamaTimeoutMs: number(env, 'AUTOROUTER_OLLAMA_TIMEOUT_MS', defaultOllamaTimeoutMs(ollamaModel), 0, 30000),
|
|
73
102
|
ollamaStateChars: 3000,
|
|
74
103
|
ollamaKeepAlive,
|
|
75
|
-
models
|
|
76
|
-
haiku: env.AUTOROUTER_HAIKU_MODEL ?? 'claude-haiku-4-5-20251001',
|
|
77
|
-
sonnet: env.AUTOROUTER_SONNET_MODEL ?? 'claude-sonnet-5',
|
|
78
|
-
opus: env.AUTOROUTER_OPUS_MODEL ?? 'claude-opus-5-5',
|
|
79
|
-
},
|
|
104
|
+
models,
|
|
80
105
|
port: number(env, 'AUTOROUTER_PORT', 8787, 0, 65535),
|
|
81
106
|
jevTimeoutMs: number(env, 'AUTOROUTER_JEV_TIMEOUT_MS', 1500, 1, 10000),
|
|
82
107
|
tokenCountTimeoutMs: number(env, 'AUTOROUTER_TOKEN_COUNT_TIMEOUT_MS', 1500, 1, 10000),
|
package/src/model-request.mjs
CHANGED
|
@@ -14,6 +14,11 @@ function sonnetNeedsAdaptive(body) {
|
|
|
14
14
|
export function prepareRequest(body, model) {
|
|
15
15
|
const request = { ...body, model };
|
|
16
16
|
const adjustments = [];
|
|
17
|
+
if (model !== body.model && body.model === 'claude-sonnet-5-5' && body.thinking?.type === 'between_tools'
|
|
18
|
+
&& Object.keys(body.thinking).length === 1 && ADAPTIVE_TARGETS.has(model)) {
|
|
19
|
+
request.thinking = { type: 'adaptive' };
|
|
20
|
+
adjustments.push('adaptive_thinking_required');
|
|
21
|
+
}
|
|
17
22
|
if (model !== body.model && body.thinking?.type === 'disabled') {
|
|
18
23
|
if (model === 'claude-sonnet-5-5') {
|
|
19
24
|
const type = sonnetNeedsAdaptive(body) ? 'adaptive' : 'between_tools';
|
package/src/onboarding.mjs
CHANGED
|
@@ -3,7 +3,7 @@ import { existsSync } from 'node:fs';
|
|
|
3
3
|
import { createInterface } from 'node:readline';
|
|
4
4
|
import { Writable } from 'node:stream';
|
|
5
5
|
import { promisify } from 'node:util';
|
|
6
|
-
import { readConfig, requireKeys, parseStopHookBlockCap } from './config.mjs';
|
|
6
|
+
import { CLIENT_PROFILES, readConfig, requireKeys, parseStopHookBlockCap, parseSessionLogDir } from './config.mjs';
|
|
7
7
|
import { buildClaudeEnv, conflictingProviders, LOCAL_AUTH_HEADER } from './auth.mjs';
|
|
8
8
|
import { getConfigPath, loadUserConfig, saveUserConfig } from './user-config.mjs';
|
|
9
9
|
import { DEFAULT_OLLAMA_MODEL, validateOllamaModel } from './ollama-models.mjs';
|
|
@@ -41,14 +41,17 @@ export async function setup(args, {
|
|
|
41
41
|
env = process.env, write = console.log, prompt = askSecret, fetchImpl = fetch, signal,
|
|
42
42
|
} = {}) {
|
|
43
43
|
let authMode = env.AUTOROUTER_AUTH_MODE ?? 'subscription';
|
|
44
|
+
let clientProfile = env.AUTOROUTER_CLIENT_PROFILE ?? 'compatible';
|
|
44
45
|
let evaluator = env.AUTOROUTER_EVALUATOR ?? 'jev';
|
|
45
46
|
let model;
|
|
46
47
|
let ollamaTimeoutMs;
|
|
47
48
|
let stopHookBlockCap = env.CLAUDE_CODE_STOP_HOOK_BLOCK_CAP;
|
|
49
|
+
let sessionLogDir = env.AUTOROUTER_SESSION_LOG_DIR;
|
|
48
50
|
let pull = false;
|
|
49
51
|
let overwrite = false;
|
|
50
52
|
for (let i = 0; i < args.length; i++) {
|
|
51
53
|
if (args[i] === '--auth-mode') authMode = args[++i];
|
|
54
|
+
else if (args[i] === '--client-profile') clientProfile = args[++i];
|
|
52
55
|
else if (args[i] === '--evaluator') evaluator = args[++i];
|
|
53
56
|
else if (args[i] === '--ollama-model') { model = args[++i]; if (model === undefined) throw new Error('--ollama-model requires a model tag'); }
|
|
54
57
|
else if (args[i] === '--ollama-timeout-ms') {
|
|
@@ -61,21 +64,29 @@ export async function setup(args, {
|
|
|
61
64
|
else if (args[i] === '--stop-hook-block-cap') {
|
|
62
65
|
stopHookBlockCap = parseStopHookBlockCap(args[++i], '--stop-hook-block-cap');
|
|
63
66
|
}
|
|
67
|
+
else if (args[i] === '--session-log-dir') {
|
|
68
|
+
sessionLogDir = args[++i];
|
|
69
|
+
if (sessionLogDir === undefined || sessionLogDir.startsWith('--')) throw new Error('--session-log-dir requires a directory path');
|
|
70
|
+
parseSessionLogDir(sessionLogDir, '--session-log-dir');
|
|
71
|
+
}
|
|
64
72
|
else if (args[i] === '--pull') pull = true;
|
|
65
73
|
else if (args[i] === '--force') overwrite = true;
|
|
66
|
-
else throw new Error('Usage: claude-autorouter setup [--auth-mode subscription|api-key] [--evaluator jev|ollama] [--ollama-model TAG] [--ollama-timeout-ms N] [--stop-hook-block-cap N] [--pull] [--force]');
|
|
74
|
+
else throw new Error('Usage: claude-autorouter setup [--auth-mode subscription|api-key] [--client-profile compatible|native|auto] [--evaluator jev|ollama] [--ollama-model TAG] [--ollama-timeout-ms N] [--stop-hook-block-cap N] [--session-log-dir DIR] [--pull] [--force]');
|
|
67
75
|
}
|
|
68
76
|
if (!['subscription', 'api-key'].includes(authMode)) throw new Error('--auth-mode must be subscription or api-key');
|
|
77
|
+
if (!CLIENT_PROFILES.includes(clientProfile)) throw new Error('--client-profile must be compatible, native or auto');
|
|
69
78
|
if (!['jev', 'ollama'].includes(evaluator)) throw new Error('--evaluator must be jev or ollama');
|
|
70
79
|
if (evaluator !== 'ollama' && (model !== undefined || ollamaTimeoutMs !== undefined || pull)) throw new Error('Ollama model, deadline and download options require --evaluator ollama');
|
|
71
80
|
if (stopHookBlockCap !== undefined) stopHookBlockCap = parseStopHookBlockCap(stopHookBlockCap);
|
|
81
|
+
if (sessionLogDir !== undefined) sessionLogDir = parseSessionLogDir(sessionLogDir) ?? '';
|
|
72
82
|
const path = getConfigPath(env);
|
|
73
83
|
if (!overwrite && existsSync(path)) throw new Error('AutoRouter configuration already exists. Use setup --force to replace it.');
|
|
74
84
|
write(evaluator === 'ollama'
|
|
75
85
|
? 'AutoRouter evaluates bounded prompt excerpts locally with Ollama. Complete requests still go to Anthropic.'
|
|
76
86
|
: 'AutoRouter sends bounded prompt excerpts to TypeSafe Jev and complete requests to Anthropic.');
|
|
77
|
-
const values = { AUTOROUTER_AUTH_MODE: authMode, AUTOROUTER_CLIENT_PROFILE:
|
|
87
|
+
const values = { AUTOROUTER_AUTH_MODE: authMode, AUTOROUTER_CLIENT_PROFILE: clientProfile, AUTOROUTER_EVALUATOR: evaluator };
|
|
78
88
|
if (stopHookBlockCap !== undefined) values.CLAUDE_CODE_STOP_HOOK_BLOCK_CAP = String(stopHookBlockCap);
|
|
89
|
+
if (sessionLogDir !== undefined) values.AUTOROUTER_SESSION_LOG_DIR = sessionLogDir;
|
|
79
90
|
if (evaluator === 'ollama') {
|
|
80
91
|
values.AUTOROUTER_OLLAMA_MODEL = validateOllamaModel(model ?? env.AUTOROUTER_OLLAMA_MODEL ?? DEFAULT_OLLAMA_MODEL);
|
|
81
92
|
for (const key of ['AUTOROUTER_OLLAMA_URL', 'AUTOROUTER_OLLAMA_TIMEOUT_MS', 'AUTOROUTER_OLLAMA_KEEP_ALIVE']) {
|
|
@@ -91,7 +102,9 @@ export async function setup(args, {
|
|
|
91
102
|
}
|
|
92
103
|
const config = readConfig(values);
|
|
93
104
|
requireKeys(config);
|
|
105
|
+
if (config.clientProfile === 'auto') write('Auto-compatible profile: Sonnet/Opus task routing. Claude controls permission-mode availability and safety checks.');
|
|
94
106
|
if (config.stopHookBlockCap !== undefined) write(stopHookCapText(config.stopHookBlockCap));
|
|
107
|
+
if (config.sessionLogDir) write('Session decision logs enabled; files include up to 500 characters of user prompt text per decision.');
|
|
95
108
|
if (evaluator === 'ollama') {
|
|
96
109
|
write(`Local evaluator: ${config.ollamaModel}; ${ollamaDeadlineText(config.ollamaTimeoutMs)}.`);
|
|
97
110
|
const controller = new AbortController();
|
|
@@ -127,6 +140,8 @@ export async function doctor({ env = process.env, write = console.log, run = exe
|
|
|
127
140
|
report(false, `Unset ${key}; AutoRouter uses the Anthropic Messages API`);
|
|
128
141
|
}
|
|
129
142
|
if (config?.stopHookBlockCap !== undefined) write(stopHookCapText(config.stopHookBlockCap));
|
|
143
|
+
if (config?.clientProfile === 'auto') write('Auto-compatible profile: Sonnet/Opus task routing. Claude controls permission-mode availability and safety checks.');
|
|
144
|
+
if (config?.sessionLogDir) write('Session decision logs enabled; files include up to 500 characters of user prompt text per decision.');
|
|
130
145
|
if (config?.evaluator === 'ollama') {
|
|
131
146
|
write(`Local evaluator: ${config.ollamaModel}; ${ollamaDeadlineText(config.ollamaTimeoutMs)}.`);
|
|
132
147
|
write('Model availability is checked below; classification speed and accuracy are not tested.');
|
package/src/prompt-state.mjs
CHANGED
|
@@ -122,6 +122,54 @@ export function goalFeedbackIndexes(messages) {
|
|
|
122
122
|
return indexes;
|
|
123
123
|
}
|
|
124
124
|
|
|
125
|
+
// Optional diagnostic logs need only the current human text, not evaluator
|
|
126
|
+
// history or non-text placeholders. Bound collection before joining strings,
|
|
127
|
+
// and never visit tool input/output, attachments, or reasoning payloads.
|
|
128
|
+
export function promptExcerpt(body, maxChars = 500) {
|
|
129
|
+
if (!Number.isSafeInteger(maxChars) || maxChars < 0) throw new TypeError('maxChars must be a nonnegative safe integer');
|
|
130
|
+
const messages = body?.messages;
|
|
131
|
+
if (!maxChars || !Array.isArray(messages)) return '';
|
|
132
|
+
let feedbackIndexes;
|
|
133
|
+
for (let index = messages.length - 1; index >= 0; index--) {
|
|
134
|
+
const message = messages[index];
|
|
135
|
+
if (message?.role !== 'user') continue;
|
|
136
|
+
const content = message.content;
|
|
137
|
+
if (Array.isArray(content) && content.some(block => block?.type === 'tool_result')) continue;
|
|
138
|
+
const standalone = typeof content === 'string' ? content
|
|
139
|
+
: Array.isArray(content) && content.length === 1 && content[0]?.type === 'text'
|
|
140
|
+
&& typeof content[0].text === 'string' ? content[0].text : undefined;
|
|
141
|
+
if (standalone?.startsWith('Stop hook feedback:\n[')) {
|
|
142
|
+
feedbackIndexes ??= goalFeedbackIndexes(messages);
|
|
143
|
+
if (feedbackIndexes.has(index)) continue;
|
|
144
|
+
}
|
|
145
|
+
const blocks = typeof content === 'string' ? [{ type: 'text', text: content }]
|
|
146
|
+
: Array.isArray(content) ? content : [];
|
|
147
|
+
const characters = [];
|
|
148
|
+
let nonText = false;
|
|
149
|
+
for (const block of blocks) {
|
|
150
|
+
if (block?.type !== 'text' || typeof block.text !== 'string') {
|
|
151
|
+
nonText = true;
|
|
152
|
+
continue;
|
|
153
|
+
}
|
|
154
|
+
const value = block.text;
|
|
155
|
+
// Also omit complete wrappers in string messages. Unlike classifier
|
|
156
|
+
// input, a diagnostic excerpt should never log a reminder-only turn.
|
|
157
|
+
if (!/\S/.test(value) || isReminderBlock(value)) continue;
|
|
158
|
+
if (characters.length) characters.push('\n');
|
|
159
|
+
for (const character of value) {
|
|
160
|
+
if (characters.length >= maxChars) break;
|
|
161
|
+
characters.push(character);
|
|
162
|
+
}
|
|
163
|
+
if (characters.length >= maxChars) break;
|
|
164
|
+
}
|
|
165
|
+
if (characters.length) return characters.join('').toWellFormed();
|
|
166
|
+
// An image/document-only human turn is a new task with no safe excerpt;
|
|
167
|
+
// do not incorrectly label it with the preceding human task's text.
|
|
168
|
+
if (nonText) return '';
|
|
169
|
+
}
|
|
170
|
+
return '';
|
|
171
|
+
}
|
|
172
|
+
|
|
125
173
|
export function buildState(body, limit = 12000) {
|
|
126
174
|
const messages = body.messages ?? [];
|
|
127
175
|
const feedbackIndexes = goalFeedbackIndexes(messages);
|
package/src/router.mjs
CHANGED
|
@@ -2,6 +2,7 @@ import { createHash } from 'node:crypto';
|
|
|
2
2
|
import { TIERS } from './config.mjs';
|
|
3
3
|
import { buildState, goalFeedbackIndexes } from './prompt-state.mjs';
|
|
4
4
|
import { buildOllamaState, evaluateOllama } from './ollama-evaluator.mjs';
|
|
5
|
+
import { canRouteAutoRequest, hasRoutableSafeguards } from './auto-routing.mjs';
|
|
5
6
|
export { buildState } from './prompt-state.mjs';
|
|
6
7
|
|
|
7
8
|
const hash = value => createHash('sha256').update(JSON.stringify(value)).digest('hex');
|
|
@@ -226,6 +227,29 @@ export class Router {
|
|
|
226
227
|
async route(body, { scope = '', signal, requestClass = '', promptId = '', countTokens } = {}) {
|
|
227
228
|
const start = performance.now();
|
|
228
229
|
const c = this.config;
|
|
230
|
+
const autoMode = c.clientProfile === 'auto' || hasRoutableSafeguards(body);
|
|
231
|
+
// Auxiliary permission classifiers keep their model and verdicts. Main
|
|
232
|
+
// execution requests can switch between compatible Sonnet/Opus models
|
|
233
|
+
// while retaining the server review contract verbatim. Unknown contracts
|
|
234
|
+
// still pass through, including any future safeguards version.
|
|
235
|
+
if (requestClass === 'auxiliary' || (body.safeguards !== undefined
|
|
236
|
+
&& (!hasRoutableSafeguards(body) || requestClass === 'compaction'))) {
|
|
237
|
+
// A safeguarded main request still produces the next tool turn. Replace
|
|
238
|
+
// any older routing pin with the actual preserved model so a later
|
|
239
|
+
// request that omits safeguards cannot restore that stale model. Side
|
|
240
|
+
// classifiers and compaction never take ownership of the main turn.
|
|
241
|
+
if (!['auxiliary', 'compaction'].includes(requestClass)) {
|
|
242
|
+
const turn = turnInfo(body, scope, promptId);
|
|
243
|
+
if (turn.index >= 0 || promptId) {
|
|
244
|
+
const pin = { model: body.model, requestedModel: body.model };
|
|
245
|
+
this.turns.set(turn.key, pin);
|
|
246
|
+
if (turn.contentKey !== turn.key) this.turns.set(turn.contentKey, pin);
|
|
247
|
+
}
|
|
248
|
+
}
|
|
249
|
+
return { model: body.model, source: 'passthrough',
|
|
250
|
+
reason: requestClass === 'auxiliary' ? 'internal_request' : 'auto_mode_safeguards',
|
|
251
|
+
latency_ms: Math.round((performance.now() - start) * 100) / 100 };
|
|
252
|
+
}
|
|
229
253
|
const hasSystemMessage = body.messages.some(m => m.role === 'system');
|
|
230
254
|
const unknownModel = rank(body.model) < 0 && !Object.values(c.models).includes(body.model);
|
|
231
255
|
const modelSpecificThinking = body.thinking && !['disabled', 'adaptive'].includes(body.thinking.type);
|
|
@@ -243,18 +267,28 @@ export class Router {
|
|
|
243
267
|
// Check suspicious input in parallel with Jev. Byte size only triggers a
|
|
244
268
|
// check: common tool catalogs can be 200KB yet occupy far less than 200K
|
|
245
269
|
// tokens. Tiny requests keep the one-call fast path.
|
|
246
|
-
const earlyCount = !capacityLocked && countTokens && CAPACITY_UPGRADE_MODELS.has(c.models.haiku)
|
|
270
|
+
const earlyCount = !autoMode && !capacityLocked && countTokens && CAPACITY_UPGRADE_MODELS.has(c.models.haiku)
|
|
247
271
|
&& (contextSizeBytes(body, c.models.haiku) > 150000 || hasAttachments)
|
|
248
272
|
? safelyCount(c.models.haiku) : undefined;
|
|
249
273
|
const decision = await this.classify(body, signal);
|
|
250
274
|
let model = c.models[decision.tier];
|
|
251
275
|
let reason = decision.reason;
|
|
276
|
+
// Auto permission mode requires a supported execution model. Retain the
|
|
277
|
+
// evaluator's verdict for observability; stronger compatibility and turn
|
|
278
|
+
// constraints below still decide whether this ordinary choice can apply.
|
|
279
|
+
if (autoMode && decision.tier === 'haiku') {
|
|
280
|
+
model = c.models.sonnet;
|
|
281
|
+
reason = 'auto_mode_floor';
|
|
282
|
+
}
|
|
252
283
|
const turn = turnInfo(body, scope, promptId);
|
|
253
284
|
const promptPin = promptId ? this.turns.get(turn.key) : undefined;
|
|
254
285
|
const turnPin = promptPin ?? this.turns.get(turn.contentKey);
|
|
255
286
|
let previous = turnPin?.model;
|
|
256
287
|
const textTurn = !turn.continuation || turn.goalFeedback;
|
|
257
288
|
const textPin = promptPin ?? (!promptId && turn.goalFeedback ? turnPin : undefined);
|
|
289
|
+
const pinnedTarget = (turn.continuation || (textTurn && textPin?.requestedModel === body.model))
|
|
290
|
+
? previous : undefined;
|
|
291
|
+
const sharedAutoRequest = autoMode && canRouteAutoRequest(body, pinnedTarget ?? model);
|
|
258
292
|
// A new human prompt can still carry signed thinking from the preceding
|
|
259
293
|
// turn. Recover that turn's actual routed model when it is known.
|
|
260
294
|
if (!turn.continuation && body.messages.length > 1 && !previous) {
|
|
@@ -269,23 +303,24 @@ export class Router {
|
|
|
269
303
|
|
|
270
304
|
// A tool result belongs to the model that requested it. Do not bounce the
|
|
271
305
|
// agent between models partway through one human turn.
|
|
272
|
-
if (requestClass === 'compaction'
|
|
306
|
+
if (requestClass === 'compaction') preserve(body.model, 'internal_request');
|
|
273
307
|
// Mid-conversation system messages are only supported by certain models.
|
|
274
308
|
// Keep the client's capable model and all message fields (including
|
|
275
309
|
// clear_at, tool changes, and output_config) instead of down-routing.
|
|
276
|
-
else if (hasSystemMessage) preserve(body.model, 'mid_conversation_system');
|
|
310
|
+
else if (hasSystemMessage && !sharedAutoRequest) preserve(body.model, 'mid_conversation_system');
|
|
277
311
|
else if (unknownModel) preserve(body.model, 'unknown_model');
|
|
278
312
|
// A new native request can explicitly select a model-specific thinking
|
|
279
313
|
// mode, including between_tools. An earlier turn's model is not evidence
|
|
280
314
|
// that it accepts that mode. Existing tool turns retain their pin below.
|
|
281
|
-
else if (modelSpecificThinking && body.thinking.type !== 'enabled' && textTurn) preserve(body.model, 'model_specific_features');
|
|
315
|
+
else if (modelSpecificThinking && body.thinking.type !== 'enabled' && textTurn && !sharedAutoRequest) preserve(body.model, 'model_specific_features');
|
|
282
316
|
// Stop hooks (including /goal) return feedback as user-role text, even
|
|
283
317
|
// though it still serves the same human prompt. Trust the scoped gateway
|
|
284
318
|
// identity instead of treating that text as a new task. A client model
|
|
285
319
|
// change can be an explicit fallback after a failure; do not undo it.
|
|
286
320
|
// Local /goal commands can omit the gateway prompt ID. Exact feedback for
|
|
287
321
|
// a known goal then uses the original conversation anchor as a fallback.
|
|
288
|
-
else if (textTurn && textPin?.requestedModel === body.model && !modelSpecificFeatures
|
|
322
|
+
else if (textTurn && textPin?.requestedModel === body.model && (!modelSpecificFeatures
|
|
323
|
+
|| (autoMode && canRouteAutoRequest(body, textPin.model)))) {
|
|
289
324
|
const needsSonnet = body.thinking?.type === 'adaptive' || body.output_config?.effort || body.max_tokens > 64000;
|
|
290
325
|
if (needsSonnet && (textPin.model === c.models.haiku || rank(textPin.model) === 0)) {
|
|
291
326
|
preserve(c.models.sonnet, 'requires_sonnet_capabilities');
|
|
@@ -295,10 +330,10 @@ export class Router {
|
|
|
295
330
|
else if (turn.continuation && !turn.goalFeedback) keep(previous ? 'tool_turn_pinned' : 'unknown_continuation');
|
|
296
331
|
// Unknown or model-specific features are preserved, never silently removed.
|
|
297
332
|
else if (decision.source === 'fallback' && rank(body.model) >= 1) keep('classifier_unavailable');
|
|
298
|
-
else if (modelSpecificFeatures) keep('model_specific_features');
|
|
299
|
-
else if (thinkingHistory) keep('thinking_history');
|
|
333
|
+
else if (modelSpecificFeatures && !sharedAutoRequest) keep('model_specific_features');
|
|
334
|
+
else if (thinkingHistory && !sharedAutoRequest) keep('thinking_history');
|
|
300
335
|
else if (body.thinking?.type === 'adaptive' || body.output_config?.effort || body.max_tokens > 64000) {
|
|
301
|
-
if (decision.tier === 'haiku') { model = c.models.sonnet; reason = 'requires_sonnet_capabilities'; }
|
|
336
|
+
if (decision.tier === 'haiku') { model = c.models.sonnet; if (!autoMode) reason = 'requires_sonnet_capabilities'; }
|
|
302
337
|
}
|
|
303
338
|
// Account for all context, including system instructions and loaded tool
|
|
304
339
|
// schemas that are intentionally omitted from Jev's bounded excerpt.
|
|
@@ -317,7 +352,10 @@ export class Router {
|
|
|
317
352
|
const configured = TIERS.findIndex(tier => c.models[tier] === value);
|
|
318
353
|
return configured >= 0 ? configured : rank(value);
|
|
319
354
|
};
|
|
320
|
-
|
|
355
|
+
// The verified modern Auto pair shares a native 1M input window. A large
|
|
356
|
+
// prompt is not a reason to pin Opus forever after the task becomes easy.
|
|
357
|
+
// This does not assert that the prompt fits the upstream context limit.
|
|
358
|
+
if (!preserved && largeContext && !sharedAutoRequest) {
|
|
321
359
|
const baseline = previous ?? body.model;
|
|
322
360
|
// Prevent a downgrade; a compatible Haiku client must still be able to
|
|
323
361
|
// upgrade a demanding request to a larger-context, stronger model.
|
|
@@ -335,8 +373,11 @@ export class Router {
|
|
|
335
373
|
capacityUpgraded = true;
|
|
336
374
|
}
|
|
337
375
|
}
|
|
376
|
+
if (autoMode && model !== body.model && !canRouteAutoRequest(body, model)) {
|
|
377
|
+
preserve(body.model, 'auto_mode_incompatible');
|
|
378
|
+
}
|
|
338
379
|
const identifiableUpgrade = capacityUpgraded && (turn.index >= 0 || promptId);
|
|
339
|
-
if ((!turn.continuation || previous || identifiableUpgrade || reason === 'mid_conversation_system') &&
|
|
380
|
+
if ((!turn.continuation || previous || identifiableUpgrade || reason === 'mid_conversation_system') && requestClass !== 'compaction') {
|
|
340
381
|
const pin = { model, requestedModel: body.model };
|
|
341
382
|
this.turns.set(turn.key, pin);
|
|
342
383
|
// Keep the content key too: later human turns carry signed thinking but
|
package/src/server.mjs
CHANGED
|
@@ -7,6 +7,7 @@ import { LOCAL_AUTH_HEADER, isSubscriptionRequest } from './auth.mjs';
|
|
|
7
7
|
import { prepareRequest } from './model-request.mjs';
|
|
8
8
|
import { createResponseObserver } from './response-observer.mjs';
|
|
9
9
|
import { createTokenCounter } from './token-counter.mjs';
|
|
10
|
+
import { promptExcerpt } from './prompt-state.mjs';
|
|
10
11
|
|
|
11
12
|
function cleanHeaders(headers) {
|
|
12
13
|
const blocked = new Set(['host', 'connection', 'keep-alive', 'proxy-authenticate', 'proxy-authorization', 'te', 'trailer', 'transfer-encoding', 'upgrade', 'content-length']);
|
|
@@ -101,7 +102,7 @@ async function forward(url, req, res, body, config, signal, log, status) {
|
|
|
101
102
|
}
|
|
102
103
|
}
|
|
103
104
|
|
|
104
|
-
export function createRouterServer(config, { router = new Router(config), tokenCounter = createTokenCounter(config), log = entry => process.stderr.write(`${JSON.stringify(entry)}\n`), onStatus = () => {} } = {}) {
|
|
105
|
+
export function createRouterServer(config, { router = new Router(config), tokenCounter = createTokenCounter(config), log = entry => process.stderr.write(`${JSON.stringify(entry)}\n`), onStatus = () => {}, onDecision } = {}) {
|
|
105
106
|
if (!config.localToken || config.localToken.length < 16) throw new Error('AUTOROUTER_TOKEN must contain at least 16 characters');
|
|
106
107
|
const server = http.createServer(async (req, res) => {
|
|
107
108
|
const controller = new AbortController();
|
|
@@ -175,6 +176,21 @@ export function createRouterServer(config, { router = new Router(config), tokenC
|
|
|
175
176
|
if (controller.signal.aborted) { status('request_cancelled'); return; }
|
|
176
177
|
const prepared = prepareRequest(parsed, decision.model);
|
|
177
178
|
body = Buffer.from(JSON.stringify(prepared.request));
|
|
179
|
+
// Prompt excerpts go only to this explicit opt-in sink, never to
|
|
180
|
+
// ordinary diagnostics or the status snapshot. Optional logging
|
|
181
|
+
// cannot delay or fail forwarding, including an async sink failure.
|
|
182
|
+
if (onDecision) {
|
|
183
|
+
try {
|
|
184
|
+
const chars = [...(!context.request_class || context.request_class === 'main' ? promptExcerpt(parsed, 501) : '')];
|
|
185
|
+
Promise.resolve(onDecision({
|
|
186
|
+
schema_version: 1, event: 'decision', timestamp: new Date().toISOString(), ...context,
|
|
187
|
+
prompt_excerpt: chars.slice(0, 500).join(''), prompt_truncated: chars.length > 500,
|
|
188
|
+
requested_model: parsed.model, selected_model: decision.model, decision_latency_ms: decision.latency_ms,
|
|
189
|
+
source: decision.source, reason: decision.reason, evaluator: decision.evaluator,
|
|
190
|
+
classified_tier: decision.classified_tier, classifier_error: decision.classifier_error,
|
|
191
|
+
})).catch(() => {});
|
|
192
|
+
} catch {}
|
|
193
|
+
}
|
|
178
194
|
log({ event: 'route', requested_model: parsed.model, ...decision, request_adjustments: prepared.adjustments });
|
|
179
195
|
const { model, source, evaluator, reason, latency_ms, classifier_error, classifier_status, classified_tier, context_check, counted_input_tokens } = decision;
|
|
180
196
|
const pricingValue = (field, allowed, fallback) => prepared.request[field] === undefined ? fallback
|
|
@@ -0,0 +1,171 @@
|
|
|
1
|
+
import { constants } from 'node:fs';
|
|
2
|
+
import fs from 'node:fs/promises';
|
|
3
|
+
import { createHash, randomBytes } from 'node:crypto';
|
|
4
|
+
import { join, resolve } from 'node:path';
|
|
5
|
+
|
|
6
|
+
const MAX_PENDING_BYTES = 1024 * 1024;
|
|
7
|
+
const MAX_SESSIONS = 128;
|
|
8
|
+
const WARNING = 'AutoRouter session logging disabled.';
|
|
9
|
+
const SOURCES = new Set(['jev', 'ollama', 'cache', 'fallback', 'passthrough']);
|
|
10
|
+
const ERRORS = new Set(['timeout', 'http_error', 'invalid_response', 'network_error']);
|
|
11
|
+
const identifier = value => typeof value === 'string' && /^[A-Za-z0-9_.:-]{1,200}$/.test(value) ? value : undefined;
|
|
12
|
+
const model = value => typeof value === 'string' && /^[A-Za-z0-9_.:/-]{1,120}$/.test(value) ? value : undefined;
|
|
13
|
+
const code = value => typeof value === 'string' && /^[a-z][a-z0-9_]{0,79}$/.test(value) ? value : undefined;
|
|
14
|
+
|
|
15
|
+
function excerpt(value) {
|
|
16
|
+
if (typeof value !== 'string') return { text: '', truncated: false };
|
|
17
|
+
let text = '', count = 0;
|
|
18
|
+
for (const character of value) {
|
|
19
|
+
if (count++ === 500) return { text: text.toWellFormed(), truncated: true };
|
|
20
|
+
text += character;
|
|
21
|
+
}
|
|
22
|
+
return { text: text.toWellFormed(), truncated: false };
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
function normalize(entry) {
|
|
26
|
+
if (!entry || typeof entry !== 'object' || Array.isArray(entry) || entry.event !== 'decision') return;
|
|
27
|
+
const requestId = identifier(entry.request_id);
|
|
28
|
+
const selectedModel = model(entry.selected_model);
|
|
29
|
+
const requestedModel = model(entry.requested_model);
|
|
30
|
+
if (!requestId || !selectedModel || !requestedModel) return;
|
|
31
|
+
const anonymous = entry.session_id === undefined || entry.session_id === null || entry.session_id === '';
|
|
32
|
+
const sessionId = anonymous ? undefined : identifier(entry.session_id);
|
|
33
|
+
// An invalid explicit identity must not mix records into an anonymous file.
|
|
34
|
+
if (!anonymous && !sessionId) return;
|
|
35
|
+
const requestClass = code(entry.request_class);
|
|
36
|
+
const foreground = entry.request_class === undefined || entry.request_class === null || entry.request_class === '' || entry.request_class === 'main';
|
|
37
|
+
const prompt = foreground ? excerpt(entry.prompt_excerpt) : { text: '', truncated: false };
|
|
38
|
+
const timestamp = typeof entry.timestamp === 'string' && /^\d{4}-\d\d-\d\dT\d\d:\d\d:\d\d\.\d{3}Z$/.test(entry.timestamp)
|
|
39
|
+
&& Number.isFinite(Date.parse(entry.timestamp)) ? entry.timestamp : new Date().toISOString();
|
|
40
|
+
const row = { schema_version: 1, event: 'decision', timestamp, request_id: requestId,
|
|
41
|
+
...(sessionId ? { session_id: sessionId } : {}),
|
|
42
|
+
prompt_excerpt: prompt.text, prompt_truncated: prompt.truncated || (foreground && entry.prompt_truncated === true),
|
|
43
|
+
requested_model: requestedModel, selected_model: selectedModel };
|
|
44
|
+
for (const field of ['agent_id', 'prompt_id']) {
|
|
45
|
+
const value = identifier(entry[field]);
|
|
46
|
+
if (value) row[field] = value;
|
|
47
|
+
}
|
|
48
|
+
if (requestClass) row.request_class = requestClass;
|
|
49
|
+
if (typeof entry.decision_latency_ms === 'number' && Number.isFinite(entry.decision_latency_ms)
|
|
50
|
+
&& entry.decision_latency_ms >= 0) row.decision_latency_ms = entry.decision_latency_ms;
|
|
51
|
+
if (SOURCES.has(entry.source)) row.source = entry.source;
|
|
52
|
+
const reason = code(entry.reason);
|
|
53
|
+
if (reason) row.reason = reason;
|
|
54
|
+
if (['jev', 'ollama'].includes(entry.evaluator)) row.evaluator = entry.evaluator;
|
|
55
|
+
if (['haiku', 'sonnet', 'opus'].includes(entry.classified_tier)) row.classified_tier = entry.classified_tier;
|
|
56
|
+
if (ERRORS.has(entry.classifier_error)) row.classifier_error = entry.classifier_error;
|
|
57
|
+
return { sessionKey: sessionId ? `session:${sessionId}` : 'anonymous', row };
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
// Inference never waits on this writer. Each launch creates new files; only
|
|
61
|
+
// accepted, bounded JSON lines are retained until the serialized writer drains.
|
|
62
|
+
export async function createSessionLog(directory, { warn = () => {} } = {}) {
|
|
63
|
+
let accepting = true, failed = false, warned = false, pendingBytes = 0;
|
|
64
|
+
let root, directoryIdentity, pump, closePromise;
|
|
65
|
+
const sessions = new Map(), queue = [];
|
|
66
|
+
const launch = `${new Date().toISOString().replace(/[-:.]/g, '')}-${randomBytes(12).toString('hex')}`;
|
|
67
|
+
const disable = () => {
|
|
68
|
+
accepting = false;
|
|
69
|
+
if (warned) return;
|
|
70
|
+
warned = true;
|
|
71
|
+
try { Promise.resolve(warn(WARNING)).catch(() => {}); } catch {}
|
|
72
|
+
};
|
|
73
|
+
async function checkDirectory() {
|
|
74
|
+
const current = await fs.lstat(root);
|
|
75
|
+
if (!current.isDirectory() || current.isSymbolicLink()
|
|
76
|
+
|| (directoryIdentity && (current.dev !== directoryIdentity.dev || current.ino !== directoryIdentity.ino))) {
|
|
77
|
+
throw new Error('Invalid session log directory');
|
|
78
|
+
}
|
|
79
|
+
return current;
|
|
80
|
+
}
|
|
81
|
+
try {
|
|
82
|
+
if (typeof directory !== 'string' || !directory.trim() || typeof constants.O_NOFOLLOW !== 'number') throw new Error('Invalid session log directory');
|
|
83
|
+
root = resolve(directory);
|
|
84
|
+
try { await checkDirectory(); }
|
|
85
|
+
catch (error) { if (error.code !== 'ENOENT') throw error; }
|
|
86
|
+
await fs.mkdir(root, { recursive: true, mode: 0o700 });
|
|
87
|
+
directoryIdentity = await checkDirectory();
|
|
88
|
+
} catch { failed = true; disable(); }
|
|
89
|
+
|
|
90
|
+
async function handleFor(session) {
|
|
91
|
+
if (session.handle) return session.handle;
|
|
92
|
+
await checkDirectory();
|
|
93
|
+
const handle = await fs.open(session.path, constants.O_WRONLY | constants.O_APPEND | constants.O_CREAT | constants.O_EXCL | constants.O_NOFOLLOW, 0o600);
|
|
94
|
+
// Track immediately, including when a subsequent check fails, so shutdown
|
|
95
|
+
// always closes the descriptor. Never reopen or follow an existing path.
|
|
96
|
+
session.handle = handle;
|
|
97
|
+
const stat = await handle.stat();
|
|
98
|
+
if (!stat.isFile() || stat.nlink !== 1) throw new Error('Invalid session log file');
|
|
99
|
+
await handle.chmod(0o600);
|
|
100
|
+
await checkDirectory();
|
|
101
|
+
return handle;
|
|
102
|
+
}
|
|
103
|
+
async function drain() {
|
|
104
|
+
let index = 0;
|
|
105
|
+
try {
|
|
106
|
+
while (index < queue.length && !failed) {
|
|
107
|
+
const item = queue[index];
|
|
108
|
+
queue[index++] = undefined;
|
|
109
|
+
const handle = await handleFor(item.session);
|
|
110
|
+
await handle.writeFile(item.line);
|
|
111
|
+
pendingBytes -= item.bytes;
|
|
112
|
+
// A steady producer can keep the queue nonempty indefinitely. Bound
|
|
113
|
+
// processed slots as well as the pending strings they once held.
|
|
114
|
+
if (index >= 256) { queue.splice(0, index); index = 0; }
|
|
115
|
+
}
|
|
116
|
+
} catch {
|
|
117
|
+
failed = true;
|
|
118
|
+
disable();
|
|
119
|
+
} finally {
|
|
120
|
+
queue.length = 0;
|
|
121
|
+
pendingBytes = 0;
|
|
122
|
+
}
|
|
123
|
+
}
|
|
124
|
+
function schedule() {
|
|
125
|
+
if (pump || failed || !queue.length) return;
|
|
126
|
+
pump = Promise.resolve().then(drain).finally(() => {
|
|
127
|
+
pump = undefined;
|
|
128
|
+
// A record can arrive between drain resolving and this continuation.
|
|
129
|
+
// Keep it scheduled even if shutdown has already stopped new records.
|
|
130
|
+
schedule();
|
|
131
|
+
});
|
|
132
|
+
}
|
|
133
|
+
function record(entry) {
|
|
134
|
+
if (!accepting) return false;
|
|
135
|
+
try {
|
|
136
|
+
const normalized = normalize(entry);
|
|
137
|
+
if (!normalized) return false;
|
|
138
|
+
const line = `${JSON.stringify(normalized.row)}\n`;
|
|
139
|
+
const bytes = Buffer.byteLength(line);
|
|
140
|
+
let session = sessions.get(normalized.sessionKey);
|
|
141
|
+
if (pendingBytes + bytes > MAX_PENDING_BYTES || (!session && sessions.size >= MAX_SESSIONS)) {
|
|
142
|
+
disable();
|
|
143
|
+
return false;
|
|
144
|
+
}
|
|
145
|
+
if (!session) {
|
|
146
|
+
const sessionHash = createHash('sha256').update(normalized.sessionKey).digest('hex');
|
|
147
|
+
session = { path: join(root, `autorouter-session-${launch}-${sessionHash}.jsonl`) };
|
|
148
|
+
sessions.set(normalized.sessionKey, session);
|
|
149
|
+
}
|
|
150
|
+
queue.push({ session, line, bytes });
|
|
151
|
+
pendingBytes += bytes;
|
|
152
|
+
schedule();
|
|
153
|
+
return true;
|
|
154
|
+
} catch { return false; }
|
|
155
|
+
}
|
|
156
|
+
function close() {
|
|
157
|
+
if (!closePromise) {
|
|
158
|
+
accepting = false;
|
|
159
|
+
closePromise = (async () => {
|
|
160
|
+
while (pump) await pump;
|
|
161
|
+
for (const session of sessions.values()) {
|
|
162
|
+
if (!session.handle) continue;
|
|
163
|
+
try { await session.handle.close(); } catch { disable(); }
|
|
164
|
+
session.handle = undefined;
|
|
165
|
+
}
|
|
166
|
+
})().catch(() => { disable(); });
|
|
167
|
+
}
|
|
168
|
+
return closePromise;
|
|
169
|
+
}
|
|
170
|
+
return { record, close };
|
|
171
|
+
}
|
package/src/status-state.mjs
CHANGED
|
@@ -102,7 +102,7 @@ export function createStatusState(options = {}) {
|
|
|
102
102
|
switch (event.event) {
|
|
103
103
|
case 'route': {
|
|
104
104
|
const fields = { requested_model: modelName(event.requested_model), selected_model: modelName(event.model),
|
|
105
|
-
source: ['jev', 'ollama', 'cache', 'fallback'].includes(event.source) ? event.source : undefined,
|
|
105
|
+
source: ['jev', 'ollama', 'cache', 'fallback', 'passthrough'].includes(event.source) ? event.source : undefined,
|
|
106
106
|
evaluator: ['jev', 'ollama'].includes(event.evaluator) ? event.evaluator : undefined,
|
|
107
107
|
reason: code(event.reason), latency_ms: latency(event.latency_ms),
|
|
108
108
|
classified_tier: ['haiku', 'sonnet', 'opus'].includes(event.classified_tier) ? event.classified_tier : undefined,
|
package/src/statusline.mjs
CHANGED
|
@@ -6,6 +6,7 @@ const REASONS = {
|
|
|
6
6
|
model_specific_features: 'model features', large_or_multimodal_request: 'large request',
|
|
7
7
|
context_capacity: 'large context',
|
|
8
8
|
internal_request: 'internal request', unknown_model: 'custom model', low_confidence: 'low confidence',
|
|
9
|
+
auto_mode_floor: 'Auto mode floor', auto_mode_safeguards: 'Auto safety', auto_mode_incompatible: 'Auto model guard',
|
|
9
10
|
};
|
|
10
11
|
const CLASSIFIER_ERRORS = {
|
|
11
12
|
timeout: 'timeout', http_error: 'HTTP error', invalid_response: 'invalid response', network_error: 'network error',
|
|
@@ -128,14 +129,14 @@ export function renderStatusLine(input, snapshot, { now = Date.now(), color = tr
|
|
|
128
129
|
? `error ${state.status}` : errorType ? `error ${errorType}` : 'error';
|
|
129
130
|
|
|
130
131
|
const details = [];
|
|
131
|
-
const source = ['jev', 'ollama', 'cache', 'fallback'].includes(state?.source) ? state.source : undefined;
|
|
132
|
+
const source = ['jev', 'ollama', 'cache', 'fallback', 'passthrough'].includes(state?.source) ? state.source : undefined;
|
|
132
133
|
const evaluator = ['jev', 'ollama'].includes(state?.evaluator) ? state.evaluator : undefined;
|
|
133
134
|
const evaluatorLabel = evaluator === 'ollama' ? 'Ollama' : evaluator === 'jev' ? 'Jev' : '';
|
|
134
135
|
let fallbackCause = '';
|
|
135
136
|
let fallbackPhase = '';
|
|
136
137
|
let compactFallback = false;
|
|
137
138
|
if (source) {
|
|
138
|
-
const sourceLabel = source === 'jev' ? 'Jev' : source === 'ollama' ? 'Ollama'
|
|
139
|
+
const sourceLabel = source === 'passthrough' ? 'pass-through' : source === 'jev' ? 'Jev' : source === 'ollama' ? 'Ollama'
|
|
139
140
|
: evaluatorLabel ? `${evaluatorLabel} ${source}` : source;
|
|
140
141
|
const timing = Number.isFinite(state.latency_ms) && state.latency_ms >= 0 ? ` ${Math.round(Math.min(state.latency_ms, 999999))}ms` : '';
|
|
141
142
|
const classified = ['haiku', 'sonnet', 'opus'].includes(state.classified_tier) ? state.classified_tier : undefined;
|
package/src/user-config.mjs
CHANGED
|
@@ -13,6 +13,7 @@ const CONFIG_KEYS = new Set([
|
|
|
13
13
|
'AUTOROUTER_HAIKU_MODEL', 'AUTOROUTER_SONNET_MODEL', 'AUTOROUTER_OPUS_MODEL',
|
|
14
14
|
'AUTOROUTER_PORT', 'AUTOROUTER_JEV_TIMEOUT_MS', 'AUTOROUTER_TOKEN_COUNT_TIMEOUT_MS',
|
|
15
15
|
'AUTOROUTER_MIN_CONFIDENCE', 'AUTOROUTER_STATUSLINE', 'AUTOROUTER_DEBUG',
|
|
16
|
+
'AUTOROUTER_SESSION_LOG_DIR',
|
|
16
17
|
'ENABLE_TOOL_SEARCH', 'CLAUDE_CODE_STOP_HOOK_BLOCK_CAP',
|
|
17
18
|
'AUTOROUTER_EVALUATOR', 'AUTOROUTER_OLLAMA_URL', 'AUTOROUTER_OLLAMA_MODEL',
|
|
18
19
|
'AUTOROUTER_OLLAMA_TIMEOUT_MS', 'AUTOROUTER_OLLAMA_KEEP_ALIVE',
|