claude-autorouter 0.3.3 → 0.3.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +4 -0
- package/README.md +1 -0
- package/bin/autorouter.mjs +4 -0
- package/docs/reference.md +28 -1
- package/docs/releasing.md +4 -0
- package/package.json +1 -1
- package/src/auth.mjs +5 -0
- package/src/config.mjs +11 -0
- package/src/onboarding.mjs +14 -2
- package/src/prompt-state.mjs +48 -0
- package/src/router.mjs +28 -8
- package/src/statusline.mjs +2 -1
- package/src/user-config.mjs +1 -1
package/.env.example
CHANGED
|
@@ -7,6 +7,10 @@ AUTOROUTER_EVALUATOR=jev
|
|
|
7
7
|
AUTOROUTER_STATUSLINE=1
|
|
8
8
|
# Optional metadata logs on stderr. Redirect stderr to a file when using the UI.
|
|
9
9
|
# AUTOROUTER_DEBUG=1
|
|
10
|
+
# Optional: allow two tool-free Stop-hook continuations, then end the turn on
|
|
11
|
+
# the third block. Applies to /goal and all Stop/SubagentStop hooks.
|
|
12
|
+
# Unset keeps Claude's default (currently 8); 0 DISABLES the cap.
|
|
13
|
+
# CLAUDE_CODE_STOP_HOOK_BLOCK_CAP=2
|
|
10
14
|
# Required for Jev only; subscription + Ollama needs no API keys.
|
|
11
15
|
TYPESAFE_API_KEY=
|
|
12
16
|
# For API billing instead, set AUTOROUTER_AUTH_MODE=api-key and fill this in.
|
package/README.md
CHANGED
|
@@ -104,5 +104,6 @@ Historical measurements before 0.3.2, on a 16 GiB M4: Tev1 0.8B matched 18/24 he
|
|
|
104
104
|
- The selected evaluator receives bounded excerpts that can contain source code and tool results: TypeSafe with Jev, or the local service with Ollama. Jev also receives system-text excerpts; the local path excludes Claude's executor system instructions. Anthropic receives the complete request. Images, document payloads, and private thinking are omitted from classifier input. [Data flow and authentication](docs/reference.md#data-flow-and-authentication).
|
|
105
105
|
- Subscription access and usage limits still apply. Model switches can reduce cache reuse; cheaper token prices do not guarantee cheaper completed tasks. Run ordinary `claude` to bypass routing.
|
|
106
106
|
- The launcher is quiet by default. Use `AUTOROUTER_DEBUG=1` for metadata diagnostics or `AUTOROUTER_STATUSLINE=0` to retain your existing status line. [Troubleshooting](docs/reference.md#troubleshooting).
|
|
107
|
+
- For blocked `/goal` loops, optionally launch with `env CLAUDE_CODE_STOP_HOOK_BLOCK_CAP=2 claude-autorouter claude`. Claude then ends the turn on the third consecutive blocking verdict without tool use, leaving the goal unmet. This also affects other Stop/SubagentStop hooks; defaults are unchanged. [Scope and saved configuration](docs/reference.md#shorter-stop-hook-loops-opt-in).
|
|
107
108
|
|
|
108
109
|
[Reference](docs/reference.md) · [Development and validation](docs/development.md) · [CI and npm release setup](docs/releasing.md) · [Apache-2.0 license](LICENSE)
|
package/bin/autorouter.mjs
CHANGED
|
@@ -23,6 +23,7 @@ Usage:
|
|
|
23
23
|
claude-autorouter setup [--auth-mode subscription|api-key] [--force]
|
|
24
24
|
[--evaluator jev|ollama]
|
|
25
25
|
[--ollama-model MODEL] [--ollama-timeout-ms N] [--pull]
|
|
26
|
+
[--stop-hook-block-cap N]
|
|
26
27
|
claude-autorouter doctor
|
|
27
28
|
claude-autorouter claude [Claude Code arguments]
|
|
28
29
|
claude-autorouter serve
|
|
@@ -47,6 +48,9 @@ AUTOROUTER_AUTH_MODE=subscription uses your saved Claude Code login.
|
|
|
47
48
|
Without setup, AUTOROUTER_AUTH_MODE defaults to api-key and also requires ANTHROPIC_API_KEY.
|
|
48
49
|
AUTOROUTER_CLIENT_PROFILE=compatible (default) enables all three routing tiers.
|
|
49
50
|
Use AUTOROUTER_CLIENT_PROFILE=native to retain Claude Code's own model/thinking settings.
|
|
51
|
+
Optional CLAUDE_CODE_STOP_HOOK_BLOCK_CAP=N limits consecutive tool-free Stop-hook continuations.
|
|
52
|
+
Use 2 to stop on the third block; applies to /goal and all Stop/SubagentStop hooks.
|
|
53
|
+
Unset preserves Claude's default; 0 disables the cap. Setup --stop-hook-block-cap N saves it.
|
|
50
54
|
Standalone serve also requires AUTOROUTER_TOKEN (at least 16 characters).
|
|
51
55
|
The claude launcher creates a temporary credential and an ephemeral port.
|
|
52
56
|
It enables an AutoRouter status line for this session (AUTOROUTER_STATUSLINE=0 to opt out).
|
package/docs/reference.md
CHANGED
|
@@ -6,6 +6,7 @@
|
|
|
6
6
|
| --- | --- |
|
|
7
7
|
| `claude-autorouter setup` | Save subscription-mode configuration and a Jev key |
|
|
8
8
|
| `claude-autorouter setup --auth-mode api-key` | Configure Jev and Anthropic API-key billing |
|
|
9
|
+
| `claude-autorouter setup --stop-hook-block-cap 2` | Opt into a shorter native Stop-hook continuation cap during setup |
|
|
9
10
|
| `claude-autorouter setup --evaluator ollama --pull` | Configure the native local evaluator and download its selected model if missing |
|
|
10
11
|
| `claude-autorouter setup --evaluator ollama --ollama-timeout-ms 0 --force` | Save a disabled runtime evaluator deadline |
|
|
11
12
|
| `claude-autorouter setup --force` | Replace an existing user config |
|
|
@@ -45,6 +46,7 @@ For an environment-only subscription launch, set `AUTOROUTER_AUTH_MODE=subscript
|
|
|
45
46
|
| `AUTOROUTER_CLIENT_PROFILE` | `compatible` | `native` retains Claude's own model and thinking settings |
|
|
46
47
|
| `AUTOROUTER_STATUSLINE` | enabled | `0` retains your existing status line |
|
|
47
48
|
| `AUTOROUTER_DEBUG` | off | `1` enables launcher metadata logs on stderr |
|
|
49
|
+
| `CLAUDE_CODE_STOP_HOOK_BLOCK_CAP` | unset; Claude currently uses `8` | Optional cap on consecutive Stop/SubagentStop continuations without tool use; `0` disables the cap |
|
|
48
50
|
| `ENABLE_TOOL_SEARCH` | `true` in launcher when unset | Load MCP tool definitions on demand; explicit values are preserved |
|
|
49
51
|
| `AUTOROUTER_HAIKU_MODEL` | `claude-haiku-4-5-20251001` | Routine tier |
|
|
50
52
|
| `AUTOROUTER_SONNET_MODEL` | `claude-sonnet-5` | Standard tier |
|
|
@@ -166,7 +168,8 @@ The following policy applies after classification:
|
|
|
166
168
|
|
|
167
169
|
- Jev's 1,500 ms deadline covers the response body and has no retry. Successful calls return immediately. Timeouts, HTTP errors, and invalid responses fall back to Sonnet or retain an existing stronger model.
|
|
168
170
|
- Jev confidence below 0.75 prevents a downgrade below Sonnet or the requested tier. Ollama returns a tier without calibrated confidence; its failure handling and compatibility guards still apply.
|
|
169
|
-
- Tool continuations retain the model chosen at the start of the human turn. Session, agent, and prompt headers identify turns; normalized conversation content provides a fallback. Moving prompt-cache markers does not create a new turn.
|
|
171
|
+
- Tool continuations retain the model chosen at the start of the human turn. Session, agent, and prompt headers identify turns; normalized conversation content provides a fallback. Text feedback from a Stop hook also retains the model when it serves the same gateway prompt ID and the client has not changed its requested model, subject to capability and context checks. Moving prompt-cache markers does not create a new turn.
|
|
172
|
+
- Claude's local `/goal` command can omit the prompt-ID header. For that path, an exact feedback label matching a preceding expanded `/goal` command keeps the original task and conversation anchor. This narrow text fallback also recognizes Claude's repeated-goal truncation format; arbitrary hook text is not treated as a goal. Feedback remains in the evaluator's recent conversation and the full API request. A new human message becomes the current task normally. The status line shows `prompt pinned` or `goal pinned` when either text-continuation rule applies.
|
|
170
173
|
- Thinking history, fixed-budget thinking, server tools, context management, and other recognized model-specific features preserve the current model. Adaptive thinking, effort, and output above 64K prevent a Haiku choice. Fields are never stripped to force a downgrade.
|
|
171
174
|
- Mid-conversation `system` messages preserve the requested model and pass through unchanged. They do not count as a tool continuation by themselves.
|
|
172
175
|
- Recognized compaction and auxiliary requests otherwise preserve their requested model. Token counting and model discovery pass through without classification.
|
|
@@ -233,6 +236,30 @@ The launcher is quiet by default. Standalone `serve` logs to stderr by default.
|
|
|
233
236
|
|
|
234
237
|
**Claude says Haiku while AutoRouter says Opus:** the built-in model label is Claude's starting/requested model. The AutoRouter confirmed-model label comes from Anthropic. Claude's own token-cost estimate can likewise be attributed to the requested model.
|
|
235
238
|
|
|
239
|
+
**`/goal` repeats a question or says it is blocked:** Claude Code runs its own completion checker after each turn, separately from AutoRouter's Jev/Ollama classifier. AutoRouter preserves the requested model for that auxiliary check and forwards its verdict unchanged. The check cannot authorize GitHub SAML, approve a tool, or resolve an external dependency. A tool's authentication error is also different from a Claude API authentication failure.
|
|
240
|
+
|
|
241
|
+
If the worker reports a blocker but the checker keeps returning “not yet met,” Claude can repeat its answer until its no-progress guard pauses the goal. Repeated tool calls can keep the loop running longer. Use `/goal clear` to end the loop, resolve the external blocker, and set the goal again. For tasks that may require human action, explicitly allow reporting a blocker as an alternative end condition, for example: `/goal Verify the discrepancy against upstream main and run the relevant tests, or report an external authorization blocker and stop.` This changes what counts as completion; AutoRouter does not declare blocked work successful or rewrite goal instructions. See [Claude Code goal evaluation](https://code.claude.com/docs/en/goal#how-evaluation-works).
|
|
242
|
+
|
|
243
|
+
### Shorter Stop-hook loops (opt-in)
|
|
244
|
+
|
|
245
|
+
To return control sooner when a goal keeps reporting the same unmet condition, set Claude's native continuation cap for one launch:
|
|
246
|
+
|
|
247
|
+
```sh
|
|
248
|
+
env CLAUDE_CODE_STOP_HOOK_BLOCK_CAP=2 claude-autorouter claude
|
|
249
|
+
```
|
|
250
|
+
|
|
251
|
+
This permits two consecutive continuations without tool use; the third blocking verdict ends the turn. The goal remains set and unmet, and a new message can resume it. Tool activity resets the counter, so this is not a total turn or request limit and cannot bound repeated failed tool calls. It applies to **all Stop and SubagentStop hooks**, including `/goal`. A smaller cap can pause useful work sooner. Unset preserves Claude's default (currently `8`); **`0` disables the guard**. AutoRouter does not install a Stop hook or change completion verdicts. See [Claude's environment-variable reference](https://code.claude.com/docs/en/env-vars) and [Stop-hook loop behavior](https://code.claude.com/docs/en/hooks#stop).
|
|
252
|
+
|
|
253
|
+
The environment-only command also works on AutoRouter 0.3.4. Saved configuration and the setup flag require AutoRouter 0.3.5 or newer. To save the preference, add the following property to your existing AutoRouter config JSON, preserving its other values:
|
|
254
|
+
|
|
255
|
+
```json
|
|
256
|
+
"CLAUDE_CODE_STOP_HOOK_BLOCK_CAP": "2"
|
|
257
|
+
```
|
|
258
|
+
|
|
259
|
+
For a new configuration, use `claude-autorouter setup --stop-hook-block-cap 2`; the flag works with either evaluator and overrides the environment during setup. Runtime environment values override saved configuration. `setup --force` replaces the entire config, so keep your existing evaluator/authentication options if using it. `doctor` reports the cap when configured. AutoRouter accepts nonnegative safe integers and leaves the setting absent unless you opt in.
|
|
260
|
+
|
|
261
|
+
### Other session issues
|
|
262
|
+
|
|
236
263
|
**After restarting mid-conversation:** turn state is in memory and expires after 30 minutes. Unknown continuations preserve the incoming model. Start a fresh conversation when restarting around signed thinking; AutoRouter cannot reconstruct the prior actual model from lost turn state.
|
|
237
264
|
|
|
238
265
|
Switching models can lose prompt-cache reuse. A cheaper price per token does not guarantee a cheaper or faster task. Only requests using Claude's configured base URL are visible to this proxy. Alternate provider modes such as Bedrock, Vertex, Foundry, Mantle, and `ANTHROPIC_AWS` are unsupported; unset their enable flags before launching. Use ordinary `claude` to bypass routing.
|
package/docs/releasing.md
CHANGED
|
@@ -8,6 +8,10 @@ Version `0.3.2` fixes local timeout fallbacks with model-specific deadlines, rem
|
|
|
8
8
|
|
|
9
9
|
Version `0.3.3` fixes HTTP 400 errors when a compatible request with disabled thinking is routed to Sonnet 5.5. Inference and token counting translate that setting to `between_tools`, or adaptive thinking when effort settings require it. Model defaults are unchanged; select Sonnet 5.5 with `AUTOROUTER_SONNET_MODEL=claude-sonnet-5-5`.
|
|
10
10
|
|
|
11
|
+
Version `0.3.4` preserves the selected model across Stop-hook feedback for the same prompt, including `/goal` commands that omit the gateway prompt-ID header. Recognized goal feedback remains conversation context rather than replacing the human task in evaluator excerpts. Goal-checker verdicts remain unchanged; external authorization blockers can still cause Claude's own goal loop to repeat. See [goal troubleshooting](reference.md#troubleshooting).
|
|
12
|
+
|
|
13
|
+
Version `0.3.5` adds opt-in saved configuration for Claude's native `CLAUDE_CODE_STOP_HOOK_BLOCK_CAP`, with `setup --stop-hook-block-cap N`, validation, and `doctor` reporting. A value of `2` permits two consecutive Stop-hook continuations without tool use and ends the turn on the third blocking verdict, leaving an unmet goal set. This affects all Stop/SubagentStop hooks, and tool activity resets the counter. Defaults and completion verdicts are unchanged; `0` disables the guard. See [shorter Stop-hook loops](reference.md#shorter-stop-hook-loops-opt-in).
|
|
14
|
+
|
|
11
15
|
The GitHub repository is private. Publishing to npm makes the tarball's runtime source, README, configuration example, license, and shipped documentation public. Model weights, user configuration, credentials, transcripts, local artifacts, and test fixtures are excluded. Review the archive before the first publication and whenever the package allowlist changes.
|
|
12
16
|
|
|
13
17
|
## What runs automatically
|
package/package.json
CHANGED
package/src/auth.mjs
CHANGED
|
@@ -16,6 +16,11 @@ export function isSubscriptionRequest(headers) {
|
|
|
16
16
|
|
|
17
17
|
export function buildClaudeEnv(config, baseUrl, parent = process.env) {
|
|
18
18
|
const env = { ...parent, ANTHROPIC_BASE_URL: baseUrl, CLAUDE_CODE_GATEWAY_HINT_HEADERS: '1' };
|
|
19
|
+
// Opt into Claude's own loop guard without installing hooks or altering
|
|
20
|
+
// their verdicts. An unset cap leaves Claude's default in control.
|
|
21
|
+
if (env.CLAUDE_CODE_STOP_HOOK_BLOCK_CAP === undefined && config.stopHookBlockCap !== undefined) {
|
|
22
|
+
env.CLAUDE_CODE_STOP_HOOK_BLOCK_CAP = String(config.stopHookBlockCap);
|
|
23
|
+
}
|
|
19
24
|
// Claude Code otherwise disables MCP tool search for a non-first-party
|
|
20
25
|
// base URL and loads every schema into context. This proxy preserves both
|
|
21
26
|
// tool_reference blocks and their beta headers. Respect explicit choices.
|
package/src/config.mjs
CHANGED
|
@@ -2,6 +2,15 @@ import { DEFAULT_OLLAMA_MODEL, defaultOllamaTimeoutMs, validateOllamaEndpoint, v
|
|
|
2
2
|
|
|
3
3
|
export const TIERS = ['haiku', 'sonnet', 'opus'];
|
|
4
4
|
|
|
5
|
+
export function parseStopHookBlockCap(value, name = 'CLAUDE_CODE_STOP_HOOK_BLOCK_CAP') {
|
|
6
|
+
if (!['string', 'number'].includes(typeof value)
|
|
7
|
+
|| (typeof value === 'string' && !/^[0-9]+$/.test(value.trim()))
|
|
8
|
+
|| !Number.isSafeInteger(Number(value)) || Number(value) < 0) {
|
|
9
|
+
throw new Error(`${name} requires a nonnegative safe integer (0 disables the Stop-hook continuation cap)`);
|
|
10
|
+
}
|
|
11
|
+
return Number(value);
|
|
12
|
+
}
|
|
13
|
+
|
|
5
14
|
function number(env, key, fallback, min, max, integer = true) {
|
|
6
15
|
const value = Number(env[key] ?? fallback);
|
|
7
16
|
if (!Number.isFinite(value) || value < min || value > max || (integer && !Number.isInteger(value))) {
|
|
@@ -50,6 +59,8 @@ export function readConfig(env = process.env) {
|
|
|
50
59
|
evaluator,
|
|
51
60
|
authMode,
|
|
52
61
|
clientProfile,
|
|
62
|
+
stopHookBlockCap: env.CLAUDE_CODE_STOP_HOOK_BLOCK_CAP === undefined
|
|
63
|
+
? undefined : parseStopHookBlockCap(env.CLAUDE_CODE_STOP_HOOK_BLOCK_CAP),
|
|
53
64
|
anthropicKey: authMode === 'api-key' ? env.ANTHROPIC_API_KEY : undefined,
|
|
54
65
|
jevKey: env.TYPESAFE_API_KEY,
|
|
55
66
|
localToken: env.AUTOROUTER_TOKEN,
|
package/src/onboarding.mjs
CHANGED
|
@@ -3,7 +3,7 @@ import { existsSync } from 'node:fs';
|
|
|
3
3
|
import { createInterface } from 'node:readline';
|
|
4
4
|
import { Writable } from 'node:stream';
|
|
5
5
|
import { promisify } from 'node:util';
|
|
6
|
-
import { readConfig, requireKeys } from './config.mjs';
|
|
6
|
+
import { readConfig, requireKeys, parseStopHookBlockCap } from './config.mjs';
|
|
7
7
|
import { buildClaudeEnv, conflictingProviders, LOCAL_AUTH_HEADER } from './auth.mjs';
|
|
8
8
|
import { getConfigPath, loadUserConfig, saveUserConfig } from './user-config.mjs';
|
|
9
9
|
import { DEFAULT_OLLAMA_MODEL, validateOllamaModel } from './ollama-models.mjs';
|
|
@@ -14,6 +14,10 @@ const execute = promisify(execFile);
|
|
|
14
14
|
export const ollamaDeadlineText = timeoutMs => timeoutMs === 0
|
|
15
15
|
? 'routing deadline disabled' : `routing deadline ${timeoutMs} ms per request`;
|
|
16
16
|
|
|
17
|
+
const stopHookCapText = cap => cap === 0
|
|
18
|
+
? 'Claude Stop/SubagentStop continuation cap disabled (0).'
|
|
19
|
+
: `Claude Stop/SubagentStop cap: ${cap} continuations without tool use.`;
|
|
20
|
+
|
|
17
21
|
// Readline manages editing and restores terminal state; its output is discarded
|
|
18
22
|
// so neither typing nor pasted credentials are echoed to the terminal.
|
|
19
23
|
export async function askSecret(label, { input = process.stdin, output = process.stderr } = {}) {
|
|
@@ -40,6 +44,7 @@ export async function setup(args, {
|
|
|
40
44
|
let evaluator = env.AUTOROUTER_EVALUATOR ?? 'jev';
|
|
41
45
|
let model;
|
|
42
46
|
let ollamaTimeoutMs;
|
|
47
|
+
let stopHookBlockCap = env.CLAUDE_CODE_STOP_HOOK_BLOCK_CAP;
|
|
43
48
|
let pull = false;
|
|
44
49
|
let overwrite = false;
|
|
45
50
|
for (let i = 0; i < args.length; i++) {
|
|
@@ -53,19 +58,24 @@ export async function setup(args, {
|
|
|
53
58
|
}
|
|
54
59
|
ollamaTimeoutMs = String(Number(value));
|
|
55
60
|
}
|
|
61
|
+
else if (args[i] === '--stop-hook-block-cap') {
|
|
62
|
+
stopHookBlockCap = parseStopHookBlockCap(args[++i], '--stop-hook-block-cap');
|
|
63
|
+
}
|
|
56
64
|
else if (args[i] === '--pull') pull = true;
|
|
57
65
|
else if (args[i] === '--force') overwrite = true;
|
|
58
|
-
else throw new Error('Usage: claude-autorouter setup [--auth-mode subscription|api-key] [--evaluator jev|ollama] [--ollama-model TAG] [--ollama-timeout-ms N] [--pull] [--force]');
|
|
66
|
+
else throw new Error('Usage: claude-autorouter setup [--auth-mode subscription|api-key] [--evaluator jev|ollama] [--ollama-model TAG] [--ollama-timeout-ms N] [--stop-hook-block-cap N] [--pull] [--force]');
|
|
59
67
|
}
|
|
60
68
|
if (!['subscription', 'api-key'].includes(authMode)) throw new Error('--auth-mode must be subscription or api-key');
|
|
61
69
|
if (!['jev', 'ollama'].includes(evaluator)) throw new Error('--evaluator must be jev or ollama');
|
|
62
70
|
if (evaluator !== 'ollama' && (model !== undefined || ollamaTimeoutMs !== undefined || pull)) throw new Error('Ollama model, deadline and download options require --evaluator ollama');
|
|
71
|
+
if (stopHookBlockCap !== undefined) stopHookBlockCap = parseStopHookBlockCap(stopHookBlockCap);
|
|
63
72
|
const path = getConfigPath(env);
|
|
64
73
|
if (!overwrite && existsSync(path)) throw new Error('AutoRouter configuration already exists. Use setup --force to replace it.');
|
|
65
74
|
write(evaluator === 'ollama'
|
|
66
75
|
? 'AutoRouter evaluates bounded prompt excerpts locally with Ollama. Complete requests still go to Anthropic.'
|
|
67
76
|
: 'AutoRouter sends bounded prompt excerpts to TypeSafe Jev and complete requests to Anthropic.');
|
|
68
77
|
const values = { AUTOROUTER_AUTH_MODE: authMode, AUTOROUTER_CLIENT_PROFILE: 'compatible', AUTOROUTER_EVALUATOR: evaluator };
|
|
78
|
+
if (stopHookBlockCap !== undefined) values.CLAUDE_CODE_STOP_HOOK_BLOCK_CAP = String(stopHookBlockCap);
|
|
69
79
|
if (evaluator === 'ollama') {
|
|
70
80
|
values.AUTOROUTER_OLLAMA_MODEL = validateOllamaModel(model ?? env.AUTOROUTER_OLLAMA_MODEL ?? DEFAULT_OLLAMA_MODEL);
|
|
71
81
|
for (const key of ['AUTOROUTER_OLLAMA_URL', 'AUTOROUTER_OLLAMA_TIMEOUT_MS', 'AUTOROUTER_OLLAMA_KEEP_ALIVE']) {
|
|
@@ -81,6 +91,7 @@ export async function setup(args, {
|
|
|
81
91
|
}
|
|
82
92
|
const config = readConfig(values);
|
|
83
93
|
requireKeys(config);
|
|
94
|
+
if (config.stopHookBlockCap !== undefined) write(stopHookCapText(config.stopHookBlockCap));
|
|
84
95
|
if (evaluator === 'ollama') {
|
|
85
96
|
write(`Local evaluator: ${config.ollamaModel}; ${ollamaDeadlineText(config.ollamaTimeoutMs)}.`);
|
|
86
97
|
const controller = new AbortController();
|
|
@@ -115,6 +126,7 @@ export async function doctor({ env = process.env, write = console.log, run = exe
|
|
|
115
126
|
for (const key of conflictingProviders(effectiveEnv)) {
|
|
116
127
|
report(false, `Unset ${key}; AutoRouter uses the Anthropic Messages API`);
|
|
117
128
|
}
|
|
129
|
+
if (config?.stopHookBlockCap !== undefined) write(stopHookCapText(config.stopHookBlockCap));
|
|
118
130
|
if (config?.evaluator === 'ollama') {
|
|
119
131
|
write(`Local evaluator: ${config.ollamaModel}; ${ollamaDeadlineText(config.ollamaTimeoutMs)}.`);
|
|
120
132
|
write('Model availability is checked below; classification speed and accuracy are not tested.');
|
package/src/prompt-state.mjs
CHANGED
|
@@ -76,12 +76,60 @@ function humanTask(message) {
|
|
|
76
76
|
return contentText(message.content, true);
|
|
77
77
|
}
|
|
78
78
|
|
|
79
|
+
// Claude Code emits /goal Stop-hook feedback as user text. Recognize only its
|
|
80
|
+
// observed wrapper and a condition established by an earlier expanded command;
|
|
81
|
+
// ordinary messages mentioning hooks remain human tasks. This does not alter
|
|
82
|
+
// message content, and the feedback remains available as classifier history.
|
|
83
|
+
export function goalFeedbackIndexes(messages) {
|
|
84
|
+
const indexes = new Set();
|
|
85
|
+
if (!Array.isArray(messages)) return indexes;
|
|
86
|
+
let condition;
|
|
87
|
+
let shortCondition;
|
|
88
|
+
let sawFullFeedback = false;
|
|
89
|
+
for (let index = 0; index < messages.length; index++) {
|
|
90
|
+
const message = messages[index];
|
|
91
|
+
if (message?.role !== 'user') continue;
|
|
92
|
+
const command = /^\s*<command-name>\/goal<\/command-name>\s*<command-message>goal<\/command-message>\s*<command-args>([\s\S]*?)<\/command-args>(?:\s|$)/.exec(humanTask(message));
|
|
93
|
+
if (command) {
|
|
94
|
+
const value = command[1].trim();
|
|
95
|
+
// /goal with no arguments is a status query, not a replacement goal.
|
|
96
|
+
if (!value) continue;
|
|
97
|
+
condition = value.length <= 4000 && !/^(?:clear|stop|off|reset|none|cancel)$/i.test(value) ? value : undefined;
|
|
98
|
+
shortCondition = undefined;
|
|
99
|
+
sawFullFeedback = false;
|
|
100
|
+
if (condition?.length > 500) {
|
|
101
|
+
let prefix = condition.slice(0, 500);
|
|
102
|
+
const last = prefix.charCodeAt(prefix.length - 1);
|
|
103
|
+
if (last >= 0xd800 && last <= 0xdbff) prefix = prefix.slice(0, -1);
|
|
104
|
+
shortCondition = `${prefix}… [+${condition.length - prefix.length} chars]`;
|
|
105
|
+
}
|
|
106
|
+
continue;
|
|
107
|
+
}
|
|
108
|
+
if (!condition || messages[index - 1]?.role !== 'assistant') continue;
|
|
109
|
+
const text = typeof message.content === 'string' ? message.content
|
|
110
|
+
: Array.isArray(message.content) && message.content.length === 1 && message.content[0]?.type === 'text'
|
|
111
|
+
&& typeof message.content[0].text === 'string' ? message.content[0].text : undefined;
|
|
112
|
+
if (text === undefined) continue;
|
|
113
|
+
const matches = label => {
|
|
114
|
+
const prefix = `Stop hook feedback:\n[${label}]: `;
|
|
115
|
+
return text.startsWith(prefix) && Boolean(text.slice(prefix.length).trim());
|
|
116
|
+
};
|
|
117
|
+
if (matches(condition)) {
|
|
118
|
+
indexes.add(index);
|
|
119
|
+
sawFullFeedback = true;
|
|
120
|
+
} else if (sawFullFeedback && shortCondition && matches(shortCondition)) indexes.add(index);
|
|
121
|
+
}
|
|
122
|
+
return indexes;
|
|
123
|
+
}
|
|
124
|
+
|
|
79
125
|
export function buildState(body, limit = 12000) {
|
|
80
126
|
const messages = body.messages ?? [];
|
|
127
|
+
const feedbackIndexes = goalFeedbackIndexes(messages);
|
|
81
128
|
let firstTask = '';
|
|
82
129
|
let currentTask = '';
|
|
83
130
|
let currentIndex = -1;
|
|
84
131
|
for (let index = 0; index < messages.length; index++) {
|
|
132
|
+
if (feedbackIndexes.has(index)) continue;
|
|
85
133
|
const task = humanTask(messages[index]);
|
|
86
134
|
if (!task.trim()) continue;
|
|
87
135
|
if (!firstTask) firstTask = task;
|
package/src/router.mjs
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { createHash } from 'node:crypto';
|
|
2
2
|
import { TIERS } from './config.mjs';
|
|
3
|
-
import { buildState } from './prompt-state.mjs';
|
|
3
|
+
import { buildState, goalFeedbackIndexes } from './prompt-state.mjs';
|
|
4
4
|
import { buildOllamaState, evaluateOllama } from './ollama-evaluator.mjs';
|
|
5
5
|
export { buildState } from './prompt-state.mjs';
|
|
6
6
|
|
|
@@ -134,10 +134,11 @@ function turnContent(content) {
|
|
|
134
134
|
|
|
135
135
|
function turnInfo(body, scope, promptId = '') {
|
|
136
136
|
const messages = body.messages ?? [];
|
|
137
|
+
const feedback = goalFeedbackIndexes(messages);
|
|
137
138
|
let index = -1;
|
|
138
139
|
for (let i = messages.length - 1; i >= 0; i--) {
|
|
139
140
|
const message = messages[i];
|
|
140
|
-
if (message.role === 'user' && !(Array.isArray(message.content) && message.content.some(b => b.type === 'tool_result'))) {
|
|
141
|
+
if (message.role === 'user' && !feedback.has(i) && !(Array.isArray(message.content) && message.content.some(b => b.type === 'tool_result'))) {
|
|
141
142
|
index = i; break;
|
|
142
143
|
}
|
|
143
144
|
}
|
|
@@ -147,6 +148,7 @@ function turnInfo(body, scope, promptId = '') {
|
|
|
147
148
|
index,
|
|
148
149
|
key: promptId ? hash(['prompt', scope, promptId]) : contentKey,
|
|
149
150
|
contentKey,
|
|
151
|
+
goalFeedback: feedback.has(messages.length - 1),
|
|
150
152
|
// Claude Code can append turn-scoped system instructions after the human
|
|
151
153
|
// prompt. These do not start an assistant/tool continuation.
|
|
152
154
|
continuation: index < 0 || messages.slice(index + 1).some(m => m.role !== 'system'),
|
|
@@ -248,11 +250,15 @@ export class Router {
|
|
|
248
250
|
let model = c.models[decision.tier];
|
|
249
251
|
let reason = decision.reason;
|
|
250
252
|
const turn = turnInfo(body, scope, promptId);
|
|
251
|
-
|
|
253
|
+
const promptPin = promptId ? this.turns.get(turn.key) : undefined;
|
|
254
|
+
const turnPin = promptPin ?? this.turns.get(turn.contentKey);
|
|
255
|
+
let previous = turnPin?.model;
|
|
256
|
+
const textTurn = !turn.continuation || turn.goalFeedback;
|
|
257
|
+
const textPin = promptPin ?? (!promptId && turn.goalFeedback ? turnPin : undefined);
|
|
252
258
|
// A new human prompt can still carry signed thinking from the preceding
|
|
253
259
|
// turn. Recover that turn's actual routed model when it is known.
|
|
254
260
|
if (!turn.continuation && body.messages.length > 1 && !previous) {
|
|
255
|
-
previous = this.turns.get(turnInfo({ ...body, messages: body.messages.slice(0, turn.index) }, scope).key);
|
|
261
|
+
previous = this.turns.get(turnInfo({ ...body, messages: body.messages.slice(0, turn.index) }, scope).key)?.model;
|
|
256
262
|
}
|
|
257
263
|
let preserved = false;
|
|
258
264
|
const preserve = (chosen, why) => { model = chosen; reason = why; preserved = true; };
|
|
@@ -272,8 +278,21 @@ export class Router {
|
|
|
272
278
|
// A new native request can explicitly select a model-specific thinking
|
|
273
279
|
// mode, including between_tools. An earlier turn's model is not evidence
|
|
274
280
|
// that it accepts that mode. Existing tool turns retain their pin below.
|
|
275
|
-
else if (modelSpecificThinking && body.thinking.type !== 'enabled' &&
|
|
276
|
-
|
|
281
|
+
else if (modelSpecificThinking && body.thinking.type !== 'enabled' && textTurn) preserve(body.model, 'model_specific_features');
|
|
282
|
+
// Stop hooks (including /goal) return feedback as user-role text, even
|
|
283
|
+
// though it still serves the same human prompt. Trust the scoped gateway
|
|
284
|
+
// identity instead of treating that text as a new task. A client model
|
|
285
|
+
// change can be an explicit fallback after a failure; do not undo it.
|
|
286
|
+
// Local /goal commands can omit the gateway prompt ID. Exact feedback for
|
|
287
|
+
// a known goal then uses the original conversation anchor as a fallback.
|
|
288
|
+
else if (textTurn && textPin?.requestedModel === body.model && !modelSpecificFeatures) {
|
|
289
|
+
const needsSonnet = body.thinking?.type === 'adaptive' || body.output_config?.effort || body.max_tokens > 64000;
|
|
290
|
+
if (needsSonnet && (textPin.model === c.models.haiku || rank(textPin.model) === 0)) {
|
|
291
|
+
preserve(c.models.sonnet, 'requires_sonnet_capabilities');
|
|
292
|
+
} else preserve(textPin.model, promptPin ? 'prompt_turn_pinned' : 'goal_turn_pinned');
|
|
293
|
+
}
|
|
294
|
+
else if (turn.goalFeedback && !turnPin) preserve(body.model, 'unknown_continuation');
|
|
295
|
+
else if (turn.continuation && !turn.goalFeedback) keep(previous ? 'tool_turn_pinned' : 'unknown_continuation');
|
|
277
296
|
// Unknown or model-specific features are preserved, never silently removed.
|
|
278
297
|
else if (decision.source === 'fallback' && rank(body.model) >= 1) keep('classifier_unavailable');
|
|
279
298
|
else if (modelSpecificFeatures) keep('model_specific_features');
|
|
@@ -318,12 +337,13 @@ export class Router {
|
|
|
318
337
|
}
|
|
319
338
|
const identifiableUpgrade = capacityUpgraded && (turn.index >= 0 || promptId);
|
|
320
339
|
if ((!turn.continuation || previous || identifiableUpgrade || reason === 'mid_conversation_system') && !['compaction', 'auxiliary'].includes(requestClass)) {
|
|
321
|
-
|
|
340
|
+
const pin = { model, requestedModel: body.model };
|
|
341
|
+
this.turns.set(turn.key, pin);
|
|
322
342
|
// Keep the content key too: later human turns carry signed thinking but
|
|
323
343
|
// have a new prompt ID, so they must recover the preceding routed model.
|
|
324
344
|
// Refresh this alias during known continuations too, since tool discovery
|
|
325
345
|
// can change the content key without changing the gateway prompt ID.
|
|
326
|
-
if (turn.contentKey !== turn.key) this.turns.set(turn.contentKey,
|
|
346
|
+
if (turn.contentKey !== turn.key) this.turns.set(turn.contentKey, pin);
|
|
327
347
|
}
|
|
328
348
|
return { ...decision, ...contextCheck, model, reason, latency_ms: Math.round((performance.now() - start) * 100) / 100 };
|
|
329
349
|
}
|
package/src/statusline.mjs
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
const PHASES = new Set(['routing', 'connecting', 'streaming', 'ready', 'error', 'cancelled']);
|
|
2
2
|
const REASONS = {
|
|
3
|
-
tool_turn_pinned: 'turn pinned',
|
|
3
|
+
tool_turn_pinned: 'turn pinned', prompt_turn_pinned: 'prompt pinned', goal_turn_pinned: 'goal pinned',
|
|
4
|
+
thinking_history: 'thinking pinned', unknown_continuation: 'continuation pinned',
|
|
4
5
|
mid_conversation_system: 'system features', requires_sonnet_capabilities: 'capability guard',
|
|
5
6
|
model_specific_features: 'model features', large_or_multimodal_request: 'large request',
|
|
6
7
|
context_capacity: 'large context',
|
package/src/user-config.mjs
CHANGED
|
@@ -13,7 +13,7 @@ const CONFIG_KEYS = new Set([
|
|
|
13
13
|
'AUTOROUTER_HAIKU_MODEL', 'AUTOROUTER_SONNET_MODEL', 'AUTOROUTER_OPUS_MODEL',
|
|
14
14
|
'AUTOROUTER_PORT', 'AUTOROUTER_JEV_TIMEOUT_MS', 'AUTOROUTER_TOKEN_COUNT_TIMEOUT_MS',
|
|
15
15
|
'AUTOROUTER_MIN_CONFIDENCE', 'AUTOROUTER_STATUSLINE', 'AUTOROUTER_DEBUG',
|
|
16
|
-
'ENABLE_TOOL_SEARCH',
|
|
16
|
+
'ENABLE_TOOL_SEARCH', 'CLAUDE_CODE_STOP_HOOK_BLOCK_CAP',
|
|
17
17
|
'AUTOROUTER_EVALUATOR', 'AUTOROUTER_OLLAMA_URL', 'AUTOROUTER_OLLAMA_MODEL',
|
|
18
18
|
'AUTOROUTER_OLLAMA_TIMEOUT_MS', 'AUTOROUTER_OLLAMA_KEEP_ALIVE',
|
|
19
19
|
]);
|