claude-autorouter 0.3.2 → 0.3.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/docs/reference.md +3 -1
- package/docs/releasing.md +2 -0
- package/package.json +1 -1
- package/src/model-request.mjs +21 -6
- package/src/router.mjs +6 -1
- package/src/token-counter.mjs +5 -1
package/docs/reference.md
CHANGED
|
@@ -171,7 +171,9 @@ The following policy applies after classification:
|
|
|
171
171
|
- Mid-conversation `system` messages preserve the requested model and pass through unchanged. They do not count as a tool continuation by themselves.
|
|
172
172
|
- Recognized compaction and auxiliary requests otherwise preserve their requested model. Token counting and model discovery pass through without classification.
|
|
173
173
|
|
|
174
|
-
The default `compatible` profile starts Claude with Haiku-compatible requests and client-requested thinking disabled.
|
|
174
|
+
The default `compatible` profile starts Claude with Haiku-compatible requests and client-requested thinking disabled. AutoRouter uses adaptive thinking when upgrading these requests to Opus 5/5.5. Starting with 0.3.3, routing to exact `claude-sonnet-5-5` translates disabled thinking to `between_tools`, which skips up-front thinking but permits progress updates between tool calls. At `xhigh`/`max` effort, or when per-message effort differs from the top-level setting (default `high`), it uses adaptive thinking while preserving the effort settings. Token counting uses the same adaptation. Sonnet 5 still accepts disabled thinking and is unchanged. See [Sonnet 5.5 thinking requirements](https://platform.claude.com/docs/en/models/sonnet-5-5/migration-guide).
|
|
175
|
+
|
|
176
|
+
Explicit native `between_tools` and unknown thinking modes retain the incoming model on new human turns. Signed thinking blocks pass through unchanged and existing tool turns retain their model pin. `AUTOROUTER_CLIENT_PROFILE=native` preserves normal client settings, which can constrain routing. An explicit Claude `--model` argument overrides the starting model, but `/model` and `--model` are requested models, not locks on the routed result. Native same-model requests and unknown model aliases are not rewritten; clients must use settings supported by that model.
|
|
175
177
|
|
|
176
178
|
The launcher enables `ENABLE_TOOL_SEARCH=true` when unset. Claude can otherwise disable on-demand MCP discovery when using a custom API address, loading connected-tool schemas into even a fresh conversation. Explicit values, including `false` or `auto:5`, are preserved. Managed settings and always-loaded tools can still affect deferral. See [Claude Code tool search](https://code.claude.com/docs/en/mcp#configure-tool-search).
|
|
177
179
|
|
package/docs/releasing.md
CHANGED
|
@@ -6,6 +6,8 @@ Version `0.3.1` replaces the old Ollama chat evaluator and Qwen presets with the
|
|
|
6
6
|
|
|
7
7
|
Version `0.3.2` fixes local timeout fallbacks with model-specific deadlines, removes Claude executor instructions from local evaluator excerpts, and keeps fallback causes visible in compact status lines. It also adds `AUTOROUTER_OLLAMA_TIMEOUT_MS=0` and `setup --ollama-timeout-ms 0` to disable the runtime evaluation deadline while preserving caller cancellation and the separate startup warmup limit. Existing explicit timeout settings still override the defaults; Jev is unchanged.
|
|
8
8
|
|
|
9
|
+
Version `0.3.3` fixes HTTP 400 errors when a compatible request with disabled thinking is routed to Sonnet 5.5. Inference and token counting translate that setting to `between_tools`, or adaptive thinking when effort settings require it. Model defaults are unchanged; select Sonnet 5.5 with `AUTOROUTER_SONNET_MODEL=claude-sonnet-5-5`.
|
|
10
|
+
|
|
9
11
|
The GitHub repository is private. Publishing to npm makes the tarball's runtime source, README, configuration example, license, and shipped documentation public. Model weights, user configuration, credentials, transcripts, local artifacts, and test fixtures are excluded. Review the archive before the first publication and whenever the package allowlist changes.
|
|
10
12
|
|
|
11
13
|
## What runs automatically
|
package/package.json
CHANGED
package/src/model-request.mjs
CHANGED
|
@@ -1,13 +1,28 @@
|
|
|
1
|
-
//
|
|
2
|
-
//
|
|
3
|
-
const
|
|
1
|
+
// Keep adaptations explicit: models in the same family can have different
|
|
2
|
+
// thinking contracts. Preserve the existing adaptive Opus upgrade behavior.
|
|
3
|
+
const ADAPTIVE_TARGETS = new Set(['claude-opus-5', 'claude-opus-5-5']);
|
|
4
|
+
|
|
5
|
+
function sonnetNeedsAdaptive(body) {
|
|
6
|
+
const effort = body.output_config?.effort ?? 'high';
|
|
7
|
+
// Sonnet 5.5's between_tools mode accepts high effort or below, and cannot
|
|
8
|
+
// change effort through per-message overrides. Preserve those overrides by
|
|
9
|
+
// selecting adaptive thinking instead of dropping or lowering the effort.
|
|
10
|
+
return ['xhigh', 'max'].includes(effort) || body.messages?.some(message =>
|
|
11
|
+
message.output_config?.effort !== undefined && message.output_config.effort !== effort);
|
|
12
|
+
}
|
|
4
13
|
|
|
5
14
|
export function prepareRequest(body, model) {
|
|
6
15
|
const request = { ...body, model };
|
|
7
16
|
const adjustments = [];
|
|
8
|
-
if (model !== body.model &&
|
|
9
|
-
|
|
10
|
-
|
|
17
|
+
if (model !== body.model && body.thinking?.type === 'disabled') {
|
|
18
|
+
if (model === 'claude-sonnet-5-5') {
|
|
19
|
+
const type = sonnetNeedsAdaptive(body) ? 'adaptive' : 'between_tools';
|
|
20
|
+
request.thinking = { type };
|
|
21
|
+
adjustments.push(type === 'adaptive' ? 'adaptive_thinking_required' : 'between_tools_thinking_required');
|
|
22
|
+
} else if (ADAPTIVE_TARGETS.has(model)) {
|
|
23
|
+
request.thinking = { type: 'adaptive' };
|
|
24
|
+
adjustments.push('adaptive_thinking_required');
|
|
25
|
+
}
|
|
11
26
|
}
|
|
12
27
|
return { request, adjustments };
|
|
13
28
|
}
|
package/src/router.mjs
CHANGED
|
@@ -226,7 +226,8 @@ export class Router {
|
|
|
226
226
|
const c = this.config;
|
|
227
227
|
const hasSystemMessage = body.messages.some(m => m.role === 'system');
|
|
228
228
|
const unknownModel = rank(body.model) < 0 && !Object.values(c.models).includes(body.model);
|
|
229
|
-
const
|
|
229
|
+
const modelSpecificThinking = body.thinking && !['disabled', 'adaptive'].includes(body.thinking.type);
|
|
230
|
+
const modelSpecificFeatures = modelSpecificThinking || body.context_management || body.speed || body.container || body.mcp_servers || body.tools?.some(t => t.type && t.type !== 'custom');
|
|
230
231
|
const thinkingHistory = hasContentBlock(body, ['thinking', 'redacted_thinking']);
|
|
231
232
|
const knownSourceModel = LARGE_CONTEXT_MODELS.has(body.model) || CAPACITY_UPGRADE_MODELS.has(body.model);
|
|
232
233
|
const capacityLocked = hasSystemMessage || unknownModel || !knownSourceModel || modelSpecificFeatures || thinkingHistory;
|
|
@@ -268,6 +269,10 @@ export class Router {
|
|
|
268
269
|
// clear_at, tool changes, and output_config) instead of down-routing.
|
|
269
270
|
else if (hasSystemMessage) preserve(body.model, 'mid_conversation_system');
|
|
270
271
|
else if (unknownModel) preserve(body.model, 'unknown_model');
|
|
272
|
+
// A new native request can explicitly select a model-specific thinking
|
|
273
|
+
// mode, including between_tools. An earlier turn's model is not evidence
|
|
274
|
+
// that it accepts that mode. Existing tool turns retain their pin below.
|
|
275
|
+
else if (modelSpecificThinking && body.thinking.type !== 'enabled' && !turn.continuation) preserve(body.model, 'model_specific_features');
|
|
271
276
|
else if (turn.continuation) keep(previous ? 'tool_turn_pinned' : 'unknown_continuation');
|
|
272
277
|
// Unknown or model-specific features are preserved, never silently removed.
|
|
273
278
|
else if (decision.source === 'fallback' && rank(body.model) >= 1) keep('classifier_unavailable');
|
package/src/token-counter.mjs
CHANGED
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import { createHash } from 'node:crypto';
|
|
2
|
+
import { prepareRequest } from './model-request.mjs';
|
|
2
3
|
|
|
3
4
|
// Count-token API fields, including the beta fields used by Claude Code.
|
|
4
5
|
// Keep unknown extensions out of the count path: dropping new input context
|
|
@@ -80,7 +81,10 @@ export function createTokenCounter(config, { fetchImpl = fetch } = {}) {
|
|
|
80
81
|
let onAbort;
|
|
81
82
|
const controller = new AbortController();
|
|
82
83
|
try {
|
|
83
|
-
|
|
84
|
+
// Apply the same target-model compatibility changes as inference before
|
|
85
|
+
// projecting count fields. Keep the source model available to the adapter.
|
|
86
|
+
const prepared = prepareRequest(body, model).request;
|
|
87
|
+
const payload = Object.fromEntries(Object.entries(prepared).filter(([key]) => COUNT_FIELDS.has(key)));
|
|
84
88
|
const serialized = JSON.stringify(payload);
|
|
85
89
|
const requestHeaders = new Headers(headers);
|
|
86
90
|
requestHeaders.delete('content-length');
|