claude-autorouter 0.3.1 → 0.3.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +24 -5
- package/README.md +22 -4
- package/bin/autorouter.mjs +6 -3
- package/docs/development.md +29 -1
- package/docs/ollama-evaluation.md +36 -11
- package/docs/reference.md +29 -12
- package/docs/releasing.md +11 -9
- package/package.json +2 -1
- package/src/config.mjs +11 -3
- package/src/ollama-evaluator.mjs +17 -4
- package/src/ollama-models.mjs +15 -0
- package/src/onboarding.mjs +17 -3
- package/src/statusline.mjs +30 -6
package/.env.example
CHANGED
|
@@ -22,7 +22,7 @@ AUTOROUTER_JEV_TIMEOUT_MS=1500
|
|
|
22
22
|
AUTOROUTER_TOKEN_COUNT_TIMEOUT_MS=1500
|
|
23
23
|
AUTOROUTER_MIN_CONFIDENCE=0.75
|
|
24
24
|
|
|
25
|
-
# Experimental local evaluator in AutoRouter 0.3.
|
|
25
|
+
# Experimental local evaluator in AutoRouter 0.3.2+: native decision API only.
|
|
26
26
|
# Install/start Ollama 0.35+, then run:
|
|
27
27
|
# claude-autorouter setup --evaluator ollama --pull --force
|
|
28
28
|
# Defaults to nimble:9b-q4_K_M (~5.63 GB download); --force replaces user config.
|
|
@@ -31,22 +31,41 @@ AUTOROUTER_MIN_CONFIDENCE=0.75
|
|
|
31
31
|
# Add --ollama-model LOCAL_TAG_OR_ALIAS to setup to choose another suitable model.
|
|
32
32
|
# Tev1 alternatives use the same native API; choose one explicitly:
|
|
33
33
|
# claude-autorouter setup --evaluator ollama --ollama-model tev1:0.8b --pull --force
|
|
34
|
-
#
|
|
34
|
+
# claude-autorouter setup --evaluator ollama --ollama-model tev1:4b-q4_K_M --pull --force
|
|
35
35
|
# Downloads: Tev1 0.8B Q8 ~812 MB; Tev1 4B Q4_K_M ~2.7 GB.
|
|
36
36
|
# tev1:latest / tev1:4b select ~4.5 GB Q8; model terms: https://ollama.com/library/tev1
|
|
37
37
|
# Or set AUTOROUTER_EVALUATOR=ollama above and configure an installed model:
|
|
38
38
|
# AUTOROUTER_OLLAMA_URL=http://127.0.0.1:11434
|
|
39
39
|
# AUTOROUTER_OLLAMA_MODEL=nimble:9b-q4_K_M
|
|
40
|
-
#
|
|
40
|
+
# Version 0.3.2 defaults: Tev1 0.8B/custom 1500ms; Tev1 4B 15000ms; Nimble 30000ms.
|
|
41
|
+
# An explicit timeout (including an old saved 1500) always overrides the default.
|
|
42
|
+
# Setup, doctor and startup show the effective model/deadline.
|
|
43
|
+
# Set only when intentionally overriding; Jev's separate deadline is unchanged:
|
|
44
|
+
# AUTOROUTER_OLLAMA_TIMEOUT_MS=30000
|
|
45
|
+
# Set 0 to disable only the runtime evaluator timer:
|
|
46
|
+
# AUTOROUTER_OLLAMA_TIMEOUT_MS=0
|
|
47
|
+
# One launch without changing saved config:
|
|
48
|
+
# AUTOROUTER_OLLAMA_TIMEOUT_MS=0 claude-autorouter claude
|
|
49
|
+
# Persist it for an installed model (the setup flag overrides the environment):
|
|
50
|
+
# claude-autorouter setup --evaluator ollama --ollama-model tev1:4b --ollama-timeout-ms 0 --force
|
|
51
|
+
# Positive values 1..30000 retain a deadline; 2500 means a 2.5-second cutoff.
|
|
52
|
+
# Zero and --ollama-timeout-ms require AutoRouter 0.3.2 or newer.
|
|
41
53
|
# AUTOROUTER_OLLAMA_KEEP_ALIVE=5m
|
|
42
|
-
# The launcher primes the classifier before opening Claude, allowing up to 60s.
|
|
54
|
+
# The launcher primes the classifier before opening Claude, allowing up to 60s even with timeout 0.
|
|
55
|
+
# User/disconnect cancellation remains active; normal evaluator errors still use fallback.
|
|
56
|
+
# Warmup and metadata-only doctor checks do not certify speed or accuracy.
|
|
43
57
|
# Native model context is retained; evaluator state stays capped at 3,000 bytes.
|
|
44
|
-
#
|
|
58
|
+
# Local classification omits Claude executor system instructions; task/history remain.
|
|
59
|
+
# After the idle period, cold reloading may exceed an enabled deadline and use fallback.
|
|
45
60
|
# A longer keep-alive holds the model in memory longer but avoids some reloads.
|
|
46
61
|
# Native entropy confidence is not calibrated accuracy and does not use Jev's threshold.
|
|
62
|
+
# Historical measurements before 0.3.2:
|
|
47
63
|
# On the tested M4, all 12 Nimble tuning requests exceeded the 1,500 ms deadline.
|
|
48
64
|
# Tev1 0.8B: 18/24 labels, 450 ms median; 4B: 22/24, 3.15s at a 10s deadline.
|
|
49
65
|
# Both Tev1 tags have a 2,050-token native context window; see measured limits.
|
|
66
|
+
# Source-only regression: node scripts/test-ollama-routing.mjs --model tev1:0.8b
|
|
67
|
+
# Uses local inference only; no Anthropic/Jev calls, downloads or config writes.
|
|
68
|
+
# A fallback timeout is not a valid Sonnet decision; mismatches fail the regression.
|
|
50
69
|
# See docs/ollama-evaluation.md before allowing slower local classifications.
|
|
51
70
|
|
|
52
71
|
AUTOROUTER_PORT=8787
|
package/README.md
CHANGED
|
@@ -50,7 +50,23 @@ Savings are an **API-equivalent estimate for the same token counts**, using Opus
|
|
|
50
50
|
|
|
51
51
|
## Experimental local evaluator
|
|
52
52
|
|
|
53
|
-
The local setup below requires AutoRouter 0.3.
|
|
53
|
+
The local setup below requires AutoRouter 0.3.2 or newer. It uses Ollama's native `/v1/systemone` decision API with `nimble:9b-q4_K_M` by default. Jev remains the default evaluator. If upgrading from 0.2.0, replace the old Qwen model configuration using the [migration steps](docs/reference.md#migrating-an-older-ollama-config).
|
|
54
|
+
|
|
55
|
+
Version 0.3.2 excludes Claude's executor system instructions from the local classifier excerpt, retaining task and conversation excerpts. Runtime deadlines default to 1,500 ms for Tev1 0.8B/custom models, 15,000 ms for official Tev1 4B tags, and 30,000 ms for official Nimble tags. Explicit timeout settings, including a `1500` saved with 0.3.1, still override these defaults. Jev is unchanged.
|
|
56
|
+
|
|
57
|
+
Set `0` to disable AutoRouter's runtime evaluator deadline for one launch using your existing configuration:
|
|
58
|
+
|
|
59
|
+
```sh
|
|
60
|
+
AUTOROUTER_OLLAMA_TIMEOUT_MS=0 claude-autorouter claude
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
To save that setting for an installed Tev1 4B model:
|
|
64
|
+
|
|
65
|
+
```sh
|
|
66
|
+
claude-autorouter setup --evaluator ollama --ollama-model tev1:4b --ollama-timeout-ms 0 --force
|
|
67
|
+
```
|
|
68
|
+
|
|
69
|
+
The setup flag overrides the timeout environment value and saves it. Cancellation and disconnected clients still stop evaluation, normal errors still use fallback, and startup priming keeps its separate 60-second deadline.
|
|
54
70
|
|
|
55
71
|
Install and start Ollama 0.35 or newer; [version 0.35.0](https://github.com/ollama/ollama/releases/tag/v0.35.0) is a prerelease as of September 29, 2026. Then run:
|
|
56
72
|
|
|
@@ -74,16 +90,18 @@ For example, select Tev1 0.8B with:
|
|
|
74
90
|
claude-autorouter setup --evaluator ollama --ollama-model tev1:0.8b --pull --force
|
|
75
91
|
```
|
|
76
92
|
|
|
77
|
-
Use `--ollama-model tev1:4b-q4_K_M` for the listed 4B variant;
|
|
93
|
+
Use `--ollama-model tev1:4b-q4_K_M` for the listed 4B variant; `tev1:latest` and `tev1:4b` select the larger Q8 download. Model terms are linked in the listings above; download size does not measure resident memory or routing quality. Custom native model tags and aliases also work.
|
|
78
94
|
|
|
79
95
|
No Jev key is needed for local classification. The launcher primes the evaluator before opening Claude's UI, and evaluation failures fall back to Sonnet or retain Opus without contacting Jev. Claude still answers through Anthropic, with the same routing guards and subscription limits.
|
|
80
96
|
|
|
81
|
-
|
|
97
|
+
Setup, doctor, and startup show the effective model and deadline. Warmup and doctor do not certify classification speed or accuracy. `Ollama fallback: timeout` means no valid decision arrived in time; it is different from the evaluator choosing Sonnet. Source users can run the [local routing regression](docs/development.md#local-routing-regression) to check all three tiers without Claude or Jev calls.
|
|
98
|
+
|
|
99
|
+
Historical measurements before 0.3.2, on a 16 GiB M4: Tev1 0.8B matched 18/24 held-out labels with 450 ms median latency and no timeouts at 1,500 ms, including full-excerpt checks. Tev1 4B matched 22/24 with a 10-second diagnostic deadline and 3.15-second median latency. Nimble matched 23/24 with a 30-second deadline and 11.4-second median latency. Both larger models exceeded the then-default 1,500 ms. A separate six-case regression with the 0.3.2 fixes passed for both tested Tev1 4B variants and Nimble; Tev1 0.8B matched only three cases. These small tests do not establish general accuracy or Jev parity. See the [measurements and limits](docs/ollama-evaluation.md) and [Ollama reference](docs/reference.md#ollama-evaluator).
|
|
82
100
|
|
|
83
101
|
## Behavior and data
|
|
84
102
|
|
|
85
103
|
- The default client profile permits all three routing tiers. Tool continuations, thinking history, model-specific features, and context size can keep or upgrade a model even when the evaluator chooses a cheaper tier. [Routing policy](docs/reference.md#routing-policy).
|
|
86
|
-
- The selected evaluator receives bounded excerpts that can contain source code
|
|
104
|
+
- The selected evaluator receives bounded excerpts that can contain source code and tool results: TypeSafe with Jev, or the local service with Ollama. Jev also receives system-text excerpts; the local path excludes Claude's executor system instructions. Anthropic receives the complete request. Images, document payloads, and private thinking are omitted from classifier input. [Data flow and authentication](docs/reference.md#data-flow-and-authentication).
|
|
87
105
|
- Subscription access and usage limits still apply. Model switches can reduce cache reuse; cheaper token prices do not guarantee cheaper completed tasks. Run ordinary `claude` to bypass routing.
|
|
88
106
|
- The launcher is quiet by default. Use `AUTOROUTER_DEBUG=1` for metadata diagnostics or `AUTOROUTER_STATUSLINE=0` to retain your existing status line. [Troubleshooting](docs/reference.md#troubleshooting).
|
|
89
107
|
|
package/bin/autorouter.mjs
CHANGED
|
@@ -9,7 +9,7 @@ import { dirname } from 'node:path';
|
|
|
9
9
|
import { createStatusState } from '../src/status-state.mjs';
|
|
10
10
|
import { addStatusLineSettings } from '../src/status-settings.mjs';
|
|
11
11
|
import { loadUserConfig } from '../src/user-config.mjs';
|
|
12
|
-
import { setup, doctor } from '../src/onboarding.mjs';
|
|
12
|
+
import { setup, doctor, ollamaDeadlineText } from '../src/onboarding.mjs';
|
|
13
13
|
import { setupOllama } from '../src/ollama-setup.mjs';
|
|
14
14
|
|
|
15
15
|
const [command = 'help', ...args] = process.argv.slice(2);
|
|
@@ -22,7 +22,7 @@ if (['--version', '-v', 'version'].includes(command)) {
|
|
|
22
22
|
Usage:
|
|
23
23
|
claude-autorouter setup [--auth-mode subscription|api-key] [--force]
|
|
24
24
|
[--evaluator jev|ollama]
|
|
25
|
-
[--ollama-model MODEL] [--pull]
|
|
25
|
+
[--ollama-model MODEL] [--ollama-timeout-ms N] [--pull]
|
|
26
26
|
claude-autorouter doctor
|
|
27
27
|
claude-autorouter claude [Claude Code arguments]
|
|
28
28
|
claude-autorouter serve
|
|
@@ -39,6 +39,9 @@ Ollama evaluates locally and requires Ollama 0.35+ with /v1/systemone.
|
|
|
39
39
|
Use setup --evaluator ollama --pull to detect Ollama and download a missing model.
|
|
40
40
|
The local default is nimble:9b-q4_K_M; --ollama-model selects another compatible model.
|
|
41
41
|
Smaller Tev1 options: --ollama-model tev1:0.8b or --ollama-model tev1:4b-q4_K_M.
|
|
42
|
+
Local routing deadlines: Tev1 0.8B/custom 1500 ms, Tev1 4B 15000 ms, Nimble 30000 ms.
|
|
43
|
+
Setup --ollama-timeout-ms N saves a routing deadline; use 0 to disable it.
|
|
44
|
+
AUTOROUTER_OLLAMA_TIMEOUT_MS also overrides the deadline; 0 disables it.
|
|
42
45
|
Local routing is experimental; see docs/ollama-evaluation.md for measured limits.
|
|
43
46
|
AUTOROUTER_AUTH_MODE=subscription uses your saved Claude Code login.
|
|
44
47
|
Without setup, AUTOROUTER_AUTH_MODE defaults to api-key and also requires ANTHROPIC_API_KEY.
|
|
@@ -84,7 +87,7 @@ Complete inference requests still go to Anthropic. See README.md.`);
|
|
|
84
87
|
config.localToken = randomBytes(32).toString('hex');
|
|
85
88
|
}
|
|
86
89
|
if (config.evaluator === 'ollama') {
|
|
87
|
-
console.error(`Preparing local Ollama evaluator (${config.ollamaModel})…`);
|
|
90
|
+
console.error(`Preparing local Ollama evaluator (${config.ollamaModel}); ${ollamaDeadlineText(config.ollamaTimeoutMs)}…`);
|
|
88
91
|
try { await setupOllama(config, { pull: false, warm: true, write: () => {} }); }
|
|
89
92
|
catch {
|
|
90
93
|
console.error('Ollama could not be prepared. Requests will use the conservative fallback while it is unavailable; run claude-autorouter doctor.');
|
package/docs/development.md
CHANGED
|
@@ -26,9 +26,37 @@ node --env-file=.env bin/autorouter.mjs claude
|
|
|
26
26
|
|
|
27
27
|
The explicit Node flag loads `.env`; the CLI itself does not auto-load project files. Environment values override the user config. Keep keys out of source control and command arguments.
|
|
28
28
|
|
|
29
|
+
With AutoRouter 0.3.2 or newer, launch an already configured Ollama evaluator without its runtime deadline using:
|
|
30
|
+
|
|
31
|
+
```sh
|
|
32
|
+
AUTOROUTER_OLLAMA_TIMEOUT_MS=0 claude-autorouter claude
|
|
33
|
+
```
|
|
34
|
+
|
|
35
|
+
Persist the setting with `claude-autorouter setup --evaluator ollama --ollama-model tev1:4b --ollama-timeout-ms 0 --force` for that installed model. The setup flag overrides the timeout environment value; later launch-time environment values still override saved configuration. Startup priming keeps its separate 60-second limit, cancellation remains active, and normal errors still use fallback.
|
|
36
|
+
|
|
37
|
+
## Local routing regression
|
|
38
|
+
|
|
39
|
+
The source-only harness below is opt-in and is not included in the npm package. It sends synthetic Claude-shaped requests through the real router and an installed local evaluator, checking task extraction, classifier choices, selected Claude tiers, and new human turns. It makes no Anthropic or Jev calls, downloads no models, and writes no user configuration.
|
|
40
|
+
|
|
41
|
+
```sh
|
|
42
|
+
node scripts/test-ollama-routing.mjs --model tev1:0.8b
|
|
43
|
+
node scripts/test-ollama-routing.mjs --model tev1:4b
|
|
44
|
+
node scripts/test-ollama-routing.mjs --model nimble:9b-q4_K_M
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
Start Ollama 0.35+ and install the selected model first. The harness refuses to run while another model is resident. It never explicitly unloads the selected model; its keep-alive setting controls residency. It warms once, then uses the production deadline for each uncached case: 1,500 ms for Tev1 0.8B/custom tags, 15,000 ms for official Tev1 4B tags, and 30,000 ms for official Nimble tags. Environment settings or `--timeout-ms N` can override the deadline; `--timeout-ms 0` disables the runtime timer while retaining cancellation and the separate warmup limit. This harness does not load saved user configuration. Use `--output artifacts/local-routing.json` to save a metadata report.
|
|
48
|
+
|
|
49
|
+
```sh
|
|
50
|
+
node scripts/test-ollama-routing.mjs --model tev1:4b --timeout-ms 0
|
|
51
|
+
```
|
|
52
|
+
|
|
53
|
+
The command fails on a wrong classification, fallback, unexpected guard override, or missing tier coverage. A Sonnet result with `source: ollama` and `classified_tier: sonnet` is a valid prediction; `source: fallback` and `classifier_error: timeout` means classification did not complete. Passing establishes these synthetic cases only. Warmup and metadata-only `doctor` checks do not establish speed or accuracy on real tasks.
|
|
54
|
+
|
|
55
|
+
Version 0.3.2 excludes Claude's executor system instructions from local classifier input before excerpt budgeting and retains task/history excerpts. Jev is unchanged. Version 0.3.1 included the executor background locally and defaulted every local model to 1,500 ms; see the [upgrade notes](reference.md#migrating-an-older-ollama-config). Historical benchmark results must remain labeled with their original excerpt policy and explicit deadlines.
|
|
56
|
+
|
|
29
57
|
## Live integration tests
|
|
30
58
|
|
|
31
|
-
|
|
59
|
+
The Claude integration tests below make real Claude calls and invoke the configured evaluator, consuming Claude usage and, with Jev, TypeSafe usage. They use temporary synthetic fixtures and disable unrelated customizations and MCP servers. With Ollama, start the local service and install the chosen model first; the harness does not install or download it.
|
|
32
60
|
|
|
33
61
|
```sh
|
|
34
62
|
npm run test:live
|
|
@@ -2,7 +2,32 @@
|
|
|
2
2
|
|
|
3
3
|
AutoRouter supports `/v1/systemone` classification only: remote Jev with an API key, or local Ollama 0.35+ with a compatible decision model. Jev remains the default evaluator. The local default is `nimble:9b-q4_K_M`; the old Qwen chat adapter and compact/quality/auto presets have been removed. Existing downloaded models are not deleted.
|
|
4
4
|
|
|
5
|
-
##
|
|
5
|
+
## Version 0.3.2 routing regression (September 30, 2026)
|
|
6
|
+
|
|
7
|
+
The npm `0.3.1` runtime used a 1,500 ms deadline for every local model, even though startup priming allows 60 seconds. On the user's already-loaded `tev1:4b` (Q8), four synthetic tasks all timed out and selected Sonnet through `source=fallback`, `classifier_error=timeout`. With a separate 30-second diagnostic allowance, the same tasks selected Haiku, Haiku, Sonnet, and Opus. Three of those decisions took 5.6–6.1 seconds. This was a deadline failure, not a parser forcing every answer to Sonnet.
|
|
8
|
+
|
|
9
|
+
We captured three synthetic request shapes from the installed Claude client using an isolated local response stub, without paid provider calls. The human task survived intact and no compatibility guard forced Sonnet. The local evaluator nevertheless received 627–796 characters of Claude's general executor instructions. Tev1 0.8B classified all three captures as Sonnet. Removing only this system background changed the distributed-fencing case to Opus; the `[].length` example still selected Sonnet. Removing additional state fields did not improve that result, so the routing rubric and remaining state layout were retained.
|
|
10
|
+
|
|
11
|
+
Version 0.3.2 excludes executor system instructions from local classifier input, uses 15-second defaults for official Tev1 4B variants and 30 seconds for Nimble, and retains the 1.5-second default for Tev1 0.8B/custom models. Explicit timeout settings still win; `0` disables only the runtime evaluator timer. Setup accepts `--ollama-timeout-ms`; setup, doctor, and startup display the effective deadline. Compact status lines retain the fallback cause. Jev's input extraction and deadline are unchanged.
|
|
12
|
+
|
|
13
|
+
The source-only, opt-in `npm run test:ollama -- --model TAG` checks actual `Router.route` results against six fixed synthetic Claude-shaped requests: literal output, `[].length`, a bounded feature, distributed fencing, a new mechanical task after a difficult task, and a new difficult task after a simple task. It checks the evaluator choice, selected Claude model, source, and reason, and fails on a mismatch, fallback, override, or missing tier coverage. Live cases use fresh routers to avoid decision-cache passes; persistent turn-cache transitions and compatibility guards are separately covered by offline regression tests. It is not included in the npm package, downloads nothing, and does not contact Claude or Jev.
|
|
14
|
+
|
|
15
|
+
One pass with the fixes on the same M4/16 GiB Mac produced:
|
|
16
|
+
|
|
17
|
+
| Installed model | Runtime deadline | Matching cases | Fallbacks | Decision latency range |
|
|
18
|
+
| --- | ---: | ---: | ---: | ---: |
|
|
19
|
+
| `tev1:0.8b` | 1,500 ms | 3 / 6 | 0 | 48–479 ms |
|
|
20
|
+
| `tev1:4b-q4_K_M` | 15,000 ms | 6 / 6 | 0 | 2.40–2.75 s |
|
|
21
|
+
| `tev1:4b` (Q8) | 15,000 ms | 6 / 6 | 0 | 2.38–2.70 s |
|
|
22
|
+
| `nimble:9b-q4_K_M` | 30,000 ms | 6 / 6 | 0 | 4.74–7.29 s |
|
|
23
|
+
|
|
24
|
+
Every model reached all three tiers. The 0.8B failures were genuine classifications: `[].length` → Sonnet, new mechanical task → Opus, new difficult task → Sonnet. Its live suite therefore **fails**, rather than treating valid native responses as proof of accuracy. These six cases are regression checks, not a new held-out quality benchmark; the samples, machine load, residency, and prompt-cache effects do not establish a quantization speed comparison or a worst-case latency bound. The larger-model budgets allow slower decisions; they do not make those models fast or prevent every timeout. No user configuration or model files were changed; the originally resident `tev1:4b` was restored after sequential model testing.
|
|
25
|
+
|
|
26
|
+
A separate check with `tev1:4b` and the runtime deadline disabled (`timeout_ms: 0`) passed the same six cases without fallback, with decision latencies of 4.37–5.42 seconds after 13.11 seconds of priming. It made no Claude or Jev calls and downloaded nothing. This validates disabled-timer routing on those cases; it adds no new quality cases or latency guarantee.
|
|
27
|
+
|
|
28
|
+
Fixture SHA-256: `88e12f962fb42b36a00edd6b8c2560e54620d90c978b1ff13a1ddf31ee46b585`. The questions remain unchanged at `be151cedb4de4b7ef3f7162d751f70ce7d9dd14efc66fae1835f73ffd04027be`.
|
|
29
|
+
|
|
30
|
+
## Historical benchmark method (before 0.3.2)
|
|
6
31
|
|
|
7
32
|
Measurements were recorded on September 29, 2026, on an Apple M4 Mac with 16 GiB of unified memory alongside other applications. Nimble ran on an isolated Ollama 0.35.0 process; the installed Ollama 0.33.3 daemon was left unchanged during that test. Tev1 was tested later on the user's upgraded Ollama 0.35.0 service after its downloads finished. Version 0.35.0 was a prerelease at the time. These are observations under different application loads, not a controlled hardware comparison. See the [official release](https://github.com/ollama/ollama/releases/tag/v0.35.0), [Nimble catalog](https://ollama.com/library/nimble), and [Tev1 catalog](https://ollama.com/library/tev1).
|
|
8
33
|
|
|
@@ -10,15 +35,15 @@ The selected Nimble tag contains a 9B Q4_K_M model, approximately 5.63 GB to dow
|
|
|
10
35
|
|
|
11
36
|
The fixture contains 36 balanced synthetic workloads: 12 tuning cases and 24 held-out cases, with equal numbers of Haiku, Sonnet, and Opus labels. Cases include mechanical edits, ordinary implementation, difficult correctness and security work, topic changes, tool results, short follow-ups, and misleading routing instructions. These labels are judgments under the routing policy, not proof of which Claude model would complete each task successfully. The native questions preserve the existing local routing policy and were frozen before Nimble testing; no changes were made from held-out results.
|
|
12
37
|
|
|
13
|
-
The harness
|
|
38
|
+
The historical harness used the then-current production state builder and evaluator, including local metadata checks in wall-clock latency. It bypasses AutoRouter's decision cache; Ollama's own caching remains enabled. A cold measurement starts with the model unloaded, but operating-system file caches and kernels may already be warm. Cold calls have a separate 60-second deadline. The normal evaluator deadline in that benchmark was 1,500 ms for every model. Reported model allocation comes from `/api/ps`; it is not a measurement of total process or system memory, and its GPU allocation is not additional independent RAM on this unified-memory Mac. Aggregate runtime RSS includes all Ollama and llama-server processes, including the original idle daemon.
|
|
14
39
|
|
|
15
|
-
## Nimble results
|
|
40
|
+
## Historical Nimble results
|
|
16
41
|
|
|
17
42
|
The first production-deadline run returned Haiku correctly for its cold mechanical task in 17.64 seconds. **All 12 warm tuning requests timed out at 1,500 ms**, with cancellation p50/p95 of 1,503/1,513 ms. No warm classification accuracy can be inferred from that run. The model's reported allocation was 5.48 GB, with an 8,194-token context. This did not meet the desired fast-routing target on the tested Mac.
|
|
18
43
|
|
|
19
44
|
A separate first-load smoke call correctly classified `[].length` as Haiku in 25.23 seconds. It checks integration, not warm performance or classifier accuracy.
|
|
20
45
|
|
|
21
|
-
The separate held-out diagnostic used a 30,000 ms deadline and one pass over 24 distinct workloads. It
|
|
46
|
+
The separate held-out diagnostic used a 30,000 ms deadline and one pass over 24 distinct workloads. It did not change the then-default 1,500 ms budget or establish performance within it.
|
|
22
47
|
|
|
23
48
|
| Measurement | Result |
|
|
24
49
|
| --- | ---: |
|
|
@@ -36,9 +61,9 @@ All eight full-excerpt diagnostic requests completed at the 30,000 ms deadline a
|
|
|
36
61
|
|
|
37
62
|
An isolated live Claude Code test also passed using the saved Enterprise subscription login, a temporary configuration with a 30,000 ms evaluator deadline, and no Jev key. Nimble selected Haiku in 12.69 seconds; Anthropic returned HTTP 200 with the expected literal response and confirmed `claude-haiku-4-5-20251001`. Only a synthetic prompt was used, with no repository files or tools. This verifies the authentication and routing integration, not general classifier accuracy. The installed Ollama service and user configuration were left unchanged.
|
|
38
63
|
|
|
39
|
-
## Tev1 results
|
|
64
|
+
## Historical Tev1 results
|
|
40
65
|
|
|
41
|
-
Both [Tev1 variants](https://ollama.com/library/tev1) use the same production adapter and frozen questions, selected through `--ollama-model`. The 0.8B tag uses Q8_0 quantization and downloads approximately 812 MB; `tev1:4b-q4_K_M` downloads approximately 2.71 GB. The unqualified `tev1` tag selects the larger 4B Q8 model, which was not tested. Jev remains the evaluator default and Nimble remains the local-model default.
|
|
66
|
+
Both [Tev1 variants](https://ollama.com/library/tev1) use the same production adapter and frozen questions, selected through `--ollama-model`. The 0.8B tag uses Q8_0 quantization and downloads approximately 812 MB; `tev1:4b-q4_K_M` downloads approximately 2.71 GB. The unqualified `tev1` tag selects the larger 4B Q8 model, which was not tested in this historical benchmark. It was tested separately in the 0.3.2 regression above. Jev remains the evaluator default and Nimble remains the local-model default.
|
|
42
67
|
|
|
43
68
|
Each measured tag ships `num_ctx:2050`. The window includes the template, routing criteria, and excerpt. The 3,000-byte state cap is not a guarantee that every possible input fits this smaller token window. The reported stress cases used 1,664–1,752 input tokens on 0.8B. Requests exceeding model limits use the usual fallback; AutoRouter does not switch protocols or silently truncate additional content for Tev1.
|
|
44
69
|
|
|
@@ -52,7 +77,7 @@ These measurements use one pass over the same 24 held-out cases and eight separa
|
|
|
52
77
|
|
|
53
78
|
The 0.8B model matched four of eight Haiku labels, all eight Sonnet labels, and six of eight Opus labels. It over-routed four mechanical tasks and under-routed two difficult tasks to Sonnet, including a case with a misleading tier instruction. Cold wall time was 2.08 seconds for 0.8B and 6.04 seconds for 4B. Model size and fast responses do not establish sufficient accuracy for an engineering workload.
|
|
54
79
|
|
|
55
|
-
With a separate 10-second deadline, 4B matched seven of eight Haiku labels, all eight Sonnet labels, and seven of eight Opus labels. It over-routed one mechanical case to Opus and under-routed one difficult case to Sonnet. Its cold diagnostic request took 3.69 seconds. The improved agreement comes with several seconds of classification latency; it is not performance at the default deadline.
|
|
80
|
+
With a separate 10-second deadline, 4B matched seven of eight Haiku labels, all eight Sonnet labels, and seven of eight Opus labels. It over-routed one mechanical case to Opus and under-routed one difficult case to Sonnet. Its cold diagnostic request took 3.69 seconds. The improved agreement comes with several seconds of classification latency; it is not performance at the then-default 1,500 ms deadline.
|
|
56
81
|
|
|
57
82
|
All eight 0.8B full-excerpt requests completed within 1,500 ms and returned Haiku, with p50/p95 of 1,079/1,150 ms. The 4B model timed out on all eight at 1,500 ms and again on all eight at 10,000 ms. Its 10-second cancellation p50/p95 was 10,006/10,081 ms; completed full-excerpt latency was not measured. The longer deadline therefore allows the reported short held-out decisions but does not guarantee completion for full excerpts.
|
|
58
83
|
|
|
@@ -67,15 +92,15 @@ Model digests:
|
|
|
67
92
|
|
|
68
93
|
## Reproducing the evaluation
|
|
69
94
|
|
|
70
|
-
Use a source checkout; benchmark scripts and fixtures are not included in the npm package. Install Ollama 0.35+, start it, and explicitly download the model:
|
|
95
|
+
Use a source checkout; benchmark scripts and fixtures are not included in the npm package. The commands below evaluate the checked-out version. To reproduce the historical excerpt policy, use the `v0.3.1` checkout; 0.3.2 changes local input extraction. The explicit deadlines retain the historical budgets. Install Ollama 0.35+, start it, and explicitly download the model:
|
|
71
96
|
|
|
72
97
|
```sh
|
|
73
98
|
ollama pull nimble:9b-q4_K_M
|
|
74
|
-
node scripts/evaluate-ollama.mjs --models nimble:9b-q4_K_M --split tuning --rounds 1 --output artifacts/nimble-tuning.json
|
|
99
|
+
node scripts/evaluate-ollama.mjs --models nimble:9b-q4_K_M --split tuning --rounds 1 --timeout-ms 1500 --output artifacts/nimble-tuning.json
|
|
75
100
|
node scripts/evaluate-ollama.mjs --models nimble:9b-q4_K_M --split heldout --rounds 1 --stress-rounds 8 --timeout-ms 30000 --output artifacts/nimble-diagnostic.json
|
|
76
101
|
ollama pull tev1:0.8b
|
|
77
102
|
ollama pull tev1:4b-q4_K_M
|
|
78
|
-
node scripts/evaluate-ollama.mjs --models tev1:0.8b,tev1:4b-q4_K_M --split heldout --rounds 1 --stress-rounds 8 --output artifacts/tev1-default.json
|
|
103
|
+
node scripts/evaluate-ollama.mjs --models tev1:0.8b,tev1:4b-q4_K_M --split heldout --rounds 1 --stress-rounds 8 --timeout-ms 1500 --output artifacts/tev1-default.json
|
|
79
104
|
node scripts/evaluate-ollama.mjs --models tev1:4b-q4_K_M --split heldout --rounds 1 --stress-rounds 8 --timeout-ms 10000 --output artifacts/tev1-diagnostic.json
|
|
80
105
|
```
|
|
81
106
|
|
|
@@ -91,6 +116,6 @@ Reproducibility identifiers:
|
|
|
91
116
|
|
|
92
117
|
This is a small synthetic rubric-agreement benchmark, not a downstream task-quality, savings, Jev-parity, or security evaluation. Classification can miss context outside the excerpt. The cases do not establish robust resistance to prompt injection. Performance depends on hardware, memory pressure, prompt length, and residency; a successful setup or simple request does not guarantee the runtime deadline.
|
|
93
118
|
|
|
94
|
-
The launcher primes the model before opening Claude, allowing up to 60 seconds for that synthetic classification. Idle unloading can still make later requests cold. Runtime timeouts and invalid responses use the existing conservative fallback, without contacting Jev. A longer `AUTOROUTER_OLLAMA_TIMEOUT_MS` trades added prompt latency for more completed local classifications; it does not make the evaluator faster.
|
|
119
|
+
The launcher primes the model before opening Claude, allowing up to 60 seconds for that synthetic classification. Idle unloading can still make later requests cold. Runtime timeouts and invalid responses use the existing conservative fallback, without contacting Jev. A longer `AUTOROUTER_OLLAMA_TIMEOUT_MS` trades added prompt latency for more completed local classifications; it does not make the evaluator faster. In 0.3.2, `0` disables the runtime timer while preserving user/disconnect cancellation, normal error fallback, and the separate startup limit.
|
|
95
120
|
|
|
96
121
|
Earlier Qwen results used a different `/api/chat` implementation and are not measurements of this native backend. They remain available in the [historical evaluation document](https://github.com/frapposelli/claude-autorouter/blob/548a175/docs/ollama-evaluation.md). Reproduce those results from that revision, not the current native-only harness.
|
package/docs/reference.md
CHANGED
|
@@ -7,6 +7,7 @@
|
|
|
7
7
|
| `claude-autorouter setup` | Save subscription-mode configuration and a Jev key |
|
|
8
8
|
| `claude-autorouter setup --auth-mode api-key` | Configure Jev and Anthropic API-key billing |
|
|
9
9
|
| `claude-autorouter setup --evaluator ollama --pull` | Configure the native local evaluator and download its selected model if missing |
|
|
10
|
+
| `claude-autorouter setup --evaluator ollama --ollama-timeout-ms 0 --force` | Save a disabled runtime evaluator deadline |
|
|
10
11
|
| `claude-autorouter setup --force` | Replace an existing user config |
|
|
11
12
|
| `claude-autorouter doctor` | Check config, Claude executable/login, and the selected local Ollama model without paid calls |
|
|
12
13
|
| `claude-autorouter claude [arguments]` | Start a local router and pass arguments through to Claude Code |
|
|
@@ -52,7 +53,7 @@ For an environment-only subscription launch, set `AUTOROUTER_AUTH_MODE=subscript
|
|
|
52
53
|
| `AUTOROUTER_JEV_TIMEOUT_MS` | `1500` | Classifier deadline in milliseconds |
|
|
53
54
|
| `AUTOROUTER_OLLAMA_URL` | `http://127.0.0.1:11434` | Loopback Ollama base URL |
|
|
54
55
|
| `AUTOROUTER_OLLAMA_MODEL` | `nimble:9b-q4_K_M` | Installed local model tag or alias compatible with `/v1/systemone` |
|
|
55
|
-
| `AUTOROUTER_OLLAMA_TIMEOUT_MS` |
|
|
56
|
+
| `AUTOROUTER_OLLAMA_TIMEOUT_MS` | model-dependent; see below | Runtime local classification deadline, `1`–`30000` ms; `0` disables it |
|
|
56
57
|
| `AUTOROUTER_OLLAMA_KEEP_ALIVE` | `5m` | How long Ollama retains the evaluator in memory |
|
|
57
58
|
| `AUTOROUTER_TOKEN_COUNT_TIMEOUT_MS` | `1500` | Context-check deadline; runs alongside classification |
|
|
58
59
|
| `AUTOROUTER_MIN_CONFIDENCE` | `0.75` | Jev confidence threshold; does not apply to Ollama |
|
|
@@ -65,7 +66,9 @@ Model access depends on your account. The policy recognizes specific Claude mode
|
|
|
65
66
|
|
|
66
67
|
## Ollama evaluator
|
|
67
68
|
|
|
68
|
-
|
|
69
|
+
The local configuration documented here requires AutoRouter 0.3.2 or newer and remains experimental. It uses Ollama's native `/v1/systemone` decision endpoint for every model, replacing the chat backend from 0.2.0. Jev remains the default remote evaluator, using TypeSafe's `/v1/systemone` endpoint and a TypeSafe API key. Selecting Ollama never silently switches back to Jev. Haiku, Sonnet, or Opus still completes the task through Anthropic.
|
|
70
|
+
|
|
71
|
+
Version 0.3.2 excludes Claude's executor system instructions from local excerpts, uses model-specific runtime deadlines, and accepts `0` to disable that deadline. Setup, doctor, and startup show the effective model and deadline; setup accepts `--ollama-timeout-ms`. Jev is unchanged.
|
|
69
72
|
|
|
70
73
|
All local models require Ollama 0.35 or newer. Version 0.35.0 is a prerelease as of September 29, 2026; it introduces the native decision API. See the [Ollama release notes](https://github.com/ollama/ollama/releases/tag/v0.35.0). Install and start a compatible local service, then run:
|
|
71
74
|
|
|
@@ -95,31 +98,45 @@ claude-autorouter setup --evaluator ollama --ollama-model tev1:0.8b --pull --for
|
|
|
95
98
|
```
|
|
96
99
|
|
|
97
100
|
```sh
|
|
98
|
-
# Tev1 4B Q4_K_M,
|
|
99
|
-
|
|
101
|
+
# Tev1 4B Q4_K_M, with a 15-second default deadline
|
|
102
|
+
claude-autorouter setup --evaluator ollama --ollama-model tev1:4b-q4_K_M --pull --force
|
|
100
103
|
```
|
|
101
104
|
|
|
102
|
-
For Nimble, the explicit Q4_K_M tag avoids `nimble:latest`, which currently selects an approximately 9.5 GB Q8 model. For Tev1, `tev1:latest` and `tev1:4b` select approximately 4.5 GB Q8 weights; the explicit `tev1:4b-q4_K_M` tag selects the smaller 4B download. Download size is not resident memory: runtime and context allocations add to it, and other applications need memory too. Downloaded models have their own licenses and are not bundled in this package.
|
|
105
|
+
For Nimble, the explicit Q4_K_M tag avoids `nimble:latest`, which currently selects an approximately 9.5 GB Q8 model. For Tev1, `tev1:latest` and `tev1:4b` select approximately 4.5 GB Q8 weights; the explicit `tev1:4b-q4_K_M` tag selects the smaller 4B download. Download size is not resident memory: runtime and context allocations add to it, and other applications need memory too. Downloaded models have their own licenses and are not bundled in this package. In historical tests before 0.3.2 on a 16 GiB M4, Tev1 0.8B matched 18/24 held-out labels at 450 ms median latency within 1,500 ms; 4B matched 22/24 at 3.15 seconds with a separate 10-second deadline. See the [local measurements](ollama-evaluation.md) before choosing a latency deadline.
|
|
103
106
|
|
|
104
107
|
The endpoint must be loopback (`127.0.0.1`, `localhost`, or `::1`), without a path, credentials, query, or fragment. Cloud model tags and metadata identifying a remote model are rejected before sending task text. Claude and Jev credentials are never attached to Ollama requests.
|
|
105
108
|
|
|
106
109
|
### Classification and fallback
|
|
107
110
|
|
|
108
|
-
Local classification caps serialized evaluator state at both 3,000 characters and 3,000 UTF-8 bytes, including for non-ASCII prompts. `/v1/systemone` receives the bounded state and routing criteria and returns a tier directly. The router retains each model's native context setting: 8,194 tokens for the default Nimble tag and 2,050 for the listed Tev1 tags. Tev1's smaller window includes the routing criteria and template as well as the excerpt; the byte limit does not guarantee every possible input fits. Context errors use the normal fallback. Returned confidence scores summarize choice-distribution entropy; they are not calibrated accuracy probabilities. `AUTOROUTER_MIN_CONFIDENCE` applies only to Jev. All capability, tool-continuation, thinking, and context guards still apply.
|
|
111
|
+
Local classification caps serialized evaluator state at both 3,000 characters and 3,000 UTF-8 bytes, including for non-ASCII prompts. Claude's top-level executor system instructions are excluded before budgeting; the current task, original task, and recent conversation excerpts remain. `/v1/systemone` receives the bounded state and routing criteria and returns a tier directly. The router retains each model's native context setting: 8,194 tokens for the default Nimble tag and 2,050 for the listed Tev1 tags. Tev1's smaller window includes the routing criteria and template as well as the excerpt; the byte limit does not guarantee every possible input fits. Context errors use the normal fallback. Returned confidence scores summarize choice-distribution entropy; they are not calibrated accuracy probabilities. `AUTOROUTER_MIN_CONFIDENCE` applies only to Jev. All capability, tool-continuation, thinking, and context guards still apply.
|
|
109
112
|
|
|
110
|
-
|
|
113
|
+
The deadline covering local checks and classification defaults to 1,500 ms for Tev1 0.8B and custom/unrecognized tags, 15,000 ms for official Tev1 4B variants (including bare `tev1` and `latest`), and 30,000 ms for official Nimble variants. Official `library/` and `registry.ollama.ai/` aliases are recognized; a custom namespace such as `team/nimble` keeps the short default. An explicit timeout overrides the model default, including an old saved `1500`. Environment values override saved values on launch. Defaults are not written into the user config; `setup --force` replaces the config and saves an explicit timeout when supplied through `--ollama-timeout-ms` or the environment. During setup, the command-line flag takes precedence over the timeout environment value.
|
|
111
114
|
|
|
112
|
-
|
|
115
|
+
Set `AUTOROUTER_OLLAMA_TIMEOUT_MS=0` to remove AutoRouter's runtime evaluator timer while keeping the existing configuration:
|
|
113
116
|
|
|
114
|
-
|
|
117
|
+
```sh
|
|
118
|
+
AUTOROUTER_OLLAMA_TIMEOUT_MS=0 claude-autorouter claude
|
|
119
|
+
```
|
|
120
|
+
|
|
121
|
+
To persist it for an already installed Tev1 4B model, run:
|
|
115
122
|
|
|
116
123
|
```sh
|
|
117
|
-
|
|
124
|
+
claude-autorouter setup --evaluator ollama --ollama-model tev1:4b --ollama-timeout-ms 0 --force
|
|
118
125
|
```
|
|
119
126
|
|
|
127
|
+
Only the runtime evaluator deadline is disabled. User cancellation and client disconnection still abort evaluation; ordinary service, HTTP, and response errors still use fallback. Startup priming retains its separate 60-second limit, and lifecycle checks retain their own limits. Positive values from `1` to `30000` keep a finite deadline: for example, `--ollama-timeout-ms 2500` permits 2.5 seconds and can still time out on decisions near that cutoff.
|
|
128
|
+
|
|
129
|
+
Before opening Claude's UI, the launcher loads an installed model and primes the classifier rubric with a synthetic task, using a separate deadline of up to 60 seconds. Successful priming does not establish that real excerpts finish within an enabled runtime deadline or classify correctly. `doctor` checks version and model availability without inference; it does not certify speed or accuracy either. `AUTOROUTER_OLLAMA_KEEP_ALIVE` defaults to `5m`. After five idle minutes, the next request may need to reload the model and, when a runtime deadline is enabled, exceed it and use the fallback. A longer positive keep-alive reduces some reloads while retaining memory longer; keep-alive `0` unloads immediately and can make every evaluation cold. Supported keep-alive values are `0` or a positive duration such as `30s`, `5m`, or `1h`.
|
|
130
|
+
|
|
131
|
+
If startup priming fails, the launcher warns and continues. An incompatible model or Ollama version, missing model, unavailable service, malformed answer, or evaluation timeout falls back to Sonnet or retains an incoming Opus, subject to the usual compatibility policy. No Jev request is made. `Ollama fallback` with `timeout` means no valid classification completed in time; it is not a Sonnet prediction. In metadata, a valid Sonnet decision has `source: "ollama"` and `classified_tier: "sonnet"`; a timeout has `source: "fallback"` and `classifier_error: "timeout"`. Later policy guards can still change the selected Claude model. Use the source-only [local routing regression](development.md#local-routing-regression) to test classification and all three selected tiers without external provider calls.
|
|
132
|
+
|
|
133
|
+
Historical measurements before 0.3.2: Tev1 4B timed out on all eight full-excerpt checks even with a 10-second diagnostic allowance; its short-task results did not establish a full-excerpt latency bound. Tev1 0.8B completed all eight within 1,500 ms. On the tested 16 GiB M4, Nimble timed out on all 12 tuning requests at 1,500 ms. A separate 30-second diagnostic completed 24 held-out classifications with 23 matching labels, but median routing took 11.4 seconds. The one error followed a misleading tier instruction. These historical results precede the 0.3.2 excerpt changes and do not establish guarantees for the longer defaults. See the [measurements and limitations](ollama-evaluation.md).
|
|
134
|
+
|
|
120
135
|
### Migrating an older Ollama config
|
|
121
136
|
|
|
122
|
-
Version 0.3.1
|
|
137
|
+
Version 0.3.1 used a 1,500 ms deadline for every local model. After upgrading to 0.3.2, an explicitly saved or exported `AUTOROUTER_OLLAMA_TIMEOUT_MS=1500` still wins over the new model-specific defaults. Remove that override to use the defaults, or rerun setup with the desired model and `--ollama-timeout-ms N --force`. The `0` value and setup timeout flag require 0.3.2 or newer.
|
|
138
|
+
|
|
139
|
+
Version 0.3.1 removed the Qwen chat backend and presets from 0.2.0. Existing downloaded models remain on disk, but an old Qwen model selection needs to be replaced with a native decision model. Run the setup command above with `--force`; it selects Nimble unless you pass `--ollama-model` or override the model through the environment. Remove or update any old `AUTOROUTER_OLLAMA_MODEL` environment value too, because environment variables override saved configuration. Update scripts to use `--ollama-model` when selecting a custom model.
|
|
123
140
|
|
|
124
141
|
## Data flow and authentication
|
|
125
142
|
|
|
@@ -131,7 +148,7 @@ Claude Code → authenticated local gateway → Jev or local Ollama classificati
|
|
|
131
148
|
|
|
132
149
|
AutoRouter uses Claude Code's [gateway integration](https://code.claude.com/docs/en/llm-gateway-protocol), so it sees inference requests and tool continuations. It does not rely on a user-prompt hook.
|
|
133
150
|
|
|
134
|
-
The selected evaluator receives a bounded state containing the latest human request and excerpts of the original task
|
|
151
|
+
The selected evaluator receives a bounded state containing the latest human request and excerpts of the original task and recent messages: up to 12,000 serialized characters sent to TypeSafe for Jev, or 3,000 UTF-8 bytes sent to the local Ollama service. Jev also receives system-text excerpts. The local path excludes Claude's top-level executor system instructions. These excerpts can include private source code and tool results. Images, document payloads, and signed thinking are omitted. Full tool schemas and full conversation history are not sent to either classifier. Anthropic receives the complete request, including its tools and attachments. Large or multimodal requests may also go to Anthropic's token-count endpoint before inference, including when classification is local.
|
|
135
152
|
|
|
136
153
|
In subscription mode, Claude Code owns login and OAuth refresh. AutoRouter forwards the current request's authorization and beta headers to Anthropic. It does not read keychain or saved login files, persist subscription tokens, or send them to Jev. A separate temporary `X-Autorouter-Token` authenticates the local connection and is stripped upstream. Subscription forwarding is restricted to `https://api.anthropic.com`. See [subscriptions and gateways](https://code.claude.com/docs/en/llm-gateway#subscriptions-and-gateways).
|
|
137
154
|
|
package/docs/releasing.md
CHANGED
|
@@ -4,6 +4,8 @@ The package is `claude-autorouter`, licensed under [Apache-2.0](../LICENSE). Ver
|
|
|
4
4
|
|
|
5
5
|
Version `0.3.1` replaces the old Ollama chat evaluator and Qwen presets with the native `/v1/systemone` endpoint on Ollama 0.35+. It supports Nimble, Tev1, and other compatible local models through `--ollama-model`; Jev remains the default remote evaluator. Existing local users should rerun setup with a supported model, as described in the [reference](reference.md#ollama-evaluator). The [evaluation report](ollama-evaluation.md) records local model latency, accuracy, and timeout limitations.
|
|
6
6
|
|
|
7
|
+
Version `0.3.2` fixes local timeout fallbacks with model-specific deadlines, removes Claude executor instructions from local evaluator excerpts, and keeps fallback causes visible in compact status lines. It also adds `AUTOROUTER_OLLAMA_TIMEOUT_MS=0` and `setup --ollama-timeout-ms 0` to disable the runtime evaluation deadline while preserving caller cancellation and the separate startup warmup limit. Existing explicit timeout settings still override the defaults; Jev is unchanged.
|
|
8
|
+
|
|
7
9
|
The GitHub repository is private. Publishing to npm makes the tarball's runtime source, README, configuration example, license, and shipped documentation public. Model weights, user configuration, credentials, transcripts, local artifacts, and test fixtures are excluded. Review the archive before the first publication and whenever the package allowlist changes.
|
|
8
10
|
|
|
9
11
|
## What runs automatically
|
|
@@ -95,12 +97,12 @@ After a successful trusted release, npm recommends the optional **Publishing acc
|
|
|
95
97
|
|
|
96
98
|
## 3. Release subsequent versions by tag
|
|
97
99
|
|
|
98
|
-
|
|
100
|
+
The commands below illustrate the `0.3.2` release. For a new release, substitute the next unused version throughout; never reuse a published version:
|
|
99
101
|
|
|
100
102
|
```sh
|
|
101
103
|
git switch main
|
|
102
104
|
git pull --ff-only origin main
|
|
103
|
-
npm version 0.3.
|
|
105
|
+
npm version 0.3.2 --no-git-tag-version
|
|
104
106
|
```
|
|
105
107
|
|
|
106
108
|
Review the version change and update any version-specific install examples or release notes. Check the candidate using the new filename:
|
|
@@ -109,7 +111,7 @@ Review the version change and update any version-specific install examples or re
|
|
|
109
111
|
npm run check
|
|
110
112
|
npm test
|
|
111
113
|
npm run release:pack
|
|
112
|
-
npm run test:package -- --archive ./dist/claude-autorouter-0.3.
|
|
114
|
+
npm run test:package -- --archive ./dist/claude-autorouter-0.3.2.tgz
|
|
113
115
|
git diff --check
|
|
114
116
|
```
|
|
115
117
|
|
|
@@ -117,7 +119,7 @@ Commit the intended release changes and get that commit onto `main`, either thro
|
|
|
117
119
|
|
|
118
120
|
```sh
|
|
119
121
|
git add package.json
|
|
120
|
-
git commit -m "Release 0.3.
|
|
122
|
+
git commit -m "Release 0.3.2"
|
|
121
123
|
git push origin main
|
|
122
124
|
```
|
|
123
125
|
|
|
@@ -126,24 +128,24 @@ Include any intentional documentation or release-note edits in that commit too.
|
|
|
126
128
|
```sh
|
|
127
129
|
git switch main
|
|
128
130
|
git pull --ff-only origin main
|
|
129
|
-
git tag -a v0.3.
|
|
130
|
-
git push origin v0.3.
|
|
131
|
+
git tag -a v0.3.2 -m "Release 0.3.2"
|
|
132
|
+
git push origin v0.3.2
|
|
131
133
|
```
|
|
132
134
|
|
|
133
|
-
Before pushing, confirm `package.json` contains `0.3.
|
|
135
|
+
Before pushing, confirm `package.json` contains `0.3.2` and the tag points to the intended commit. For a prerelease, use a matching version/tag such as `0.4.0-beta.1` / `v0.4.0-beta.1`; it will publish under `next`, leaving `latest` unchanged.
|
|
134
136
|
|
|
135
137
|
Release stable versions in increasing version order, one tag at a time, and wait for each run to finish before pushing the next stable tag. The workflow queues releases without canceling an active run, but queue order does not sort semantic versions. Publishing an older stable version afterward could move `latest` backward; there is no registry version-order gate.
|
|
136
138
|
|
|
137
139
|
Open the tag's run under [GitHub Actions](https://github.com/frapposelli/claude-autorouter/actions). Under **Artifacts**, download `npm-package-<run-id>-<run-attempt>`, which contains the `.tgz` and checksum used for publication. Artifacts expire after 30 days, so retain them with the release record. After the publish job succeeds, verify the registry version and tags:
|
|
138
140
|
|
|
139
141
|
```sh
|
|
140
|
-
npm view claude-autorouter@0.3.
|
|
142
|
+
npm view claude-autorouter@0.3.2 version dist.integrity --registry https://registry.npmjs.org/
|
|
141
143
|
npm view claude-autorouter dist-tags --json --registry https://registry.npmjs.org/
|
|
142
144
|
```
|
|
143
145
|
|
|
144
146
|
Repeat the independent installation check for the released version. A GitHub Release page is optional; pushing the version tag is the publication trigger.
|
|
145
147
|
|
|
146
|
-
For local release diagnostics after the tag exists, `node scripts/release-check.mjs source v0.3.
|
|
148
|
+
For local release diagnostics after the tag exists, `node scripts/release-check.mjs source v0.3.2` checks the tag, clean checkout, metadata, and ancestry. `node scripts/release-check.mjs archive v0.3.2` checks the candidate checksum and contents; `dist/` must contain only that version's archive and checksum, so retain older artifacts elsewhere first. These helpers are run automatically in the release workflow; the first untagged bootstrap uses the checks in step 1 instead.
|
|
147
149
|
|
|
148
150
|
## Recovering a failed release
|
|
149
151
|
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "claude-autorouter",
|
|
3
|
-
"version": "0.3.
|
|
3
|
+
"version": "0.3.2",
|
|
4
4
|
"license": "Apache-2.0",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"description": "A Claude Code model router with Jev and local Ollama System One evaluators",
|
|
@@ -16,6 +16,7 @@
|
|
|
16
16
|
"claude": "node bin/autorouter.mjs claude",
|
|
17
17
|
"test": "node --test test/*.test.mjs",
|
|
18
18
|
"test:package": "node scripts/package-smoke.mjs",
|
|
19
|
+
"test:ollama": "node scripts/test-ollama-routing.mjs",
|
|
19
20
|
"release:pack": "node scripts/release-pack.mjs",
|
|
20
21
|
"eval": "node --env-file=.env scripts/evaluate.mjs",
|
|
21
22
|
"eval:ollama": "node scripts/evaluate-ollama.mjs",
|
package/src/config.mjs
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { DEFAULT_OLLAMA_MODEL, validateOllamaEndpoint, validateOllamaModel } from './ollama-models.mjs';
|
|
1
|
+
import { DEFAULT_OLLAMA_MODEL, defaultOllamaTimeoutMs, validateOllamaEndpoint, validateOllamaModel } from './ollama-models.mjs';
|
|
2
2
|
|
|
3
3
|
export const TIERS = ['haiku', 'sonnet', 'opus'];
|
|
4
4
|
|
|
@@ -22,6 +22,14 @@ function endpoint(value, name) {
|
|
|
22
22
|
export function readConfig(env = process.env) {
|
|
23
23
|
const evaluator = env.AUTOROUTER_EVALUATOR ?? 'jev';
|
|
24
24
|
if (!['jev', 'ollama'].includes(evaluator)) throw new Error('AUTOROUTER_EVALUATOR must be jev or ollama');
|
|
25
|
+
const ollamaModel = validateOllamaModel(env.AUTOROUTER_OLLAMA_MODEL ?? DEFAULT_OLLAMA_MODEL);
|
|
26
|
+
const rawOllamaTimeout = env.AUTOROUTER_OLLAMA_TIMEOUT_MS;
|
|
27
|
+
// Zero is an explicit opt-out. Reject blanks, coercible non-numbers, and
|
|
28
|
+
// non-integer strings (including exponents that could underflow to zero).
|
|
29
|
+
if (rawOllamaTimeout !== undefined && (!['string', 'number'].includes(typeof rawOllamaTimeout)
|
|
30
|
+
|| (typeof rawOllamaTimeout === 'string' && !/^[0-9]+$/.test(rawOllamaTimeout.trim())))) {
|
|
31
|
+
throw new Error('AUTOROUTER_OLLAMA_TIMEOUT_MS must be an integer between 0 and 30000 (0 disables the runtime deadline)');
|
|
32
|
+
}
|
|
25
33
|
const ollamaKeepAlive = env.AUTOROUTER_OLLAMA_KEEP_ALIVE ?? '5m';
|
|
26
34
|
if (!/^(?:0|[1-9]\d{0,3}(?:s|m|h))$/.test(ollamaKeepAlive)) {
|
|
27
35
|
throw new Error('AUTOROUTER_OLLAMA_KEEP_ALIVE must be 0 or a positive duration such as 5m');
|
|
@@ -49,8 +57,8 @@ export function readConfig(env = process.env) {
|
|
|
49
57
|
jevEndpoint: endpoint(env.AUTOROUTER_JEV_URL ?? 'https://api.typesafe.ai/v1/systemone', 'AUTOROUTER_JEV_URL'),
|
|
50
58
|
jevModel: env.AUTOROUTER_JEV_MODEL ?? 'jev-latest',
|
|
51
59
|
ollamaEndpoint: validateOllamaEndpoint(env.AUTOROUTER_OLLAMA_URL ?? 'http://127.0.0.1:11434'),
|
|
52
|
-
ollamaModel
|
|
53
|
-
ollamaTimeoutMs: number(env, 'AUTOROUTER_OLLAMA_TIMEOUT_MS',
|
|
60
|
+
ollamaModel,
|
|
61
|
+
ollamaTimeoutMs: number(env, 'AUTOROUTER_OLLAMA_TIMEOUT_MS', defaultOllamaTimeoutMs(ollamaModel), 0, 30000),
|
|
54
62
|
ollamaStateChars: 3000,
|
|
55
63
|
ollamaKeepAlive,
|
|
56
64
|
models: {
|
package/src/ollama-evaluator.mjs
CHANGED
|
@@ -4,11 +4,17 @@ import { buildState } from './prompt-state.mjs';
|
|
|
4
4
|
// Bound UTF-8 bytes as well as serialized characters to keep local decision
|
|
5
5
|
// excerpts small, including when the prompt contains non-ASCII text.
|
|
6
6
|
export function buildOllamaState(body, limit = 3000) {
|
|
7
|
+
// Claude's executor instructions describe the assistant and its tools, not
|
|
8
|
+
// the current task's difficulty. Small local models can mistake that global
|
|
9
|
+
// engineering background for a request and select Sonnet for every tier.
|
|
10
|
+
// Keep task/history extraction shared with Jev, but exclude this background
|
|
11
|
+
// before budgeting so it cannot displace a useful follow-up or tool result.
|
|
12
|
+
const taskBody = { ...body, system: undefined };
|
|
7
13
|
let budget = limit;
|
|
8
|
-
let state = buildState(
|
|
14
|
+
let state = buildState(taskBody, budget);
|
|
9
15
|
while (Buffer.byteLength(JSON.stringify(state)) > limit && budget > 200) {
|
|
10
16
|
budget = Math.max(200, Math.floor(budget * limit / Buffer.byteLength(JSON.stringify(state))) - 1);
|
|
11
|
-
state = buildState(
|
|
17
|
+
state = buildState(taskBody, budget);
|
|
12
18
|
}
|
|
13
19
|
return state;
|
|
14
20
|
}
|
|
@@ -122,8 +128,15 @@ export async function checkLocalOllamaModel(config, options) {
|
|
|
122
128
|
}
|
|
123
129
|
|
|
124
130
|
export async function evaluateOllama(state, config, { fetchImpl = fetch, signal } = {}) {
|
|
125
|
-
|
|
126
|
-
|
|
131
|
+
let combined = signal;
|
|
132
|
+
if (config.ollamaTimeoutMs !== 0) {
|
|
133
|
+
const timeout = AbortSignal.timeout(config.ollamaTimeoutMs);
|
|
134
|
+
combined = signal ? AbortSignal.any([signal, timeout]) : timeout;
|
|
135
|
+
}
|
|
136
|
+
// A disabled deadline still observes caller cancellation during metadata,
|
|
137
|
+
// inference, and response reading. Direct callers may omit a signal.
|
|
138
|
+
combined ??= new AbortController().signal;
|
|
139
|
+
combined.throwIfAborted();
|
|
127
140
|
const options = { fetchImpl, signal: combined };
|
|
128
141
|
await checkLocalOllamaModel(config, options);
|
|
129
142
|
return decisionAnswer(await request(config, '/v1/systemone', buildOllamaRequest(state, config), options), config.ollamaModel);
|
package/src/ollama-models.mjs
CHANGED
|
@@ -1,5 +1,20 @@
|
|
|
1
1
|
export const DEFAULT_OLLAMA_MODEL = 'nimble:9b-q4_K_M';
|
|
2
2
|
|
|
3
|
+
// Known model families need different runtime budgets. Restrict recognition to
|
|
4
|
+
// the official library and standard quantization tags; custom namespaces and
|
|
5
|
+
// unknown variants retain the short default. Explicit configuration wins.
|
|
6
|
+
const QUANTIZATION = '(?:q[2-8]_(?:0|1|k(?:_[sml])?)|iq[1-4]_(?:xxs|xs|s|m|nl)|f16|bf16|f32)';
|
|
7
|
+
const TEV_4B = new RegExp(`^(?:latest|4b(?:-${QUANTIZATION})?)$`, 'i');
|
|
8
|
+
const NIMBLE_9B = new RegExp(`^(?:latest|9b(?:-${QUANTIZATION})?)$`, 'i');
|
|
9
|
+
|
|
10
|
+
export function defaultOllamaTimeoutMs(model) {
|
|
11
|
+
const canonical = model.replace(/^registry\.ollama\.ai\//, '').replace(/^library\//, '');
|
|
12
|
+
const match = /^(tev1|nimble)(?::([^:]+))?$/.exec(canonical);
|
|
13
|
+
if (match?.[1] === 'tev1' && TEV_4B.test(match[2] ?? 'latest')) return 15000;
|
|
14
|
+
if (match?.[1] === 'nimble' && NIMBLE_9B.test(match[2] ?? 'latest')) return 30000;
|
|
15
|
+
return 1500;
|
|
16
|
+
}
|
|
17
|
+
|
|
3
18
|
export function validateOllamaEndpoint(value) {
|
|
4
19
|
let endpoint;
|
|
5
20
|
try { endpoint = new URL(value); } catch {}
|
package/src/onboarding.mjs
CHANGED
|
@@ -11,6 +11,9 @@ import { inspectOllama, setupOllama } from './ollama-setup.mjs';
|
|
|
11
11
|
|
|
12
12
|
const execute = promisify(execFile);
|
|
13
13
|
|
|
14
|
+
export const ollamaDeadlineText = timeoutMs => timeoutMs === 0
|
|
15
|
+
? 'routing deadline disabled' : `routing deadline ${timeoutMs} ms per request`;
|
|
16
|
+
|
|
14
17
|
// Readline manages editing and restores terminal state; its output is discarded
|
|
15
18
|
// so neither typing nor pasted credentials are echoed to the terminal.
|
|
16
19
|
export async function askSecret(label, { input = process.stdin, output = process.stderr } = {}) {
|
|
@@ -36,19 +39,27 @@ export async function setup(args, {
|
|
|
36
39
|
let authMode = env.AUTOROUTER_AUTH_MODE ?? 'subscription';
|
|
37
40
|
let evaluator = env.AUTOROUTER_EVALUATOR ?? 'jev';
|
|
38
41
|
let model;
|
|
42
|
+
let ollamaTimeoutMs;
|
|
39
43
|
let pull = false;
|
|
40
44
|
let overwrite = false;
|
|
41
45
|
for (let i = 0; i < args.length; i++) {
|
|
42
46
|
if (args[i] === '--auth-mode') authMode = args[++i];
|
|
43
47
|
else if (args[i] === '--evaluator') evaluator = args[++i];
|
|
44
48
|
else if (args[i] === '--ollama-model') { model = args[++i]; if (model === undefined) throw new Error('--ollama-model requires a model tag'); }
|
|
49
|
+
else if (args[i] === '--ollama-timeout-ms') {
|
|
50
|
+
const value = args[++i];
|
|
51
|
+
if (typeof value !== 'string' || !/^[0-9]+$/.test(value.trim()) || Number(value) > 30000) {
|
|
52
|
+
throw new Error('--ollama-timeout-ms requires an integer between 0 and 30000 (0 disables the routing deadline)');
|
|
53
|
+
}
|
|
54
|
+
ollamaTimeoutMs = String(Number(value));
|
|
55
|
+
}
|
|
45
56
|
else if (args[i] === '--pull') pull = true;
|
|
46
57
|
else if (args[i] === '--force') overwrite = true;
|
|
47
|
-
else throw new Error('Usage: claude-autorouter setup [--auth-mode subscription|api-key] [--evaluator jev|ollama] [--ollama-model TAG] [--pull] [--force]');
|
|
58
|
+
else throw new Error('Usage: claude-autorouter setup [--auth-mode subscription|api-key] [--evaluator jev|ollama] [--ollama-model TAG] [--ollama-timeout-ms N] [--pull] [--force]');
|
|
48
59
|
}
|
|
49
60
|
if (!['subscription', 'api-key'].includes(authMode)) throw new Error('--auth-mode must be subscription or api-key');
|
|
50
61
|
if (!['jev', 'ollama'].includes(evaluator)) throw new Error('--evaluator must be jev or ollama');
|
|
51
|
-
if (evaluator !== 'ollama' && (model !== undefined || pull)) throw new Error('Ollama model and download options require --evaluator ollama');
|
|
62
|
+
if (evaluator !== 'ollama' && (model !== undefined || ollamaTimeoutMs !== undefined || pull)) throw new Error('Ollama model, deadline and download options require --evaluator ollama');
|
|
52
63
|
const path = getConfigPath(env);
|
|
53
64
|
if (!overwrite && existsSync(path)) throw new Error('AutoRouter configuration already exists. Use setup --force to replace it.');
|
|
54
65
|
write(evaluator === 'ollama'
|
|
@@ -60,7 +71,7 @@ export async function setup(args, {
|
|
|
60
71
|
for (const key of ['AUTOROUTER_OLLAMA_URL', 'AUTOROUTER_OLLAMA_TIMEOUT_MS', 'AUTOROUTER_OLLAMA_KEEP_ALIVE']) {
|
|
61
72
|
if (env[key] !== undefined) values[key] = env[key];
|
|
62
73
|
}
|
|
63
|
-
|
|
74
|
+
if (ollamaTimeoutMs !== undefined) values.AUTOROUTER_OLLAMA_TIMEOUT_MS = ollamaTimeoutMs;
|
|
64
75
|
}
|
|
65
76
|
const keys = [...(evaluator === 'jev' ? ['TYPESAFE_API_KEY'] : []), ...(authMode === 'api-key' ? ['ANTHROPIC_API_KEY'] : [])];
|
|
66
77
|
for (const key of keys) {
|
|
@@ -71,6 +82,7 @@ export async function setup(args, {
|
|
|
71
82
|
const config = readConfig(values);
|
|
72
83
|
requireKeys(config);
|
|
73
84
|
if (evaluator === 'ollama') {
|
|
85
|
+
write(`Local evaluator: ${config.ollamaModel}; ${ollamaDeadlineText(config.ollamaTimeoutMs)}.`);
|
|
74
86
|
const controller = new AbortController();
|
|
75
87
|
const cancel = () => controller.abort();
|
|
76
88
|
if (!signal) for (const name of ['SIGINT', 'SIGTERM']) process.once(name, cancel);
|
|
@@ -104,6 +116,8 @@ export async function doctor({ env = process.env, write = console.log, run = exe
|
|
|
104
116
|
report(false, `Unset ${key}; AutoRouter uses the Anthropic Messages API`);
|
|
105
117
|
}
|
|
106
118
|
if (config?.evaluator === 'ollama') {
|
|
119
|
+
write(`Local evaluator: ${config.ollamaModel}; ${ollamaDeadlineText(config.ollamaTimeoutMs)}.`);
|
|
120
|
+
write('Model availability is checked below; classification speed and accuracy are not tested.');
|
|
107
121
|
try {
|
|
108
122
|
const result = await inspectOllama(config, { fetchImpl, signal });
|
|
109
123
|
report(result.installed, result.installed
|
package/src/statusline.mjs
CHANGED
|
@@ -6,6 +6,9 @@ const REASONS = {
|
|
|
6
6
|
context_capacity: 'large context',
|
|
7
7
|
internal_request: 'internal request', unknown_model: 'custom model', low_confidence: 'low confidence',
|
|
8
8
|
};
|
|
9
|
+
const CLASSIFIER_ERRORS = {
|
|
10
|
+
timeout: 'timeout', http_error: 'HTTP error', invalid_response: 'invalid response', network_error: 'network error',
|
|
11
|
+
};
|
|
9
12
|
const clean = (value, limit = 64) => typeof value === 'string' ? value
|
|
10
13
|
.replace(/\x1b\][\s\S]*?(?:\x07|\x1b\\)/g, '')
|
|
11
14
|
.replace(/\x1b\[[0-?]*[ -/]*[@-~]/g, '')
|
|
@@ -125,21 +128,28 @@ export function renderStatusLine(input, snapshot, { now = Date.now(), color = tr
|
|
|
125
128
|
|
|
126
129
|
const details = [];
|
|
127
130
|
const source = ['jev', 'ollama', 'cache', 'fallback'].includes(state?.source) ? state.source : undefined;
|
|
131
|
+
const evaluator = ['jev', 'ollama'].includes(state?.evaluator) ? state.evaluator : undefined;
|
|
132
|
+
const evaluatorLabel = evaluator === 'ollama' ? 'Ollama' : evaluator === 'jev' ? 'Jev' : '';
|
|
133
|
+
let fallbackCause = '';
|
|
134
|
+
let fallbackPhase = '';
|
|
135
|
+
let compactFallback = false;
|
|
128
136
|
if (source) {
|
|
129
|
-
const evaluator = ['jev', 'ollama'].includes(state.evaluator) ? state.evaluator : undefined;
|
|
130
|
-
const evaluatorLabel = evaluator === 'ollama' ? 'Ollama' : evaluator === 'jev' ? 'Jev' : '';
|
|
131
137
|
const sourceLabel = source === 'jev' ? 'Jev' : source === 'ollama' ? 'Ollama'
|
|
132
138
|
: evaluatorLabel ? `${evaluatorLabel} ${source}` : source;
|
|
133
139
|
const timing = Number.isFinite(state.latency_ms) && state.latency_ms >= 0 ? ` ${Math.round(Math.min(state.latency_ms, 999999))}ms` : '';
|
|
134
140
|
const classified = ['haiku', 'sonnet', 'opus'].includes(state.classified_tier) ? state.classified_tier : undefined;
|
|
135
141
|
const chosenFamily = /^claude-(haiku|sonnet|opus)-/.exec(state.selected_model ?? '')?.[1];
|
|
136
|
-
const override = classified && chosenFamily && classified !== chosenFamily
|
|
142
|
+
const override = source !== 'fallback' && classified && chosenFamily && classified !== chosenFamily
|
|
137
143
|
? `→${classified[0].toUpperCase()}${classified.slice(1)}` : '';
|
|
138
144
|
details.push(`${sourceLabel}${override}${timing}`);
|
|
139
145
|
}
|
|
140
146
|
if (source === 'fallback') {
|
|
141
|
-
|
|
142
|
-
|
|
147
|
+
// Snapshots normally contain allowlisted categories, but the renderer also
|
|
148
|
+
// rejects raw error messages so paths and provider response text stay out.
|
|
149
|
+
fallbackCause = Object.hasOwn(CLASSIFIER_ERRORS, state.classifier_error) ? CLASSIFIER_ERRORS[state.classifier_error] : '';
|
|
150
|
+
if (state.classifier_error === 'http_error' && Number.isInteger(state.classifier_status)
|
|
151
|
+
&& state.classifier_status >= 100 && state.classifier_status <= 599) fallbackCause = `HTTP ${state.classifier_status}`;
|
|
152
|
+
if (fallbackCause) details.push(fallbackCause);
|
|
143
153
|
}
|
|
144
154
|
if (Object.hasOwn(REASONS, state?.reason)) details.push(state.reason === 'context_capacity' && state.context_check === 'count_unavailable'
|
|
145
155
|
? 'size unverified' : REASONS[state.reason]);
|
|
@@ -157,11 +167,23 @@ export function renderStatusLine(input, snapshot, { now = Date.now(), color = tr
|
|
|
157
167
|
if (width(plain()) > available && source !== 'fallback' && phase !== 'error') detail = '';
|
|
158
168
|
if (width(plain()) > available && saving) saving = savings.compact;
|
|
159
169
|
if (width(plain()) > available) saving = '';
|
|
170
|
+
if (width(plain()) > available && source === 'fallback') {
|
|
171
|
+
// A successful Claude response can still follow evaluator failure. When
|
|
172
|
+
// space is tight, retain that cause instead of an ordinary "ready" phase,
|
|
173
|
+
// evaluator timing, or the guard details that followed the fallback.
|
|
174
|
+
compactFallback = true;
|
|
175
|
+
fallbackPhase = ['error', 'cancelled'].includes(phase) ? status : '';
|
|
176
|
+
status = [fallbackPhase, `${evaluatorLabel ? `${evaluatorLabel} ` : ''}fallback${fallbackCause ? `: ${fallbackCause}` : ''}`].filter(Boolean).join(' · ');
|
|
177
|
+
detail = '';
|
|
178
|
+
}
|
|
160
179
|
if (width(plain()) > available) detail = '';
|
|
161
180
|
if (width(plain()) > available) brand = '● AR';
|
|
162
181
|
if (width(plain()) > available) brand = '';
|
|
182
|
+
if (width(plain()) > available && compactFallback) {
|
|
183
|
+
status = [fallbackPhase, `fallback${fallbackCause ? `: ${fallbackCause}` : ''}`].filter(Boolean).join(' · ');
|
|
184
|
+
}
|
|
163
185
|
if (width(plain()) > available && model) {
|
|
164
|
-
if (phase === 'streaming') status = 'stream';
|
|
186
|
+
if (phase === 'streaming' && !compactFallback) status = 'stream';
|
|
165
187
|
const room = available - width(prefix + suffix + status) - 3;
|
|
166
188
|
if (room >= 1) model = shorten(model, room);
|
|
167
189
|
else {
|
|
@@ -172,6 +194,8 @@ export function renderStatusLine(input, snapshot, { now = Date.now(), color = tr
|
|
|
172
194
|
else if (width('● AR · ' + status) <= available) brand = '● AR';
|
|
173
195
|
}
|
|
174
196
|
}
|
|
197
|
+
if (width(plain()) > available && compactFallback && !fallbackPhase && available >= width('fallback')
|
|
198
|
+
&& available < width('fallback: ') + 2) status = 'fallback';
|
|
175
199
|
if (width(plain()) > available) status = shorten(status, Math.max(1, available - width(modelLabel()) - (model ? 3 : 0)));
|
|
176
200
|
const attention = phase === 'error' ? '31' : source === 'fallback' ? '33' : phase === 'cancelled' ? '2' : phase === 'streaming' || phase === 'ready' ? '32' : '36';
|
|
177
201
|
const chunks = [];
|