claude-autorouter 0.3.1 → 0.3.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/.env.example CHANGED
@@ -22,7 +22,7 @@ AUTOROUTER_JEV_TIMEOUT_MS=1500
22
22
  AUTOROUTER_TOKEN_COUNT_TIMEOUT_MS=1500
23
23
  AUTOROUTER_MIN_CONFIDENCE=0.75
24
24
 
25
- # Experimental local evaluator in AutoRouter 0.3.1+: native decision API only.
25
+ # Experimental local evaluator in AutoRouter 0.3.2+: native decision API only.
26
26
  # Install/start Ollama 0.35+, then run:
27
27
  # claude-autorouter setup --evaluator ollama --pull --force
28
28
  # Defaults to nimble:9b-q4_K_M (~5.63 GB download); --force replaces user config.
@@ -31,22 +31,41 @@ AUTOROUTER_MIN_CONFIDENCE=0.75
31
31
  # Add --ollama-model LOCAL_TAG_OR_ALIAS to setup to choose another suitable model.
32
32
  # Tev1 alternatives use the same native API; choose one explicitly:
33
33
  # claude-autorouter setup --evaluator ollama --ollama-model tev1:0.8b --pull --force
34
- # AUTOROUTER_OLLAMA_TIMEOUT_MS=10000 claude-autorouter setup --evaluator ollama --ollama-model tev1:4b-q4_K_M --pull --force
34
+ # claude-autorouter setup --evaluator ollama --ollama-model tev1:4b-q4_K_M --pull --force
35
35
  # Downloads: Tev1 0.8B Q8 ~812 MB; Tev1 4B Q4_K_M ~2.7 GB.
36
36
  # tev1:latest / tev1:4b select ~4.5 GB Q8; model terms: https://ollama.com/library/tev1
37
37
  # Or set AUTOROUTER_EVALUATOR=ollama above and configure an installed model:
38
38
  # AUTOROUTER_OLLAMA_URL=http://127.0.0.1:11434
39
39
  # AUTOROUTER_OLLAMA_MODEL=nimble:9b-q4_K_M
40
- # AUTOROUTER_OLLAMA_TIMEOUT_MS=1500
40
+ # Version 0.3.2 defaults: Tev1 0.8B/custom 1500ms; Tev1 4B 15000ms; Nimble 30000ms.
41
+ # An explicit timeout (including an old saved 1500) always overrides the default.
42
+ # Setup, doctor and startup show the effective model/deadline.
43
+ # Set only when intentionally overriding; Jev's separate deadline is unchanged:
44
+ # AUTOROUTER_OLLAMA_TIMEOUT_MS=30000
45
+ # Set 0 to disable only the runtime evaluator timer:
46
+ # AUTOROUTER_OLLAMA_TIMEOUT_MS=0
47
+ # One launch without changing saved config:
48
+ # AUTOROUTER_OLLAMA_TIMEOUT_MS=0 claude-autorouter claude
49
+ # Persist it for an installed model (the setup flag overrides the environment):
50
+ # claude-autorouter setup --evaluator ollama --ollama-model tev1:4b --ollama-timeout-ms 0 --force
51
+ # Positive values 1..30000 retain a deadline; 2500 means a 2.5-second cutoff.
52
+ # Zero and --ollama-timeout-ms require AutoRouter 0.3.2 or newer.
41
53
  # AUTOROUTER_OLLAMA_KEEP_ALIVE=5m
42
- # The launcher primes the classifier before opening Claude, allowing up to 60s.
54
+ # The launcher primes the classifier before opening Claude, allowing up to 60s even with timeout 0.
55
+ # User/disconnect cancellation remains active; normal evaluator errors still use fallback.
56
+ # Warmup and metadata-only doctor checks do not certify speed or accuracy.
43
57
  # Native model context is retained; evaluator state stays capped at 3,000 bytes.
44
- # After the idle period, cold reloading may exceed the deadline and use fallback.
58
+ # Local classification omits Claude executor system instructions; task/history remain.
59
+ # After the idle period, cold reloading may exceed an enabled deadline and use fallback.
45
60
  # A longer keep-alive holds the model in memory longer but avoids some reloads.
46
61
  # Native entropy confidence is not calibrated accuracy and does not use Jev's threshold.
62
+ # Historical measurements before 0.3.2:
47
63
  # On the tested M4, all 12 Nimble tuning requests exceeded the 1,500 ms deadline.
48
64
  # Tev1 0.8B: 18/24 labels, 450 ms median; 4B: 22/24, 3.15s at a 10s deadline.
49
65
  # Both Tev1 tags have a 2,050-token native context window; see measured limits.
66
+ # Source-only regression: node scripts/test-ollama-routing.mjs --model tev1:0.8b
67
+ # Uses local inference only; no Anthropic/Jev calls, downloads or config writes.
68
+ # A fallback timeout is not a valid Sonnet decision; mismatches fail the regression.
50
69
  # See docs/ollama-evaluation.md before allowing slower local classifications.
51
70
 
52
71
  AUTOROUTER_PORT=8787
package/README.md CHANGED
@@ -50,7 +50,23 @@ Savings are an **API-equivalent estimate for the same token counts**, using Opus
50
50
 
51
51
  ## Experimental local evaluator
52
52
 
53
- The local setup below requires AutoRouter 0.3.1 or newer. It uses Ollama's native `/v1/systemone` decision API with `nimble:9b-q4_K_M` by default. Jev remains the default evaluator. If upgrading from 0.2.0, replace the old Qwen model configuration using the [migration steps](docs/reference.md#migrating-an-older-ollama-config).
53
+ The local setup below requires AutoRouter 0.3.2 or newer. It uses Ollama's native `/v1/systemone` decision API with `nimble:9b-q4_K_M` by default. Jev remains the default evaluator. If upgrading from 0.2.0, replace the old Qwen model configuration using the [migration steps](docs/reference.md#migrating-an-older-ollama-config).
54
+
55
+ Version 0.3.2 excludes Claude's executor system instructions from the local classifier excerpt, retaining task and conversation excerpts. Runtime deadlines default to 1,500 ms for Tev1 0.8B/custom models, 15,000 ms for official Tev1 4B tags, and 30,000 ms for official Nimble tags. Explicit timeout settings, including a `1500` saved with 0.3.1, still override these defaults. Jev is unchanged.
56
+
57
+ Set `0` to disable AutoRouter's runtime evaluator deadline for one launch using your existing configuration:
58
+
59
+ ```sh
60
+ AUTOROUTER_OLLAMA_TIMEOUT_MS=0 claude-autorouter claude
61
+ ```
62
+
63
+ To save that setting for an installed Tev1 4B model:
64
+
65
+ ```sh
66
+ claude-autorouter setup --evaluator ollama --ollama-model tev1:4b --ollama-timeout-ms 0 --force
67
+ ```
68
+
69
+ The setup flag overrides the timeout environment value and saves it. Cancellation and disconnected clients still stop evaluation, normal errors still use fallback, and startup priming keeps its separate 60-second deadline.
54
70
 
55
71
  Install and start Ollama 0.35 or newer; [version 0.35.0](https://github.com/ollama/ollama/releases/tag/v0.35.0) is a prerelease as of September 29, 2026. Then run:
56
72
 
@@ -74,16 +90,18 @@ For example, select Tev1 0.8B with:
74
90
  claude-autorouter setup --evaluator ollama --ollama-model tev1:0.8b --pull --force
75
91
  ```
76
92
 
77
- Use `--ollama-model tev1:4b-q4_K_M` for the listed 4B variant; on the tested Mac it also needed a longer deadline, such as `AUTOROUTER_OLLAMA_TIMEOUT_MS=10000`, at setup. `tev1:latest` and `tev1:4b` select the larger Q8 download. Model terms are linked in the listings above; download size does not measure resident memory or routing quality. Custom native model tags and aliases also work.
93
+ Use `--ollama-model tev1:4b-q4_K_M` for the listed 4B variant; `tev1:latest` and `tev1:4b` select the larger Q8 download. Model terms are linked in the listings above; download size does not measure resident memory or routing quality. Custom native model tags and aliases also work.
78
94
 
79
95
  No Jev key is needed for local classification. The launcher primes the evaluator before opening Claude's UI, and evaluation failures fall back to Sonnet or retain Opus without contacting Jev. Claude still answers through Anthropic, with the same routing guards and subscription limits.
80
96
 
81
- On the tested 16 GiB M4, Tev1 0.8B matched 18/24 held-out labels with 450 ms median latency and no timeouts at the default 1,500 ms deadline, including full-excerpt checks. Tev1 4B matched 22/24 with a 10-second diagnostic deadline and 3.15-second median latency. Nimble matched 23/24 with a 30-second deadline and 11.4-second median latency. Both larger models exceeded the normal deadline in their standard-deadline tests. See the [measurements and limits](docs/ollama-evaluation.md) and [Ollama reference](docs/reference.md#ollama-evaluator), including configuration migration and the latency tradeoff.
97
+ Setup, doctor, and startup show the effective model and deadline. Warmup and doctor do not certify classification speed or accuracy. `Ollama fallback: timeout` means no valid decision arrived in time; it is different from the evaluator choosing Sonnet. Source users can run the [local routing regression](docs/development.md#local-routing-regression) to check all three tiers without Claude or Jev calls.
98
+
99
+ Historical measurements before 0.3.2, on a 16 GiB M4: Tev1 0.8B matched 18/24 held-out labels with 450 ms median latency and no timeouts at 1,500 ms, including full-excerpt checks. Tev1 4B matched 22/24 with a 10-second diagnostic deadline and 3.15-second median latency. Nimble matched 23/24 with a 30-second deadline and 11.4-second median latency. Both larger models exceeded the then-default 1,500 ms. A separate six-case regression with the 0.3.2 fixes passed for both tested Tev1 4B variants and Nimble; Tev1 0.8B matched only three cases. These small tests do not establish general accuracy or Jev parity. See the [measurements and limits](docs/ollama-evaluation.md) and [Ollama reference](docs/reference.md#ollama-evaluator).
82
100
 
83
101
  ## Behavior and data
84
102
 
85
103
  - The default client profile permits all three routing tiers. Tool continuations, thinking history, model-specific features, and context size can keep or upgrade a model even when the evaluator chooses a cheaper tier. [Routing policy](docs/reference.md#routing-policy).
86
- - The selected evaluator receives bounded excerpts that can contain source code, system instructions, and tool results: TypeSafe with Jev, or the local service with Ollama. Anthropic receives the complete request. Images, document payloads, and private thinking are omitted from classifier input. [Data flow and authentication](docs/reference.md#data-flow-and-authentication).
104
+ - The selected evaluator receives bounded excerpts that can contain source code and tool results: TypeSafe with Jev, or the local service with Ollama. Jev also receives system-text excerpts; the local path excludes Claude's executor system instructions. Anthropic receives the complete request. Images, document payloads, and private thinking are omitted from classifier input. [Data flow and authentication](docs/reference.md#data-flow-and-authentication).
87
105
  - Subscription access and usage limits still apply. Model switches can reduce cache reuse; cheaper token prices do not guarantee cheaper completed tasks. Run ordinary `claude` to bypass routing.
88
106
  - The launcher is quiet by default. Use `AUTOROUTER_DEBUG=1` for metadata diagnostics or `AUTOROUTER_STATUSLINE=0` to retain your existing status line. [Troubleshooting](docs/reference.md#troubleshooting).
89
107
 
@@ -9,7 +9,7 @@ import { dirname } from 'node:path';
9
9
  import { createStatusState } from '../src/status-state.mjs';
10
10
  import { addStatusLineSettings } from '../src/status-settings.mjs';
11
11
  import { loadUserConfig } from '../src/user-config.mjs';
12
- import { setup, doctor } from '../src/onboarding.mjs';
12
+ import { setup, doctor, ollamaDeadlineText } from '../src/onboarding.mjs';
13
13
  import { setupOllama } from '../src/ollama-setup.mjs';
14
14
 
15
15
  const [command = 'help', ...args] = process.argv.slice(2);
@@ -22,7 +22,7 @@ if (['--version', '-v', 'version'].includes(command)) {
22
22
  Usage:
23
23
  claude-autorouter setup [--auth-mode subscription|api-key] [--force]
24
24
  [--evaluator jev|ollama]
25
- [--ollama-model MODEL] [--pull]
25
+ [--ollama-model MODEL] [--ollama-timeout-ms N] [--pull]
26
26
  claude-autorouter doctor
27
27
  claude-autorouter claude [Claude Code arguments]
28
28
  claude-autorouter serve
@@ -39,6 +39,9 @@ Ollama evaluates locally and requires Ollama 0.35+ with /v1/systemone.
39
39
  Use setup --evaluator ollama --pull to detect Ollama and download a missing model.
40
40
  The local default is nimble:9b-q4_K_M; --ollama-model selects another compatible model.
41
41
  Smaller Tev1 options: --ollama-model tev1:0.8b or --ollama-model tev1:4b-q4_K_M.
42
+ Local routing deadlines: Tev1 0.8B/custom 1500 ms, Tev1 4B 15000 ms, Nimble 30000 ms.
43
+ Setup --ollama-timeout-ms N saves a routing deadline; use 0 to disable it.
44
+ AUTOROUTER_OLLAMA_TIMEOUT_MS also overrides the deadline; 0 disables it.
42
45
  Local routing is experimental; see docs/ollama-evaluation.md for measured limits.
43
46
  AUTOROUTER_AUTH_MODE=subscription uses your saved Claude Code login.
44
47
  Without setup, AUTOROUTER_AUTH_MODE defaults to api-key and also requires ANTHROPIC_API_KEY.
@@ -84,7 +87,7 @@ Complete inference requests still go to Anthropic. See README.md.`);
84
87
  config.localToken = randomBytes(32).toString('hex');
85
88
  }
86
89
  if (config.evaluator === 'ollama') {
87
- console.error(`Preparing local Ollama evaluator (${config.ollamaModel})…`);
90
+ console.error(`Preparing local Ollama evaluator (${config.ollamaModel}); ${ollamaDeadlineText(config.ollamaTimeoutMs)}…`);
88
91
  try { await setupOllama(config, { pull: false, warm: true, write: () => {} }); }
89
92
  catch {
90
93
  console.error('Ollama could not be prepared. Requests will use the conservative fallback while it is unavailable; run claude-autorouter doctor.');
@@ -26,9 +26,37 @@ node --env-file=.env bin/autorouter.mjs claude
26
26
 
27
27
  The explicit Node flag loads `.env`; the CLI itself does not auto-load project files. Environment values override the user config. Keep keys out of source control and command arguments.
28
28
 
29
+ With AutoRouter 0.3.2 or newer, launch an already configured Ollama evaluator without its runtime deadline using:
30
+
31
+ ```sh
32
+ AUTOROUTER_OLLAMA_TIMEOUT_MS=0 claude-autorouter claude
33
+ ```
34
+
35
+ Persist the setting with `claude-autorouter setup --evaluator ollama --ollama-model tev1:4b --ollama-timeout-ms 0 --force` for that installed model. The setup flag overrides the timeout environment value; later launch-time environment values still override saved configuration. Startup priming keeps its separate 60-second limit, cancellation remains active, and normal errors still use fallback.
36
+
37
+ ## Local routing regression
38
+
39
+ The source-only harness below is opt-in and is not included in the npm package. It sends synthetic Claude-shaped requests through the real router and an installed local evaluator, checking task extraction, classifier choices, selected Claude tiers, and new human turns. It makes no Anthropic or Jev calls, downloads no models, and writes no user configuration.
40
+
41
+ ```sh
42
+ node scripts/test-ollama-routing.mjs --model tev1:0.8b
43
+ node scripts/test-ollama-routing.mjs --model tev1:4b
44
+ node scripts/test-ollama-routing.mjs --model nimble:9b-q4_K_M
45
+ ```
46
+
47
+ Start Ollama 0.35+ and install the selected model first. The harness refuses to run while another model is resident. It never explicitly unloads the selected model; its keep-alive setting controls residency. It warms once, then uses the production deadline for each uncached case: 1,500 ms for Tev1 0.8B/custom tags, 15,000 ms for official Tev1 4B tags, and 30,000 ms for official Nimble tags. Environment settings or `--timeout-ms N` can override the deadline; `--timeout-ms 0` disables the runtime timer while retaining cancellation and the separate warmup limit. This harness does not load saved user configuration. Use `--output artifacts/local-routing.json` to save a metadata report.
48
+
49
+ ```sh
50
+ node scripts/test-ollama-routing.mjs --model tev1:4b --timeout-ms 0
51
+ ```
52
+
53
+ The command fails on a wrong classification, fallback, unexpected guard override, or missing tier coverage. A Sonnet result with `source: ollama` and `classified_tier: sonnet` is a valid prediction; `source: fallback` and `classifier_error: timeout` means classification did not complete. Passing establishes these synthetic cases only. Warmup and metadata-only `doctor` checks do not establish speed or accuracy on real tasks.
54
+
55
+ Version 0.3.2 excludes Claude's executor system instructions from local classifier input before excerpt budgeting and retains task/history excerpts. Jev is unchanged. Version 0.3.1 included the executor background locally and defaulted every local model to 1,500 ms; see the [upgrade notes](reference.md#migrating-an-older-ollama-config). Historical benchmark results must remain labeled with their original excerpt policy and explicit deadlines.
56
+
29
57
  ## Live integration tests
30
58
 
31
- These tests make real Claude calls and invoke the configured evaluator, consuming Claude usage and, with Jev, TypeSafe usage. They use temporary synthetic fixtures and disable unrelated customizations and MCP servers. With Ollama, start the local service and install the chosen model first; the harness does not install or download it.
59
+ The Claude integration tests below make real Claude calls and invoke the configured evaluator, consuming Claude usage and, with Jev, TypeSafe usage. They use temporary synthetic fixtures and disable unrelated customizations and MCP servers. With Ollama, start the local service and install the chosen model first; the harness does not install or download it.
32
60
 
33
61
  ```sh
34
62
  npm run test:live
@@ -2,7 +2,32 @@
2
2
 
3
3
  AutoRouter supports `/v1/systemone` classification only: remote Jev with an API key, or local Ollama 0.35+ with a compatible decision model. Jev remains the default evaluator. The local default is `nimble:9b-q4_K_M`; the old Qwen chat adapter and compact/quality/auto presets have been removed. Existing downloaded models are not deleted.
4
4
 
5
- ## Method
5
+ ## Version 0.3.2 routing regression (September 30, 2026)
6
+
7
+ The npm `0.3.1` runtime used a 1,500 ms deadline for every local model, even though startup priming allows 60 seconds. On the user's already-loaded `tev1:4b` (Q8), four synthetic tasks all timed out and selected Sonnet through `source=fallback`, `classifier_error=timeout`. With a separate 30-second diagnostic allowance, the same tasks selected Haiku, Haiku, Sonnet, and Opus. Three of those decisions took 5.6–6.1 seconds. This was a deadline failure, not a parser forcing every answer to Sonnet.
8
+
9
+ We captured three synthetic request shapes from the installed Claude client using an isolated local response stub, without paid provider calls. The human task survived intact and no compatibility guard forced Sonnet. The local evaluator nevertheless received 627–796 characters of Claude's general executor instructions. Tev1 0.8B classified all three captures as Sonnet. Removing only this system background changed the distributed-fencing case to Opus; the `[].length` example still selected Sonnet. Removing additional state fields did not improve that result, so the routing rubric and remaining state layout were retained.
10
+
11
+ Version 0.3.2 excludes executor system instructions from local classifier input, uses 15-second defaults for official Tev1 4B variants and 30 seconds for Nimble, and retains the 1.5-second default for Tev1 0.8B/custom models. Explicit timeout settings still win; `0` disables only the runtime evaluator timer. Setup accepts `--ollama-timeout-ms`; setup, doctor, and startup display the effective deadline. Compact status lines retain the fallback cause. Jev's input extraction and deadline are unchanged.
12
+
13
+ The source-only, opt-in `npm run test:ollama -- --model TAG` checks actual `Router.route` results against six fixed synthetic Claude-shaped requests: literal output, `[].length`, a bounded feature, distributed fencing, a new mechanical task after a difficult task, and a new difficult task after a simple task. It checks the evaluator choice, selected Claude model, source, and reason, and fails on a mismatch, fallback, override, or missing tier coverage. Live cases use fresh routers to avoid decision-cache passes; persistent turn-cache transitions and compatibility guards are separately covered by offline regression tests. It is not included in the npm package, downloads nothing, and does not contact Claude or Jev.
14
+
15
+ One pass with the fixes on the same M4/16 GiB Mac produced:
16
+
17
+ | Installed model | Runtime deadline | Matching cases | Fallbacks | Decision latency range |
18
+ | --- | ---: | ---: | ---: | ---: |
19
+ | `tev1:0.8b` | 1,500 ms | 3 / 6 | 0 | 48–479 ms |
20
+ | `tev1:4b-q4_K_M` | 15,000 ms | 6 / 6 | 0 | 2.40–2.75 s |
21
+ | `tev1:4b` (Q8) | 15,000 ms | 6 / 6 | 0 | 2.38–2.70 s |
22
+ | `nimble:9b-q4_K_M` | 30,000 ms | 6 / 6 | 0 | 4.74–7.29 s |
23
+
24
+ Every model reached all three tiers. The 0.8B failures were genuine classifications: `[].length` → Sonnet, new mechanical task → Opus, new difficult task → Sonnet. Its live suite therefore **fails**, rather than treating valid native responses as proof of accuracy. These six cases are regression checks, not a new held-out quality benchmark; the samples, machine load, residency, and prompt-cache effects do not establish a quantization speed comparison or a worst-case latency bound. The larger-model budgets allow slower decisions; they do not make those models fast or prevent every timeout. No user configuration or model files were changed; the originally resident `tev1:4b` was restored after sequential model testing.
25
+
26
+ A separate check with `tev1:4b` and the runtime deadline disabled (`timeout_ms: 0`) passed the same six cases without fallback, with decision latencies of 4.37–5.42 seconds after 13.11 seconds of priming. It made no Claude or Jev calls and downloaded nothing. This validates disabled-timer routing on those cases; it adds no new quality cases or latency guarantee.
27
+
28
+ Fixture SHA-256: `88e12f962fb42b36a00edd6b8c2560e54620d90c978b1ff13a1ddf31ee46b585`. The questions remain unchanged at `be151cedb4de4b7ef3f7162d751f70ce7d9dd14efc66fae1835f73ffd04027be`.
29
+
30
+ ## Historical benchmark method (before 0.3.2)
6
31
 
7
32
  Measurements were recorded on September 29, 2026, on an Apple M4 Mac with 16 GiB of unified memory alongside other applications. Nimble ran on an isolated Ollama 0.35.0 process; the installed Ollama 0.33.3 daemon was left unchanged during that test. Tev1 was tested later on the user's upgraded Ollama 0.35.0 service after its downloads finished. Version 0.35.0 was a prerelease at the time. These are observations under different application loads, not a controlled hardware comparison. See the [official release](https://github.com/ollama/ollama/releases/tag/v0.35.0), [Nimble catalog](https://ollama.com/library/nimble), and [Tev1 catalog](https://ollama.com/library/tev1).
8
33
 
@@ -10,15 +35,15 @@ The selected Nimble tag contains a 9B Q4_K_M model, approximately 5.63 GB to dow
10
35
 
11
36
  The fixture contains 36 balanced synthetic workloads: 12 tuning cases and 24 held-out cases, with equal numbers of Haiku, Sonnet, and Opus labels. Cases include mechanical edits, ordinary implementation, difficult correctness and security work, topic changes, tool results, short follow-ups, and misleading routing instructions. These labels are judgments under the routing policy, not proof of which Claude model would complete each task successfully. The native questions preserve the existing local routing policy and were frozen before Nimble testing; no changes were made from held-out results.
12
37
 
13
- The harness uses the production state builder and evaluator, including local metadata checks in wall-clock latency. It bypasses AutoRouter's decision cache; Ollama's own caching remains enabled. A cold measurement starts with the model unloaded, but operating-system file caches and kernels may already be warm. Cold calls have a separate 60-second deadline. The normal evaluator deadline is 1,500 ms. Reported model allocation comes from `/api/ps`; it is not a measurement of total process or system memory, and its GPU allocation is not additional independent RAM on this unified-memory Mac. Aggregate runtime RSS includes all Ollama and llama-server processes, including the original idle daemon.
38
+ The historical harness used the then-current production state builder and evaluator, including local metadata checks in wall-clock latency. It bypasses AutoRouter's decision cache; Ollama's own caching remains enabled. A cold measurement starts with the model unloaded, but operating-system file caches and kernels may already be warm. Cold calls have a separate 60-second deadline. The normal evaluator deadline in that benchmark was 1,500 ms for every model. Reported model allocation comes from `/api/ps`; it is not a measurement of total process or system memory, and its GPU allocation is not additional independent RAM on this unified-memory Mac. Aggregate runtime RSS includes all Ollama and llama-server processes, including the original idle daemon.
14
39
 
15
- ## Nimble results
40
+ ## Historical Nimble results
16
41
 
17
42
  The first production-deadline run returned Haiku correctly for its cold mechanical task in 17.64 seconds. **All 12 warm tuning requests timed out at 1,500 ms**, with cancellation p50/p95 of 1,503/1,513 ms. No warm classification accuracy can be inferred from that run. The model's reported allocation was 5.48 GB, with an 8,194-token context. This did not meet the desired fast-routing target on the tested Mac.
18
43
 
19
44
  A separate first-load smoke call correctly classified `[].length` as Haiku in 25.23 seconds. It checks integration, not warm performance or classifier accuracy.
20
45
 
21
- The separate held-out diagnostic used a 30,000 ms deadline and one pass over 24 distinct workloads. It does not change the production default or establish performance at 1,500 ms.
46
+ The separate held-out diagnostic used a 30,000 ms deadline and one pass over 24 distinct workloads. It did not change the then-default 1,500 ms budget or establish performance within it.
22
47
 
23
48
  | Measurement | Result |
24
49
  | --- | ---: |
@@ -36,9 +61,9 @@ All eight full-excerpt diagnostic requests completed at the 30,000 ms deadline a
36
61
 
37
62
  An isolated live Claude Code test also passed using the saved Enterprise subscription login, a temporary configuration with a 30,000 ms evaluator deadline, and no Jev key. Nimble selected Haiku in 12.69 seconds; Anthropic returned HTTP 200 with the expected literal response and confirmed `claude-haiku-4-5-20251001`. Only a synthetic prompt was used, with no repository files or tools. This verifies the authentication and routing integration, not general classifier accuracy. The installed Ollama service and user configuration were left unchanged.
38
63
 
39
- ## Tev1 results
64
+ ## Historical Tev1 results
40
65
 
41
- Both [Tev1 variants](https://ollama.com/library/tev1) use the same production adapter and frozen questions, selected through `--ollama-model`. The 0.8B tag uses Q8_0 quantization and downloads approximately 812 MB; `tev1:4b-q4_K_M` downloads approximately 2.71 GB. The unqualified `tev1` tag selects the larger 4B Q8 model, which was not tested. Jev remains the evaluator default and Nimble remains the local-model default.
66
+ Both [Tev1 variants](https://ollama.com/library/tev1) use the same production adapter and frozen questions, selected through `--ollama-model`. The 0.8B tag uses Q8_0 quantization and downloads approximately 812 MB; `tev1:4b-q4_K_M` downloads approximately 2.71 GB. The unqualified `tev1` tag selects the larger 4B Q8 model, which was not tested in this historical benchmark. It was tested separately in the 0.3.2 regression above. Jev remains the evaluator default and Nimble remains the local-model default.
42
67
 
43
68
  Each measured tag ships `num_ctx:2050`. The window includes the template, routing criteria, and excerpt. The 3,000-byte state cap is not a guarantee that every possible input fits this smaller token window. The reported stress cases used 1,664–1,752 input tokens on 0.8B. Requests exceeding model limits use the usual fallback; AutoRouter does not switch protocols or silently truncate additional content for Tev1.
44
69
 
@@ -52,7 +77,7 @@ These measurements use one pass over the same 24 held-out cases and eight separa
52
77
 
53
78
  The 0.8B model matched four of eight Haiku labels, all eight Sonnet labels, and six of eight Opus labels. It over-routed four mechanical tasks and under-routed two difficult tasks to Sonnet, including a case with a misleading tier instruction. Cold wall time was 2.08 seconds for 0.8B and 6.04 seconds for 4B. Model size and fast responses do not establish sufficient accuracy for an engineering workload.
54
79
 
55
- With a separate 10-second deadline, 4B matched seven of eight Haiku labels, all eight Sonnet labels, and seven of eight Opus labels. It over-routed one mechanical case to Opus and under-routed one difficult case to Sonnet. Its cold diagnostic request took 3.69 seconds. The improved agreement comes with several seconds of classification latency; it is not performance at the default deadline.
80
+ With a separate 10-second deadline, 4B matched seven of eight Haiku labels, all eight Sonnet labels, and seven of eight Opus labels. It over-routed one mechanical case to Opus and under-routed one difficult case to Sonnet. Its cold diagnostic request took 3.69 seconds. The improved agreement comes with several seconds of classification latency; it is not performance at the then-default 1,500 ms deadline.
56
81
 
57
82
  All eight 0.8B full-excerpt requests completed within 1,500 ms and returned Haiku, with p50/p95 of 1,079/1,150 ms. The 4B model timed out on all eight at 1,500 ms and again on all eight at 10,000 ms. Its 10-second cancellation p50/p95 was 10,006/10,081 ms; completed full-excerpt latency was not measured. The longer deadline therefore allows the reported short held-out decisions but does not guarantee completion for full excerpts.
58
83
 
@@ -67,15 +92,15 @@ Model digests:
67
92
 
68
93
  ## Reproducing the evaluation
69
94
 
70
- Use a source checkout; benchmark scripts and fixtures are not included in the npm package. Install Ollama 0.35+, start it, and explicitly download the model:
95
+ Use a source checkout; benchmark scripts and fixtures are not included in the npm package. The commands below evaluate the checked-out version. To reproduce the historical excerpt policy, use the `v0.3.1` checkout; 0.3.2 changes local input extraction. The explicit deadlines retain the historical budgets. Install Ollama 0.35+, start it, and explicitly download the model:
71
96
 
72
97
  ```sh
73
98
  ollama pull nimble:9b-q4_K_M
74
- node scripts/evaluate-ollama.mjs --models nimble:9b-q4_K_M --split tuning --rounds 1 --output artifacts/nimble-tuning.json
99
+ node scripts/evaluate-ollama.mjs --models nimble:9b-q4_K_M --split tuning --rounds 1 --timeout-ms 1500 --output artifacts/nimble-tuning.json
75
100
  node scripts/evaluate-ollama.mjs --models nimble:9b-q4_K_M --split heldout --rounds 1 --stress-rounds 8 --timeout-ms 30000 --output artifacts/nimble-diagnostic.json
76
101
  ollama pull tev1:0.8b
77
102
  ollama pull tev1:4b-q4_K_M
78
- node scripts/evaluate-ollama.mjs --models tev1:0.8b,tev1:4b-q4_K_M --split heldout --rounds 1 --stress-rounds 8 --output artifacts/tev1-default.json
103
+ node scripts/evaluate-ollama.mjs --models tev1:0.8b,tev1:4b-q4_K_M --split heldout --rounds 1 --stress-rounds 8 --timeout-ms 1500 --output artifacts/tev1-default.json
79
104
  node scripts/evaluate-ollama.mjs --models tev1:4b-q4_K_M --split heldout --rounds 1 --stress-rounds 8 --timeout-ms 10000 --output artifacts/tev1-diagnostic.json
80
105
  ```
81
106
 
@@ -91,6 +116,6 @@ Reproducibility identifiers:
91
116
 
92
117
  This is a small synthetic rubric-agreement benchmark, not a downstream task-quality, savings, Jev-parity, or security evaluation. Classification can miss context outside the excerpt. The cases do not establish robust resistance to prompt injection. Performance depends on hardware, memory pressure, prompt length, and residency; a successful setup or simple request does not guarantee the runtime deadline.
93
118
 
94
- The launcher primes the model before opening Claude, allowing up to 60 seconds for that synthetic classification. Idle unloading can still make later requests cold. Runtime timeouts and invalid responses use the existing conservative fallback, without contacting Jev. A longer `AUTOROUTER_OLLAMA_TIMEOUT_MS` trades added prompt latency for more completed local classifications; it does not make the evaluator faster.
119
+ The launcher primes the model before opening Claude, allowing up to 60 seconds for that synthetic classification. Idle unloading can still make later requests cold. Runtime timeouts and invalid responses use the existing conservative fallback, without contacting Jev. A longer `AUTOROUTER_OLLAMA_TIMEOUT_MS` trades added prompt latency for more completed local classifications; it does not make the evaluator faster. In 0.3.2, `0` disables the runtime timer while preserving user/disconnect cancellation, normal error fallback, and the separate startup limit.
95
120
 
96
121
  Earlier Qwen results used a different `/api/chat` implementation and are not measurements of this native backend. They remain available in the [historical evaluation document](https://github.com/frapposelli/claude-autorouter/blob/548a175/docs/ollama-evaluation.md). Reproduce those results from that revision, not the current native-only harness.
package/docs/reference.md CHANGED
@@ -7,6 +7,7 @@
7
7
  | `claude-autorouter setup` | Save subscription-mode configuration and a Jev key |
8
8
  | `claude-autorouter setup --auth-mode api-key` | Configure Jev and Anthropic API-key billing |
9
9
  | `claude-autorouter setup --evaluator ollama --pull` | Configure the native local evaluator and download its selected model if missing |
10
+ | `claude-autorouter setup --evaluator ollama --ollama-timeout-ms 0 --force` | Save a disabled runtime evaluator deadline |
10
11
  | `claude-autorouter setup --force` | Replace an existing user config |
11
12
  | `claude-autorouter doctor` | Check config, Claude executable/login, and the selected local Ollama model without paid calls |
12
13
  | `claude-autorouter claude [arguments]` | Start a local router and pass arguments through to Claude Code |
@@ -52,7 +53,7 @@ For an environment-only subscription launch, set `AUTOROUTER_AUTH_MODE=subscript
52
53
  | `AUTOROUTER_JEV_TIMEOUT_MS` | `1500` | Classifier deadline in milliseconds |
53
54
  | `AUTOROUTER_OLLAMA_URL` | `http://127.0.0.1:11434` | Loopback Ollama base URL |
54
55
  | `AUTOROUTER_OLLAMA_MODEL` | `nimble:9b-q4_K_M` | Installed local model tag or alias compatible with `/v1/systemone` |
55
- | `AUTOROUTER_OLLAMA_TIMEOUT_MS` | `1500` | Whole local classification deadline in milliseconds |
56
+ | `AUTOROUTER_OLLAMA_TIMEOUT_MS` | model-dependent; see below | Runtime local classification deadline, `1`–`30000` ms; `0` disables it |
56
57
  | `AUTOROUTER_OLLAMA_KEEP_ALIVE` | `5m` | How long Ollama retains the evaluator in memory |
57
58
  | `AUTOROUTER_TOKEN_COUNT_TIMEOUT_MS` | `1500` | Context-check deadline; runs alongside classification |
58
59
  | `AUTOROUTER_MIN_CONFIDENCE` | `0.75` | Jev confidence threshold; does not apply to Ollama |
@@ -65,7 +66,9 @@ Model access depends on your account. The policy recognizes specific Claude mode
65
66
 
66
67
  ## Ollama evaluator
67
68
 
68
- Local classification is experimental and requires AutoRouter 0.3.1 or newer. It uses Ollama's native `/v1/systemone` decision endpoint for every model, replacing the chat backend from 0.2.0. Jev remains the default remote evaluator, using TypeSafe's `/v1/systemone` endpoint and a TypeSafe API key. Selecting Ollama never silently switches back to Jev. Haiku, Sonnet, or Opus still completes the task through Anthropic.
69
+ The local configuration documented here requires AutoRouter 0.3.2 or newer and remains experimental. It uses Ollama's native `/v1/systemone` decision endpoint for every model, replacing the chat backend from 0.2.0. Jev remains the default remote evaluator, using TypeSafe's `/v1/systemone` endpoint and a TypeSafe API key. Selecting Ollama never silently switches back to Jev. Haiku, Sonnet, or Opus still completes the task through Anthropic.
70
+
71
+ Version 0.3.2 excludes Claude's executor system instructions from local excerpts, uses model-specific runtime deadlines, and accepts `0` to disable that deadline. Setup, doctor, and startup show the effective model and deadline; setup accepts `--ollama-timeout-ms`. Jev is unchanged.
69
72
 
70
73
  All local models require Ollama 0.35 or newer. Version 0.35.0 is a prerelease as of September 29, 2026; it introduces the native decision API. See the [Ollama release notes](https://github.com/ollama/ollama/releases/tag/v0.35.0). Install and start a compatible local service, then run:
71
74
 
@@ -95,31 +98,45 @@ claude-autorouter setup --evaluator ollama --ollama-model tev1:0.8b --pull --for
95
98
  ```
96
99
 
97
100
  ```sh
98
- # Tev1 4B Q4_K_M, allowing slower local decisions
99
- AUTOROUTER_OLLAMA_TIMEOUT_MS=10000 claude-autorouter setup --evaluator ollama --ollama-model tev1:4b-q4_K_M --pull --force
101
+ # Tev1 4B Q4_K_M, with a 15-second default deadline
102
+ claude-autorouter setup --evaluator ollama --ollama-model tev1:4b-q4_K_M --pull --force
100
103
  ```
101
104
 
102
- For Nimble, the explicit Q4_K_M tag avoids `nimble:latest`, which currently selects an approximately 9.5 GB Q8 model. For Tev1, `tev1:latest` and `tev1:4b` select approximately 4.5 GB Q8 weights; the explicit `tev1:4b-q4_K_M` tag selects the smaller 4B download. Download size is not resident memory: runtime and context allocations add to it, and other applications need memory too. Downloaded models have their own licenses and are not bundled in this package. On the tested 16 GiB M4, Tev1 0.8B matched 18/24 held-out labels at 450 ms median latency within the normal deadline; 4B matched 22/24 at 3.15 seconds with a separate 10-second deadline. See the [local measurements](ollama-evaluation.md) before choosing a latency deadline.
105
+ For Nimble, the explicit Q4_K_M tag avoids `nimble:latest`, which currently selects an approximately 9.5 GB Q8 model. For Tev1, `tev1:latest` and `tev1:4b` select approximately 4.5 GB Q8 weights; the explicit `tev1:4b-q4_K_M` tag selects the smaller 4B download. Download size is not resident memory: runtime and context allocations add to it, and other applications need memory too. Downloaded models have their own licenses and are not bundled in this package. In historical tests before 0.3.2 on a 16 GiB M4, Tev1 0.8B matched 18/24 held-out labels at 450 ms median latency within 1,500 ms; 4B matched 22/24 at 3.15 seconds with a separate 10-second deadline. See the [local measurements](ollama-evaluation.md) before choosing a latency deadline.
103
106
 
104
107
  The endpoint must be loopback (`127.0.0.1`, `localhost`, or `::1`), without a path, credentials, query, or fragment. Cloud model tags and metadata identifying a remote model are rejected before sending task text. Claude and Jev credentials are never attached to Ollama requests.
105
108
 
106
109
  ### Classification and fallback
107
110
 
108
- Local classification caps serialized evaluator state at both 3,000 characters and 3,000 UTF-8 bytes, including for non-ASCII prompts. `/v1/systemone` receives the bounded state and routing criteria and returns a tier directly. The router retains each model's native context setting: 8,194 tokens for the default Nimble tag and 2,050 for the listed Tev1 tags. Tev1's smaller window includes the routing criteria and template as well as the excerpt; the byte limit does not guarantee every possible input fits. Context errors use the normal fallback. Returned confidence scores summarize choice-distribution entropy; they are not calibrated accuracy probabilities. `AUTOROUTER_MIN_CONFIDENCE` applies only to Jev. All capability, tool-continuation, thinking, and context guards still apply.
111
+ Local classification caps serialized evaluator state at both 3,000 characters and 3,000 UTF-8 bytes, including for non-ASCII prompts. Claude's top-level executor system instructions are excluded before budgeting; the current task, original task, and recent conversation excerpts remain. `/v1/systemone` receives the bounded state and routing criteria and returns a tier directly. The router retains each model's native context setting: 8,194 tokens for the default Nimble tag and 2,050 for the listed Tev1 tags. Tev1's smaller window includes the routing criteria and template as well as the excerpt; the byte limit does not guarantee every possible input fits. Context errors use the normal fallback. Returned confidence scores summarize choice-distribution entropy; they are not calibrated accuracy probabilities. `AUTOROUTER_MIN_CONFIDENCE` applies only to Jev. All capability, tool-continuation, thinking, and context guards still apply.
109
112
 
110
- Before opening Claude's UI, the launcher loads an installed model and primes the actual classifier rubric with a synthetic task, using a separate deadline of up to 60 seconds. Each normal evaluation has a 1,500 ms deadline covering local checks and classification. Priming reduces first-request overhead but does not guarantee that longer excerpts finish in time. `AUTOROUTER_OLLAMA_KEEP_ALIVE` defaults to `5m`. After five idle minutes, the next request may need to reload the model, exceed that deadline, and use the fallback. A longer positive keep-alive can reduce reloads while retaining memory longer; `0` unloads immediately and can make every evaluation cold. Supported values are `0` or a positive duration such as `30s`, `5m`, or `1h`.
113
+ The deadline covering local checks and classification defaults to 1,500 ms for Tev1 0.8B and custom/unrecognized tags, 15,000 ms for official Tev1 4B variants (including bare `tev1` and `latest`), and 30,000 ms for official Nimble variants. Official `library/` and `registry.ollama.ai/` aliases are recognized; a custom namespace such as `team/nimble` keeps the short default. An explicit timeout overrides the model default, including an old saved `1500`. Environment values override saved values on launch. Defaults are not written into the user config; `setup --force` replaces the config and saves an explicit timeout when supplied through `--ollama-timeout-ms` or the environment. During setup, the command-line flag takes precedence over the timeout environment value.
111
114
 
112
- If startup priming fails, the launcher warns and continues. An incompatible model or Ollama version, missing model, unavailable service, malformed answer, or evaluation timeout falls back to Sonnet or retains an incoming Opus, subject to the usual compatibility policy. No Jev request is made. The status line identifies `Ollama fallback` and its error category. Run `doctor` to inspect the local service and installed model; it does not download or generate. A live classification is needed to verify the selected model's decision-API behavior.
115
+ Set `AUTOROUTER_OLLAMA_TIMEOUT_MS=0` to remove AutoRouter's runtime evaluator timer while keeping the existing configuration:
113
116
 
114
- Tev1 4B timed out on all eight full-excerpt checks even with the 10-second allowance shown above; its short-task results do not establish a full-excerpt latency bound. Tev1 0.8B completed all eight within the default deadline. On the tested 16 GiB M4, Nimble timed out on all 12 tuning requests at the default deadline. A separate 30-second diagnostic completed 24 held-out classifications with 23 matching labels, but median routing took 11.4 seconds. The one error followed a misleading tier instruction. See the [measurements and limitations](ollama-evaluation.md). If you accept several seconds of added latency, configure a longer deadline explicitly; this example is a diagnostic allowance, not a speed recommendation:
117
+ ```sh
118
+ AUTOROUTER_OLLAMA_TIMEOUT_MS=0 claude-autorouter claude
119
+ ```
120
+
121
+ To persist it for an already installed Tev1 4B model, run:
115
122
 
116
123
  ```sh
117
- AUTOROUTER_OLLAMA_TIMEOUT_MS=30000 claude-autorouter setup --evaluator ollama --pull --force
124
+ claude-autorouter setup --evaluator ollama --ollama-model tev1:4b --ollama-timeout-ms 0 --force
118
125
  ```
119
126
 
127
+ Only the runtime evaluator deadline is disabled. User cancellation and client disconnection still abort evaluation; ordinary service, HTTP, and response errors still use fallback. Startup priming retains its separate 60-second limit, and lifecycle checks retain their own limits. Positive values from `1` to `30000` keep a finite deadline: for example, `--ollama-timeout-ms 2500` permits 2.5 seconds and can still time out on decisions near that cutoff.
128
+
129
+ Before opening Claude's UI, the launcher loads an installed model and primes the classifier rubric with a synthetic task, using a separate deadline of up to 60 seconds. Successful priming does not establish that real excerpts finish within an enabled runtime deadline or classify correctly. `doctor` checks version and model availability without inference; it does not certify speed or accuracy either. `AUTOROUTER_OLLAMA_KEEP_ALIVE` defaults to `5m`. After five idle minutes, the next request may need to reload the model and, when a runtime deadline is enabled, exceed it and use the fallback. A longer positive keep-alive reduces some reloads while retaining memory longer; keep-alive `0` unloads immediately and can make every evaluation cold. Supported keep-alive values are `0` or a positive duration such as `30s`, `5m`, or `1h`.
130
+
131
+ If startup priming fails, the launcher warns and continues. An incompatible model or Ollama version, missing model, unavailable service, malformed answer, or evaluation timeout falls back to Sonnet or retains an incoming Opus, subject to the usual compatibility policy. No Jev request is made. `Ollama fallback` with `timeout` means no valid classification completed in time; it is not a Sonnet prediction. In metadata, a valid Sonnet decision has `source: "ollama"` and `classified_tier: "sonnet"`; a timeout has `source: "fallback"` and `classifier_error: "timeout"`. Later policy guards can still change the selected Claude model. Use the source-only [local routing regression](development.md#local-routing-regression) to test classification and all three selected tiers without external provider calls.
132
+
133
+ Historical measurements before 0.3.2: Tev1 4B timed out on all eight full-excerpt checks even with a 10-second diagnostic allowance; its short-task results did not establish a full-excerpt latency bound. Tev1 0.8B completed all eight within 1,500 ms. On the tested 16 GiB M4, Nimble timed out on all 12 tuning requests at 1,500 ms. A separate 30-second diagnostic completed 24 held-out classifications with 23 matching labels, but median routing took 11.4 seconds. The one error followed a misleading tier instruction. These historical results precede the 0.3.2 excerpt changes and do not establish guarantees for the longer defaults. See the [measurements and limitations](ollama-evaluation.md).
134
+
120
135
  ### Migrating an older Ollama config
121
136
 
122
- Version 0.3.1 removes the Qwen chat backend and presets from 0.2.0. Existing downloaded models remain on disk, but an old Qwen model selection needs to be replaced with a native decision model. Run the setup command above with `--force`; it selects Nimble unless you pass `--ollama-model` or override the model through the environment. Remove or update any old `AUTOROUTER_OLLAMA_MODEL` environment value too, because environment variables override saved configuration. Update scripts to use `--ollama-model` when selecting a custom model.
137
+ Version 0.3.1 used a 1,500 ms deadline for every local model. After upgrading to 0.3.2, an explicitly saved or exported `AUTOROUTER_OLLAMA_TIMEOUT_MS=1500` still wins over the new model-specific defaults. Remove that override to use the defaults, or rerun setup with the desired model and `--ollama-timeout-ms N --force`. The `0` value and setup timeout flag require 0.3.2 or newer.
138
+
139
+ Version 0.3.1 removed the Qwen chat backend and presets from 0.2.0. Existing downloaded models remain on disk, but an old Qwen model selection needs to be replaced with a native decision model. Run the setup command above with `--force`; it selects Nimble unless you pass `--ollama-model` or override the model through the environment. Remove or update any old `AUTOROUTER_OLLAMA_MODEL` environment value too, because environment variables override saved configuration. Update scripts to use `--ollama-model` when selecting a custom model.
123
140
 
124
141
  ## Data flow and authentication
125
142
 
@@ -131,7 +148,7 @@ Claude Code → authenticated local gateway → Jev or local Ollama classificati
131
148
 
132
149
  AutoRouter uses Claude Code's [gateway integration](https://code.claude.com/docs/en/llm-gateway-protocol), so it sees inference requests and tool continuations. It does not rely on a user-prompt hook.
133
150
 
134
- The selected evaluator receives a bounded state containing the latest human request and excerpts of the original task, system text, and recent messages: up to 12,000 serialized characters sent to TypeSafe for Jev, or 3,000 UTF-8 bytes sent to the local Ollama service. These excerpts can include private source code and tool results. Images, document payloads, and signed thinking are omitted. Full tool schemas and full conversation history are not sent to either classifier. Anthropic receives the complete request, including its tools and attachments. Large or multimodal requests may also go to Anthropic's token-count endpoint before inference, including when classification is local.
151
+ The selected evaluator receives a bounded state containing the latest human request and excerpts of the original task and recent messages: up to 12,000 serialized characters sent to TypeSafe for Jev, or 3,000 UTF-8 bytes sent to the local Ollama service. Jev also receives system-text excerpts. The local path excludes Claude's top-level executor system instructions. These excerpts can include private source code and tool results. Images, document payloads, and signed thinking are omitted. Full tool schemas and full conversation history are not sent to either classifier. Anthropic receives the complete request, including its tools and attachments. Large or multimodal requests may also go to Anthropic's token-count endpoint before inference, including when classification is local.
135
152
 
136
153
  In subscription mode, Claude Code owns login and OAuth refresh. AutoRouter forwards the current request's authorization and beta headers to Anthropic. It does not read keychain or saved login files, persist subscription tokens, or send them to Jev. A separate temporary `X-Autorouter-Token` authenticates the local connection and is stripped upstream. Subscription forwarding is restricted to `https://api.anthropic.com`. See [subscriptions and gateways](https://code.claude.com/docs/en/llm-gateway#subscriptions-and-gateways).
137
154
 
package/docs/releasing.md CHANGED
@@ -4,6 +4,8 @@ The package is `claude-autorouter`, licensed under [Apache-2.0](../LICENSE). Ver
4
4
 
5
5
  Version `0.3.1` replaces the old Ollama chat evaluator and Qwen presets with the native `/v1/systemone` endpoint on Ollama 0.35+. It supports Nimble, Tev1, and other compatible local models through `--ollama-model`; Jev remains the default remote evaluator. Existing local users should rerun setup with a supported model, as described in the [reference](reference.md#ollama-evaluator). The [evaluation report](ollama-evaluation.md) records local model latency, accuracy, and timeout limitations.
6
6
 
7
+ Version `0.3.2` fixes local timeout fallbacks with model-specific deadlines, removes Claude executor instructions from local evaluator excerpts, and keeps fallback causes visible in compact status lines. It also adds `AUTOROUTER_OLLAMA_TIMEOUT_MS=0` and `setup --ollama-timeout-ms 0` to disable the runtime evaluation deadline while preserving caller cancellation and the separate startup warmup limit. Existing explicit timeout settings still override the defaults; Jev is unchanged.
8
+
7
9
  The GitHub repository is private. Publishing to npm makes the tarball's runtime source, README, configuration example, license, and shipped documentation public. Model weights, user configuration, credentials, transcripts, local artifacts, and test fixtures are excluded. Review the archive before the first publication and whenever the package allowlist changes.
8
10
 
9
11
  ## What runs automatically
@@ -95,12 +97,12 @@ After a successful trusted release, npm recommends the optional **Publishing acc
95
97
 
96
98
  ## 3. Release subsequent versions by tag
97
99
 
98
- For the System One release, prepare `0.3.1` on `main` or through a pull request. For later releases, substitute the next unused version throughout:
100
+ The commands below illustrate the `0.3.2` release. For a new release, substitute the next unused version throughout; never reuse a published version:
99
101
 
100
102
  ```sh
101
103
  git switch main
102
104
  git pull --ff-only origin main
103
- npm version 0.3.1 --no-git-tag-version
105
+ npm version 0.3.2 --no-git-tag-version
104
106
  ```
105
107
 
106
108
  Review the version change and update any version-specific install examples or release notes. Check the candidate using the new filename:
@@ -109,7 +111,7 @@ Review the version change and update any version-specific install examples or re
109
111
  npm run check
110
112
  npm test
111
113
  npm run release:pack
112
- npm run test:package -- --archive ./dist/claude-autorouter-0.3.1.tgz
114
+ npm run test:package -- --archive ./dist/claude-autorouter-0.3.2.tgz
113
115
  git diff --check
114
116
  ```
115
117
 
@@ -117,7 +119,7 @@ Commit the intended release changes and get that commit onto `main`, either thro
117
119
 
118
120
  ```sh
119
121
  git add package.json
120
- git commit -m "Release 0.3.1"
122
+ git commit -m "Release 0.3.2"
121
123
  git push origin main
122
124
  ```
123
125
 
@@ -126,24 +128,24 @@ Include any intentional documentation or release-note edits in that commit too.
126
128
  ```sh
127
129
  git switch main
128
130
  git pull --ff-only origin main
129
- git tag -a v0.3.1 -m "Release 0.3.1"
130
- git push origin v0.3.1
131
+ git tag -a v0.3.2 -m "Release 0.3.2"
132
+ git push origin v0.3.2
131
133
  ```
132
134
 
133
- Before pushing, confirm `package.json` contains `0.3.1` and the tag points to the intended commit. For a prerelease, use a matching version/tag such as `0.4.0-beta.1` / `v0.4.0-beta.1`; it will publish under `next`, leaving `latest` unchanged.
135
+ Before pushing, confirm `package.json` contains `0.3.2` and the tag points to the intended commit. For a prerelease, use a matching version/tag such as `0.4.0-beta.1` / `v0.4.0-beta.1`; it will publish under `next`, leaving `latest` unchanged.
134
136
 
135
137
  Release stable versions in increasing version order, one tag at a time, and wait for each run to finish before pushing the next stable tag. The workflow queues releases without canceling an active run, but queue order does not sort semantic versions. Publishing an older stable version afterward could move `latest` backward; there is no registry version-order gate.
136
138
 
137
139
  Open the tag's run under [GitHub Actions](https://github.com/frapposelli/claude-autorouter/actions). Under **Artifacts**, download `npm-package-<run-id>-<run-attempt>`, which contains the `.tgz` and checksum used for publication. Artifacts expire after 30 days, so retain them with the release record. After the publish job succeeds, verify the registry version and tags:
138
140
 
139
141
  ```sh
140
- npm view claude-autorouter@0.3.1 version dist.integrity --registry https://registry.npmjs.org/
142
+ npm view claude-autorouter@0.3.2 version dist.integrity --registry https://registry.npmjs.org/
141
143
  npm view claude-autorouter dist-tags --json --registry https://registry.npmjs.org/
142
144
  ```
143
145
 
144
146
  Repeat the independent installation check for the released version. A GitHub Release page is optional; pushing the version tag is the publication trigger.
145
147
 
146
- For local release diagnostics after the tag exists, `node scripts/release-check.mjs source v0.3.1` checks the tag, clean checkout, metadata, and ancestry. `node scripts/release-check.mjs archive v0.3.1` checks the candidate checksum and contents; `dist/` must contain only that version's archive and checksum, so retain older artifacts elsewhere first. These helpers are run automatically in the release workflow; the first untagged bootstrap uses the checks in step 1 instead.
148
+ For local release diagnostics after the tag exists, `node scripts/release-check.mjs source v0.3.2` checks the tag, clean checkout, metadata, and ancestry. `node scripts/release-check.mjs archive v0.3.2` checks the candidate checksum and contents; `dist/` must contain only that version's archive and checksum, so retain older artifacts elsewhere first. These helpers are run automatically in the release workflow; the first untagged bootstrap uses the checks in step 1 instead.
147
149
 
148
150
  ## Recovering a failed release
149
151
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "claude-autorouter",
3
- "version": "0.3.1",
3
+ "version": "0.3.2",
4
4
  "license": "Apache-2.0",
5
5
  "type": "module",
6
6
  "description": "A Claude Code model router with Jev and local Ollama System One evaluators",
@@ -16,6 +16,7 @@
16
16
  "claude": "node bin/autorouter.mjs claude",
17
17
  "test": "node --test test/*.test.mjs",
18
18
  "test:package": "node scripts/package-smoke.mjs",
19
+ "test:ollama": "node scripts/test-ollama-routing.mjs",
19
20
  "release:pack": "node scripts/release-pack.mjs",
20
21
  "eval": "node --env-file=.env scripts/evaluate.mjs",
21
22
  "eval:ollama": "node scripts/evaluate-ollama.mjs",
package/src/config.mjs CHANGED
@@ -1,4 +1,4 @@
1
- import { DEFAULT_OLLAMA_MODEL, validateOllamaEndpoint, validateOllamaModel } from './ollama-models.mjs';
1
+ import { DEFAULT_OLLAMA_MODEL, defaultOllamaTimeoutMs, validateOllamaEndpoint, validateOllamaModel } from './ollama-models.mjs';
2
2
 
3
3
  export const TIERS = ['haiku', 'sonnet', 'opus'];
4
4
 
@@ -22,6 +22,14 @@ function endpoint(value, name) {
22
22
  export function readConfig(env = process.env) {
23
23
  const evaluator = env.AUTOROUTER_EVALUATOR ?? 'jev';
24
24
  if (!['jev', 'ollama'].includes(evaluator)) throw new Error('AUTOROUTER_EVALUATOR must be jev or ollama');
25
+ const ollamaModel = validateOllamaModel(env.AUTOROUTER_OLLAMA_MODEL ?? DEFAULT_OLLAMA_MODEL);
26
+ const rawOllamaTimeout = env.AUTOROUTER_OLLAMA_TIMEOUT_MS;
27
+ // Zero is an explicit opt-out. Reject blanks, coercible non-numbers, and
28
+ // non-integer strings (including exponents that could underflow to zero).
29
+ if (rawOllamaTimeout !== undefined && (!['string', 'number'].includes(typeof rawOllamaTimeout)
30
+ || (typeof rawOllamaTimeout === 'string' && !/^[0-9]+$/.test(rawOllamaTimeout.trim())))) {
31
+ throw new Error('AUTOROUTER_OLLAMA_TIMEOUT_MS must be an integer between 0 and 30000 (0 disables the runtime deadline)');
32
+ }
25
33
  const ollamaKeepAlive = env.AUTOROUTER_OLLAMA_KEEP_ALIVE ?? '5m';
26
34
  if (!/^(?:0|[1-9]\d{0,3}(?:s|m|h))$/.test(ollamaKeepAlive)) {
27
35
  throw new Error('AUTOROUTER_OLLAMA_KEEP_ALIVE must be 0 or a positive duration such as 5m');
@@ -49,8 +57,8 @@ export function readConfig(env = process.env) {
49
57
  jevEndpoint: endpoint(env.AUTOROUTER_JEV_URL ?? 'https://api.typesafe.ai/v1/systemone', 'AUTOROUTER_JEV_URL'),
50
58
  jevModel: env.AUTOROUTER_JEV_MODEL ?? 'jev-latest',
51
59
  ollamaEndpoint: validateOllamaEndpoint(env.AUTOROUTER_OLLAMA_URL ?? 'http://127.0.0.1:11434'),
52
- ollamaModel: validateOllamaModel(env.AUTOROUTER_OLLAMA_MODEL ?? DEFAULT_OLLAMA_MODEL),
53
- ollamaTimeoutMs: number(env, 'AUTOROUTER_OLLAMA_TIMEOUT_MS', 1500, 1, 30000),
60
+ ollamaModel,
61
+ ollamaTimeoutMs: number(env, 'AUTOROUTER_OLLAMA_TIMEOUT_MS', defaultOllamaTimeoutMs(ollamaModel), 0, 30000),
54
62
  ollamaStateChars: 3000,
55
63
  ollamaKeepAlive,
56
64
  models: {
@@ -4,11 +4,17 @@ import { buildState } from './prompt-state.mjs';
4
4
  // Bound UTF-8 bytes as well as serialized characters to keep local decision
5
5
  // excerpts small, including when the prompt contains non-ASCII text.
6
6
  export function buildOllamaState(body, limit = 3000) {
7
+ // Claude's executor instructions describe the assistant and its tools, not
8
+ // the current task's difficulty. Small local models can mistake that global
9
+ // engineering background for a request and select Sonnet for every tier.
10
+ // Keep task/history extraction shared with Jev, but exclude this background
11
+ // before budgeting so it cannot displace a useful follow-up or tool result.
12
+ const taskBody = { ...body, system: undefined };
7
13
  let budget = limit;
8
- let state = buildState(body, budget);
14
+ let state = buildState(taskBody, budget);
9
15
  while (Buffer.byteLength(JSON.stringify(state)) > limit && budget > 200) {
10
16
  budget = Math.max(200, Math.floor(budget * limit / Buffer.byteLength(JSON.stringify(state))) - 1);
11
- state = buildState(body, budget);
17
+ state = buildState(taskBody, budget);
12
18
  }
13
19
  return state;
14
20
  }
@@ -122,8 +128,15 @@ export async function checkLocalOllamaModel(config, options) {
122
128
  }
123
129
 
124
130
  export async function evaluateOllama(state, config, { fetchImpl = fetch, signal } = {}) {
125
- const timeout = AbortSignal.timeout(config.ollamaTimeoutMs);
126
- const combined = signal ? AbortSignal.any([signal, timeout]) : timeout;
131
+ let combined = signal;
132
+ if (config.ollamaTimeoutMs !== 0) {
133
+ const timeout = AbortSignal.timeout(config.ollamaTimeoutMs);
134
+ combined = signal ? AbortSignal.any([signal, timeout]) : timeout;
135
+ }
136
+ // A disabled deadline still observes caller cancellation during metadata,
137
+ // inference, and response reading. Direct callers may omit a signal.
138
+ combined ??= new AbortController().signal;
139
+ combined.throwIfAborted();
127
140
  const options = { fetchImpl, signal: combined };
128
141
  await checkLocalOllamaModel(config, options);
129
142
  return decisionAnswer(await request(config, '/v1/systemone', buildOllamaRequest(state, config), options), config.ollamaModel);
@@ -1,5 +1,20 @@
1
1
  export const DEFAULT_OLLAMA_MODEL = 'nimble:9b-q4_K_M';
2
2
 
3
+ // Known model families need different runtime budgets. Restrict recognition to
4
+ // the official library and standard quantization tags; custom namespaces and
5
+ // unknown variants retain the short default. Explicit configuration wins.
6
+ const QUANTIZATION = '(?:q[2-8]_(?:0|1|k(?:_[sml])?)|iq[1-4]_(?:xxs|xs|s|m|nl)|f16|bf16|f32)';
7
+ const TEV_4B = new RegExp(`^(?:latest|4b(?:-${QUANTIZATION})?)$`, 'i');
8
+ const NIMBLE_9B = new RegExp(`^(?:latest|9b(?:-${QUANTIZATION})?)$`, 'i');
9
+
10
+ export function defaultOllamaTimeoutMs(model) {
11
+ const canonical = model.replace(/^registry\.ollama\.ai\//, '').replace(/^library\//, '');
12
+ const match = /^(tev1|nimble)(?::([^:]+))?$/.exec(canonical);
13
+ if (match?.[1] === 'tev1' && TEV_4B.test(match[2] ?? 'latest')) return 15000;
14
+ if (match?.[1] === 'nimble' && NIMBLE_9B.test(match[2] ?? 'latest')) return 30000;
15
+ return 1500;
16
+ }
17
+
3
18
  export function validateOllamaEndpoint(value) {
4
19
  let endpoint;
5
20
  try { endpoint = new URL(value); } catch {}
@@ -11,6 +11,9 @@ import { inspectOllama, setupOllama } from './ollama-setup.mjs';
11
11
 
12
12
  const execute = promisify(execFile);
13
13
 
14
+ export const ollamaDeadlineText = timeoutMs => timeoutMs === 0
15
+ ? 'routing deadline disabled' : `routing deadline ${timeoutMs} ms per request`;
16
+
14
17
  // Readline manages editing and restores terminal state; its output is discarded
15
18
  // so neither typing nor pasted credentials are echoed to the terminal.
16
19
  export async function askSecret(label, { input = process.stdin, output = process.stderr } = {}) {
@@ -36,19 +39,27 @@ export async function setup(args, {
36
39
  let authMode = env.AUTOROUTER_AUTH_MODE ?? 'subscription';
37
40
  let evaluator = env.AUTOROUTER_EVALUATOR ?? 'jev';
38
41
  let model;
42
+ let ollamaTimeoutMs;
39
43
  let pull = false;
40
44
  let overwrite = false;
41
45
  for (let i = 0; i < args.length; i++) {
42
46
  if (args[i] === '--auth-mode') authMode = args[++i];
43
47
  else if (args[i] === '--evaluator') evaluator = args[++i];
44
48
  else if (args[i] === '--ollama-model') { model = args[++i]; if (model === undefined) throw new Error('--ollama-model requires a model tag'); }
49
+ else if (args[i] === '--ollama-timeout-ms') {
50
+ const value = args[++i];
51
+ if (typeof value !== 'string' || !/^[0-9]+$/.test(value.trim()) || Number(value) > 30000) {
52
+ throw new Error('--ollama-timeout-ms requires an integer between 0 and 30000 (0 disables the routing deadline)');
53
+ }
54
+ ollamaTimeoutMs = String(Number(value));
55
+ }
45
56
  else if (args[i] === '--pull') pull = true;
46
57
  else if (args[i] === '--force') overwrite = true;
47
- else throw new Error('Usage: claude-autorouter setup [--auth-mode subscription|api-key] [--evaluator jev|ollama] [--ollama-model TAG] [--pull] [--force]');
58
+ else throw new Error('Usage: claude-autorouter setup [--auth-mode subscription|api-key] [--evaluator jev|ollama] [--ollama-model TAG] [--ollama-timeout-ms N] [--pull] [--force]');
48
59
  }
49
60
  if (!['subscription', 'api-key'].includes(authMode)) throw new Error('--auth-mode must be subscription or api-key');
50
61
  if (!['jev', 'ollama'].includes(evaluator)) throw new Error('--evaluator must be jev or ollama');
51
- if (evaluator !== 'ollama' && (model !== undefined || pull)) throw new Error('Ollama model and download options require --evaluator ollama');
62
+ if (evaluator !== 'ollama' && (model !== undefined || ollamaTimeoutMs !== undefined || pull)) throw new Error('Ollama model, deadline and download options require --evaluator ollama');
52
63
  const path = getConfigPath(env);
53
64
  if (!overwrite && existsSync(path)) throw new Error('AutoRouter configuration already exists. Use setup --force to replace it.');
54
65
  write(evaluator === 'ollama'
@@ -60,7 +71,7 @@ export async function setup(args, {
60
71
  for (const key of ['AUTOROUTER_OLLAMA_URL', 'AUTOROUTER_OLLAMA_TIMEOUT_MS', 'AUTOROUTER_OLLAMA_KEEP_ALIVE']) {
61
72
  if (env[key] !== undefined) values[key] = env[key];
62
73
  }
63
- write(`Local evaluator model: ${values.AUTOROUTER_OLLAMA_MODEL}.`);
74
+ if (ollamaTimeoutMs !== undefined) values.AUTOROUTER_OLLAMA_TIMEOUT_MS = ollamaTimeoutMs;
64
75
  }
65
76
  const keys = [...(evaluator === 'jev' ? ['TYPESAFE_API_KEY'] : []), ...(authMode === 'api-key' ? ['ANTHROPIC_API_KEY'] : [])];
66
77
  for (const key of keys) {
@@ -71,6 +82,7 @@ export async function setup(args, {
71
82
  const config = readConfig(values);
72
83
  requireKeys(config);
73
84
  if (evaluator === 'ollama') {
85
+ write(`Local evaluator: ${config.ollamaModel}; ${ollamaDeadlineText(config.ollamaTimeoutMs)}.`);
74
86
  const controller = new AbortController();
75
87
  const cancel = () => controller.abort();
76
88
  if (!signal) for (const name of ['SIGINT', 'SIGTERM']) process.once(name, cancel);
@@ -104,6 +116,8 @@ export async function doctor({ env = process.env, write = console.log, run = exe
104
116
  report(false, `Unset ${key}; AutoRouter uses the Anthropic Messages API`);
105
117
  }
106
118
  if (config?.evaluator === 'ollama') {
119
+ write(`Local evaluator: ${config.ollamaModel}; ${ollamaDeadlineText(config.ollamaTimeoutMs)}.`);
120
+ write('Model availability is checked below; classification speed and accuracy are not tested.');
107
121
  try {
108
122
  const result = await inspectOllama(config, { fetchImpl, signal });
109
123
  report(result.installed, result.installed
@@ -6,6 +6,9 @@ const REASONS = {
6
6
  context_capacity: 'large context',
7
7
  internal_request: 'internal request', unknown_model: 'custom model', low_confidence: 'low confidence',
8
8
  };
9
+ const CLASSIFIER_ERRORS = {
10
+ timeout: 'timeout', http_error: 'HTTP error', invalid_response: 'invalid response', network_error: 'network error',
11
+ };
9
12
  const clean = (value, limit = 64) => typeof value === 'string' ? value
10
13
  .replace(/\x1b\][\s\S]*?(?:\x07|\x1b\\)/g, '')
11
14
  .replace(/\x1b\[[0-?]*[ -/]*[@-~]/g, '')
@@ -125,21 +128,28 @@ export function renderStatusLine(input, snapshot, { now = Date.now(), color = tr
125
128
 
126
129
  const details = [];
127
130
  const source = ['jev', 'ollama', 'cache', 'fallback'].includes(state?.source) ? state.source : undefined;
131
+ const evaluator = ['jev', 'ollama'].includes(state?.evaluator) ? state.evaluator : undefined;
132
+ const evaluatorLabel = evaluator === 'ollama' ? 'Ollama' : evaluator === 'jev' ? 'Jev' : '';
133
+ let fallbackCause = '';
134
+ let fallbackPhase = '';
135
+ let compactFallback = false;
128
136
  if (source) {
129
- const evaluator = ['jev', 'ollama'].includes(state.evaluator) ? state.evaluator : undefined;
130
- const evaluatorLabel = evaluator === 'ollama' ? 'Ollama' : evaluator === 'jev' ? 'Jev' : '';
131
137
  const sourceLabel = source === 'jev' ? 'Jev' : source === 'ollama' ? 'Ollama'
132
138
  : evaluatorLabel ? `${evaluatorLabel} ${source}` : source;
133
139
  const timing = Number.isFinite(state.latency_ms) && state.latency_ms >= 0 ? ` ${Math.round(Math.min(state.latency_ms, 999999))}ms` : '';
134
140
  const classified = ['haiku', 'sonnet', 'opus'].includes(state.classified_tier) ? state.classified_tier : undefined;
135
141
  const chosenFamily = /^claude-(haiku|sonnet|opus)-/.exec(state.selected_model ?? '')?.[1];
136
- const override = classified && chosenFamily && classified !== chosenFamily
142
+ const override = source !== 'fallback' && classified && chosenFamily && classified !== chosenFamily
137
143
  ? `→${classified[0].toUpperCase()}${classified.slice(1)}` : '';
138
144
  details.push(`${sourceLabel}${override}${timing}`);
139
145
  }
140
146
  if (source === 'fallback') {
141
- const classifierError = clean(state.classifier_error, 24);
142
- if (classifierError) details.push(classifierError.replaceAll('_', ' '));
147
+ // Snapshots normally contain allowlisted categories, but the renderer also
148
+ // rejects raw error messages so paths and provider response text stay out.
149
+ fallbackCause = Object.hasOwn(CLASSIFIER_ERRORS, state.classifier_error) ? CLASSIFIER_ERRORS[state.classifier_error] : '';
150
+ if (state.classifier_error === 'http_error' && Number.isInteger(state.classifier_status)
151
+ && state.classifier_status >= 100 && state.classifier_status <= 599) fallbackCause = `HTTP ${state.classifier_status}`;
152
+ if (fallbackCause) details.push(fallbackCause);
143
153
  }
144
154
  if (Object.hasOwn(REASONS, state?.reason)) details.push(state.reason === 'context_capacity' && state.context_check === 'count_unavailable'
145
155
  ? 'size unverified' : REASONS[state.reason]);
@@ -157,11 +167,23 @@ export function renderStatusLine(input, snapshot, { now = Date.now(), color = tr
157
167
  if (width(plain()) > available && source !== 'fallback' && phase !== 'error') detail = '';
158
168
  if (width(plain()) > available && saving) saving = savings.compact;
159
169
  if (width(plain()) > available) saving = '';
170
+ if (width(plain()) > available && source === 'fallback') {
171
+ // A successful Claude response can still follow evaluator failure. When
172
+ // space is tight, retain that cause instead of an ordinary "ready" phase,
173
+ // evaluator timing, or the guard details that followed the fallback.
174
+ compactFallback = true;
175
+ fallbackPhase = ['error', 'cancelled'].includes(phase) ? status : '';
176
+ status = [fallbackPhase, `${evaluatorLabel ? `${evaluatorLabel} ` : ''}fallback${fallbackCause ? `: ${fallbackCause}` : ''}`].filter(Boolean).join(' · ');
177
+ detail = '';
178
+ }
160
179
  if (width(plain()) > available) detail = '';
161
180
  if (width(plain()) > available) brand = '● AR';
162
181
  if (width(plain()) > available) brand = '';
182
+ if (width(plain()) > available && compactFallback) {
183
+ status = [fallbackPhase, `fallback${fallbackCause ? `: ${fallbackCause}` : ''}`].filter(Boolean).join(' · ');
184
+ }
163
185
  if (width(plain()) > available && model) {
164
- if (phase === 'streaming') status = 'stream';
186
+ if (phase === 'streaming' && !compactFallback) status = 'stream';
165
187
  const room = available - width(prefix + suffix + status) - 3;
166
188
  if (room >= 1) model = shorten(model, room);
167
189
  else {
@@ -172,6 +194,8 @@ export function renderStatusLine(input, snapshot, { now = Date.now(), color = tr
172
194
  else if (width('● AR · ' + status) <= available) brand = '● AR';
173
195
  }
174
196
  }
197
+ if (width(plain()) > available && compactFallback && !fallbackPhase && available >= width('fallback')
198
+ && available < width('fallback: ') + 2) status = 'fallback';
175
199
  if (width(plain()) > available) status = shorten(status, Math.max(1, available - width(modelLabel()) - (model ? 3 : 0)));
176
200
  const attention = phase === 'error' ? '31' : source === 'fallback' ? '33' : phase === 'cancelled' ? '2' : phase === 'streaming' || phase === 'ready' ? '32' : '36';
177
201
  const chunks = [];