@alexeiled/claude-router 0.6.0 → 0.6.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/plugin.json +1 -1
- package/README.md +22 -16
- package/docs/{design.md → architecture.md} +118 -60
- package/docs/configuration.md +23 -14
- package/docs/tier-share.svg +44 -0
- package/docs/user-guide.md +32 -9
- package/lib/config.mjs +7 -1
- package/lib/cost.mjs +53 -8
- package/lib/facts.mjs +2 -0
- package/lib/gateway.mjs +1 -1
- package/lib/policy.mjs +16 -12
- package/lib/router.mjs +52 -22
- package/lib/status.mjs +44 -3
- package/package.json +1 -1
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "router",
|
|
3
3
|
"displayName": "jev-router",
|
|
4
|
-
"version": "0.6.
|
|
4
|
+
"version": "0.6.1",
|
|
5
5
|
"description": "Auto-picks the right Claude model for each turn — small models for quick edits, mid-tier for code, frontier for hard problems. Uses Jev to classify each request.",
|
|
6
6
|
"author": { "name": "Alexei Ledenev", "url": "https://github.com/alexei-led" },
|
|
7
7
|
"repository": "https://github.com/alexei-led/claude-router",
|
package/README.md
CHANGED
|
@@ -7,11 +7,17 @@
|
|
|
7
7
|
|
|
8
8
|
A Claude Code plugin that auto-picks the right model and effort for each turn.
|
|
9
9
|
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
10
|
+
**Status: experimental.** I built this to dogfood Jev, TypeSafe's routing
|
|
11
|
+
model, inside Claude Code. It works for me; I don't know yet if it holds up
|
|
12
|
+
outside my setup. Try it and open an issue with what you find.
|
|
13
|
+
|
|
14
|
+
## Why
|
|
15
|
+
|
|
16
|
+
One model for every turn is a compromise: strong enough for the hard turns
|
|
17
|
+
and it burns your limits on "rename this variable"; cheap enough for the easy
|
|
18
|
+
turns and it struggles on the hard ones. Switching models by hand works, but
|
|
19
|
+
it is friction you pay on every message. This plugin asks a small router
|
|
20
|
+
model which tier a turn needs and switches for you, automatically.
|
|
15
21
|
|
|
16
22
|
## How it works
|
|
17
23
|
|
|
@@ -26,20 +32,23 @@ only requests for `jev-router`: ├─ facts.mjs prompt, continuation,
|
|
|
26
32
|
responses go through unchanged; the gateway reads `usage` (context size, cache TTL)
|
|
27
33
|
```
|
|
28
34
|
|
|
35
|
+
A local gateway on `127.0.0.1` receives each request from Claude Code. For a
|
|
36
|
+
new user turn, the gateway asks Jev which tier the turn needs, then rewrites
|
|
37
|
+
`model`, `effort` and `thinking` and sends the request to Anthropic. All other
|
|
38
|
+
data goes through unchanged. No runtime dependencies; Node 22 or later.
|
|
39
|
+
|
|
29
40
|
The tiers are `micro` (Haiku), `low` (Sonnet, the baseline), `medium` (Opus at
|
|
30
41
|
high effort) and `high` (Opus at xhigh effort). The exact model IDs are in
|
|
31
42
|
`~/.claude/router.json` and default to the current generation of each family.
|
|
32
43
|
|
|
33
44
|
A tool continuation keeps the route of its turn — the gateway does not ask Jev.
|
|
34
45
|
Side requests, for example session titles, get the baseline tier. A subagent
|
|
35
|
-
that inherits the model gets routing with its own memory. A request for
|
|
36
|
-
|
|
46
|
+
that inherits the model gets routing with its own memory. A request for any
|
|
47
|
+
other model goes through unchanged; this is how `/router:<tier>` pins and
|
|
37
48
|
subagents with their own `model` work.
|
|
38
49
|
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
`store`. The modules `facts`, `cost`, `rewrite`, `sse` and `policy` are pure.
|
|
42
|
-
Only `store` writes files. The Jev transport is injected.
|
|
50
|
+
See [Architecture](docs/architecture.md) for the module map, the switching
|
|
51
|
+
policy, and a real-usage evaluation.
|
|
43
52
|
|
|
44
53
|
## Install
|
|
45
54
|
|
|
@@ -79,14 +88,11 @@ plugin never replaces a newer gateway. The Jev API key stays in the macOS
|
|
|
79
88
|
Keychain. Run `/router:setup` once after an update: the status line command
|
|
80
89
|
path contains the plugin version.
|
|
81
90
|
|
|
82
|
-
Updating from 0.3.0 or earlier: these gateways cannot hand over, so stop the
|
|
83
|
-
old one once with `pkill -f scripts/gateway.mjs`.
|
|
84
|
-
|
|
85
91
|
## Documentation
|
|
86
92
|
|
|
87
93
|
- [User guide](docs/user-guide.md): daily use, pins, decision log, troubleshooting.
|
|
88
94
|
- [Configuration](docs/configuration.md): each key, and where the API key and the configuration file are.
|
|
89
|
-
- [
|
|
95
|
+
- [Architecture](docs/architecture.md): how the gateway works, the switching policy, a real-usage evaluation.
|
|
90
96
|
|
|
91
97
|
## Develop
|
|
92
98
|
|
|
@@ -102,4 +108,4 @@ claude --plugin-dir . --model jev-router # with ANTHROPIC_BASE_URL and TYPESAF
|
|
|
102
108
|
Releases: push a signed tag `v<version>` that matches `package.json`. The
|
|
103
109
|
release workflow publishes `@alexeiled/claude-router` to npm with trusted
|
|
104
110
|
publishing and creates the GitHub release. See
|
|
105
|
-
[docs/
|
|
111
|
+
[docs/architecture.md](docs/architecture.md#release).
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
#
|
|
1
|
+
# Architecture
|
|
2
2
|
|
|
3
3
|
Each decision has the date when the owner took it.
|
|
4
4
|
|
|
@@ -21,12 +21,17 @@ get the context size, the cache reads and the cache TTL.
|
|
|
21
21
|
For a model without thinking (Haiku), the gateway also removes the
|
|
22
22
|
`clear_thinking_*` edits from `context_management`, because the API rejects
|
|
23
23
|
them without thinking. The gateway never changes `system`, `tools` or
|
|
24
|
-
`messages`.
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
24
|
+
`messages`. Thinking and prompt caching work as if Claude Code talked to
|
|
25
|
+
Anthropic directly. Claude Code documents this gateway mode, including the
|
|
26
|
+
OAuth value for a claude.ai login. See
|
|
27
|
+
[llm-gateway](https://code.claude.com/docs/en/llm-gateway) and
|
|
28
28
|
[protocol](https://code.claude.com/docs/en/llm-gateway-protocol).
|
|
29
29
|
|
|
30
|
+
Module dependencies point in one direction: `gateway.mjs` (HTTP) →
|
|
31
|
+
`router.mjs` (orchestration) → `facts`, `jev`, `policy` → `cost`, `rewrite`,
|
|
32
|
+
`store`. The modules `facts`, `cost`, `rewrite`, `sse` and `policy` are pure.
|
|
33
|
+
Only `store` writes files. The Jev transport is injected.
|
|
34
|
+
|
|
30
35
|
The `SessionStart` hook of the plugin starts the gateway when the port does not
|
|
31
36
|
answer. `/router:setup` writes `model`, `ANTHROPIC_BASE_URL` and the picker row to
|
|
32
37
|
the user settings once. A plugin cannot set them by itself.
|
|
@@ -38,22 +43,14 @@ is the only place where the real model of the turn is visible.
|
|
|
38
43
|
|
|
39
44
|
### Why not the native skill path
|
|
40
45
|
|
|
41
|
-
The first design used a `UserPromptSubmit` hook
|
|
42
|
-
|
|
43
|
-
the rest of the turn.
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
ran in three sessions, in auto mode and in `acceptEdits` mode, with Opus and
|
|
50
|
-
Fable targets. The documentation says "when this skill is active". The
|
|
51
|
-
behavior is user invocation only. A feedback report is filed. The skills stay
|
|
52
|
-
as manual pins.
|
|
53
|
-
|
|
54
|
-
This result also removed the "Sonnet session model" argument from the peer
|
|
55
|
-
review. The gateway has no bootstrap request and no cache drop for a turn that
|
|
56
|
-
keeps its route. The baseline is a configuration value.
|
|
46
|
+
The first design used a `UserPromptSubmit` hook that asked Claude to call a
|
|
47
|
+
tier skill, relying on the skill's frontmatter `model:` and `effort:` to serve
|
|
48
|
+
the rest of the turn. On Claude Code 2.1.278 that only works for a skill the
|
|
49
|
+
user types (`/router:medium …`); the same skill called by Claude through the
|
|
50
|
+
Skill tool does not change the model — the transcript records
|
|
51
|
+
`attributionSkill`, but the session model answers. So `/router:<tier>` skills
|
|
52
|
+
stay as manual pins, and the gateway is the only path that routes turns Claude
|
|
53
|
+
calls on its own.
|
|
57
54
|
|
|
58
55
|
## Tiers
|
|
59
56
|
|
|
@@ -64,10 +61,9 @@ keeps its route. The baseline is a configuration value.
|
|
|
64
61
|
| low | sonnet | as sent | claude-sonnet-5 |
|
|
65
62
|
| micro | haiku | none | claude-haiku-4-5 |
|
|
66
63
|
|
|
67
|
-
The gateway lowers the effort to a level
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
account.
|
|
64
|
+
The gateway lowers the effort to a level the model's `efforts` list accepts
|
|
65
|
+
(see [configuration](configuration.md#models)); Haiku accepts none, so it gets
|
|
66
|
+
no effort and no adaptive thinking. The ids are configuration.
|
|
71
67
|
|
|
72
68
|
## Request classes
|
|
73
69
|
|
|
@@ -86,6 +82,10 @@ account.
|
|
|
86
82
|
429, a 529 or a dropped connection.
|
|
87
83
|
- A side endpoint with the alias, such as `/v1/messages/count_tokens`: the
|
|
88
84
|
model of the session's last route. Only `POST /v1/messages` is a turn.
|
|
85
|
+
- A history break: a main request with fewer messages than the last one (a
|
|
86
|
+
compaction or a rewind), or the header `x-claude-code-context-compacted`.
|
|
87
|
+
The gateway drops the cached prefixes, the votes and the escalation hold,
|
|
88
|
+
then routes the request as usual. The route stays until the next decision.
|
|
89
89
|
|
|
90
90
|
## Failure handling
|
|
91
91
|
|
|
@@ -103,7 +103,7 @@ request must not reach the others.
|
|
|
103
103
|
logged. The routing decision stands.
|
|
104
104
|
- Jev: one retry on a network error or a transient status, after
|
|
105
105
|
`Retry-After` when it fits the 1.5 s budget. After three failures in a row,
|
|
106
|
-
new turns skip Jev for a minute, then try once.
|
|
106
|
+
new turns skip Jev for a minute, then try once per pause.
|
|
107
107
|
- The daemon logs a stray exception instead of exiting. On `SIGTERM` it
|
|
108
108
|
releases the port at once and finishes open streams for up to 10 minutes.
|
|
109
109
|
A second daemon on a busy port exits quietly.
|
|
@@ -127,18 +127,20 @@ request must not reach the others.
|
|
|
127
127
|
All inputs come from the traffic of the gateway. The gateway does not read
|
|
128
128
|
transcripts.
|
|
129
129
|
|
|
130
|
-
| Input | Source
|
|
131
|
-
| ------------------------------------------------ |
|
|
132
|
-
| Context of the last request, cache reads, output | `usage` in the response (`message_start` and `message_delta`)
|
|
133
|
-
| Granted TTL | `usage.cache_creation.ephemeral_1h_input_tokens` or the `5m` field
|
|
134
|
-
| Cache
|
|
135
|
-
|
|
|
136
|
-
|
|
|
137
|
-
|
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
|
|
130
|
+
| Input | Source |
|
|
131
|
+
| ------------------------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------ |
|
|
132
|
+
| Context of the last request, cache reads, output | `usage` in the response (`message_start` and `message_delta`) |
|
|
133
|
+
| Granted TTL | `usage.cache_creation.ephemeral_1h_input_tokens` or the `5m` field |
|
|
134
|
+
| Cache identity | The model id and the effort the gateway sent (`claude-opus-5-5@xhigh`). An effort change rewrites the messages cache, so each effort is its own cache. |
|
|
135
|
+
| Cache warmth | The time of the last response for that cache, plus the TTL, minus 30 s. `unknown` when no response since the session started or the history broke. |
|
|
136
|
+
| Reusable prefix | The context plus the output at the last response for that cache. Cleared by a history break, or when the context shrinks by more than 20% (context editing). |
|
|
137
|
+
| Failure signal | Two `tool_result` blocks with `is_error` and the same signature, with an edit tool call between them |
|
|
138
|
+
| Continuation | The last message contains a `tool_result`. For a new prompt, a Jev Noul answers "does this prompt continue the task". |
|
|
139
|
+
|
|
140
|
+
Prices are a list-price table in the configuration (see
|
|
141
|
+
[configuration](configuration.md#models)). `modelPricing` is a managed setting
|
|
142
|
+
and is not readable. The switching tax for a candidate `c` against the current
|
|
143
|
+
route `i` is:
|
|
142
144
|
|
|
143
145
|
```
|
|
144
146
|
input_cost(m) = P_read(m) * W_m + P_write(m) * (N - W_m)
|
|
@@ -148,11 +150,13 @@ tax = max(0, input_cost(c) - input_cost(i))
|
|
|
148
150
|
The subscription economics are not symmetric. Every default model uses the plan
|
|
149
151
|
limits, and dollars give the order between them. A model with `billing:
|
|
150
152
|
"credits"` bills cash, on the 5m TTL, and behind the gateway without the
|
|
151
|
-
consent prompt of Claude Code. `policy.cashCapUsd`
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
153
|
+
consent prompt of Claude Code. `policy.cashCapUsd` is a cold-write guard: it
|
|
154
|
+
caps the estimated first cache write of an automatic route to such a model
|
|
155
|
+
when its cache is not warm. It is not a budget: a warm cache passes, and
|
|
156
|
+
output is not counted. A Claude Code turn starts at about 100k tokens (system
|
|
157
|
+
prompt and tool definitions), so the guard binds on the first switch, not
|
|
158
|
+
later. No default model bills credits; the guard stays for configurations that
|
|
159
|
+
add one.
|
|
156
160
|
|
|
157
161
|
## Switching policy v0
|
|
158
162
|
|
|
@@ -171,8 +175,9 @@ Agreed with Codex on 2026-09-22. The thresholds are start values.
|
|
|
171
175
|
`U >= 0.75 + 0.15 * tax / (tax + 0.5)`. A jump of two tiers with
|
|
172
176
|
`U >= 0.95` skips the delay. A downgrade needs `D >= 0.90` and two
|
|
173
177
|
consecutive votes.
|
|
174
|
-
5.
|
|
175
|
-
cold write below the cap. Otherwise the
|
|
178
|
+
5. Cold-write guard (reason `cash-gate`). An automatic route to a `credits`
|
|
179
|
+
model needs a warm cache, or a cold write below the cap. Otherwise the
|
|
180
|
+
strongest `plan` tier serves.
|
|
176
181
|
6. No cooldown on upgrades. Plan, then execute, then hard again is sometimes
|
|
177
182
|
the correct routing. The log separates reversals from real changes in the
|
|
178
183
|
required capability.
|
|
@@ -180,19 +185,64 @@ Agreed with Codex on 2026-09-22. The thresholds are start values.
|
|
|
180
185
|
and cache reads of each routed response. The memory changes only from
|
|
181
186
|
responses that the gateway sent.
|
|
182
187
|
|
|
183
|
-
##
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
-
|
|
189
|
-
the
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
|
|
188
|
+
## Cache identity, history breaks and shadow economics (2026-09-23)
|
|
189
|
+
|
|
190
|
+
From a design review with the architect of pi-model-router, the Pi router
|
|
191
|
+
that uses the same Jev tiers.
|
|
192
|
+
|
|
193
|
+
- Cache identity is the model and the effort. A top-level effort change
|
|
194
|
+
invalidates the messages cache; the cache-preserving per-message effort is
|
|
195
|
+
not available on Opus 5.5. Before 0.6.1 the key was the model alone, so
|
|
196
|
+
`medium` ↔ `high` (Opus at `high` and `xhigh`) looked like a free switch
|
|
197
|
+
between warm caches.
|
|
198
|
+
- A history break (fewer messages, or the compaction header) drops the cached
|
|
199
|
+
prefixes, the votes and the escalation hold: they were about turns that are
|
|
200
|
+
gone. Context editing shrinks the context but keeps the messages; it drops
|
|
201
|
+
only the prefixes. The gateway has no branch id, so it does not restore state
|
|
202
|
+
from before a rewind; `unknown` is the honest cache state after one.
|
|
203
|
+
- `/router:status` and the status line show the reason; the report adds the
|
|
204
|
+
estimate. The dollars are list prices, a list-price equivalent for plan
|
|
205
|
+
models.
|
|
206
|
+
- Shadow economics in `decisions.jsonl`, not read by the policy: for a turn
|
|
207
|
+
where Jev's choice differs from the current route, the extra cost of the
|
|
208
|
+
next turn, the difference for each later turn (input and output), and the
|
|
209
|
+
turns until a cheaper route repays its cache write. The owner's first
|
|
210
|
+
request (2026-09-22) was to stay while the model works on a warm cache and
|
|
211
|
+
to step down once the thinking is done. Rule 2 covers the first half; the
|
|
212
|
+
shadow estimate measures the second before any rule acts on it.
|
|
213
|
+
- The default prices match `test/fixtures/list-prices.json`, which names its
|
|
214
|
+
source and date. A test pins how a tenfold cache-read error changes one
|
|
215
|
+
decision: the bar moves within `upgradeBase` and `upgradeBase +
|
|
216
|
+
upgradeSlope`, and a confident jump ignores it. That error happened once
|
|
217
|
+
(a838f7c).
|
|
218
|
+
|
|
219
|
+
Not taken: the agent type as a routing signal; a payback check on downgrades
|
|
220
|
+
before shadow data; removing the switching tax from the upgrade bar before a
|
|
221
|
+
replay of the logs. The Pi router rejects the tax-to-confidence formula
|
|
222
|
+
because it mixes dollars with an uncalibrated probability; the replay decides.
|
|
223
|
+
|
|
224
|
+
## Real-world evaluation (2026-09-23)
|
|
225
|
+
|
|
226
|
+
A day of dogfooding this repository on the installed plugin: 31 Claude Code
|
|
227
|
+
sessions, 1,578 routed requests, one machine. `decisions.jsonl` holds the
|
|
228
|
+
tier, the reason and the token counts for every request — no prompt text.
|
|
229
|
+
|
|
230
|
+

|
|
231
|
+
|
|
232
|
+
82% of turns never needed more than Sonnet, 10% stayed on Haiku, and 8% needed
|
|
233
|
+
Opus. `medium` (Opus at `high` effort) fired once: a confident vote jumps two
|
|
234
|
+
tiers straight to `high` instead of stopping at `medium` (switching policy,
|
|
235
|
+
rule 4).
|
|
236
|
+
|
|
237
|
+
Repricing that same traffic — same tokens, same observed cache reads — at
|
|
238
|
+
Opus's rates puts the input-token bill 18.8% above what the router actually
|
|
239
|
+
spent. That number covers input tokens only: the price table has no output
|
|
240
|
+
price (see [Cache and cost inputs](#cache-and-cost-inputs)), so it cannot say
|
|
241
|
+
how much of the real saving is left out — likely more, since Haiku and Sonnet
|
|
242
|
+
also bill less per output token than Opus. Answer quality isn't measured
|
|
243
|
+
here either.
|
|
244
|
+
|
|
245
|
+
One developer, one day: a dogfood snapshot, not a benchmark.
|
|
196
246
|
|
|
197
247
|
## Layout
|
|
198
248
|
|
|
@@ -209,7 +259,7 @@ Agreed with Codex on 2026-09-22. The thresholds are start values.
|
|
|
209
259
|
lib/config.mjs defaults, user file, validation
|
|
210
260
|
lib/facts.mjs request body and memory -> facts (pure)
|
|
211
261
|
lib/jev.mjs request, injected transport, parse
|
|
212
|
-
lib/cost.mjs warmth, input cost, switching tax
|
|
262
|
+
lib/cost.mjs cache key, warmth, input cost, switching tax, shadow economics
|
|
213
263
|
lib/policy.mjs switching policy v0
|
|
214
264
|
lib/rewrite.mjs model, effort, thinking per model family
|
|
215
265
|
lib/sse.mjs usage reader for SSE and JSON bodies
|
|
@@ -232,8 +282,16 @@ Agreed with Codex on 2026-09-22. The thresholds are start values.
|
|
|
232
282
|
- The gateway reads the configuration once. A reload without a restart is not
|
|
233
283
|
implemented.
|
|
234
284
|
- Without `CLAUDE_CODE_GATEWAY_HINT_HEADERS=1`, the gateway guesses side
|
|
235
|
-
requests
|
|
285
|
+
requests from the body; the guesses miss some. A missed side request with
|
|
286
|
+
a short history also counts as a history break.
|
|
236
287
|
- `x-claude-code-agent-type` is logged, not used: no policy rule reads it yet.
|
|
288
|
+
- Does a downgrade that the shadow estimate says never repays deserve a rule,
|
|
289
|
+
and does the switching tax belong in the upgrade bar? A replay of
|
|
290
|
+
`decisions.jsonl` against the `shadow` and `observed` lines decides.
|
|
291
|
+
- Effort changes how much a model writes. The shadow estimate uses the last
|
|
292
|
+
output size for both routes.
|
|
293
|
+
- Whether tools and system survive an effort change is model-specific; the
|
|
294
|
+
gateway counts the whole prefix as lost, an upper bound.
|
|
237
295
|
|
|
238
296
|
## Release
|
|
239
297
|
|
package/docs/configuration.md
CHANGED
|
@@ -77,7 +77,7 @@ If you agree, it also wraps the status line command:
|
|
|
77
77
|
```
|
|
78
78
|
|
|
79
79
|
The wrapper runs the command after it, then adds one line for a routed
|
|
80
|
-
session, for example `jev-router ▸ opus-5-5 · xhigh`. Without a command
|
|
80
|
+
session, for example `jev-router ▸ opus-5-5 · xhigh · upgrade`. Without a command
|
|
81
81
|
after it, it prints only that line. The path contains the plugin version, so
|
|
82
82
|
run `/router:setup` again after a plugin update.
|
|
83
83
|
|
|
@@ -88,12 +88,16 @@ hint headers:
|
|
|
88
88
|
get routing. `auxiliary` and `compaction` are side requests and get
|
|
89
89
|
`gateway.auxiliaryTier`.
|
|
90
90
|
- `x-claude-code-context-compacted` on the first request after a compaction.
|
|
91
|
-
The gateway then drops the cached prefixes of every model
|
|
91
|
+
The gateway then drops the cached prefixes of every model, the pending
|
|
92
|
+
votes and the escalation hold.
|
|
92
93
|
- `x-claude-code-agent-type`, for example `Explore` or `Plan`. It goes to
|
|
93
94
|
`decisions.jsonl` only.
|
|
94
95
|
|
|
95
|
-
Without the headers, the gateway identifies side requests by their shape
|
|
96
|
-
|
|
96
|
+
Without the headers, the gateway identifies side requests by their shape. A
|
|
97
|
+
main request with fewer messages than the last one is a compaction or a
|
|
98
|
+
rewind, with or without the headers: the gateway drops the same state. A
|
|
99
|
+
context that shrank by more than 20% with no fewer messages is context
|
|
100
|
+
editing; it drops only the cached prefixes.
|
|
97
101
|
|
|
98
102
|
A subagent with `model: inherit` sends `x-claude-code-agent-id` even without
|
|
99
103
|
the hint headers. Each subagent keeps its own routing memory, so its turns do
|
|
@@ -123,6 +127,7 @@ path. Nested objects merge.
|
|
|
123
127
|
"sonnet": {
|
|
124
128
|
"id": "claude-sonnet-5",
|
|
125
129
|
"input": 2,
|
|
130
|
+
"output": 10,
|
|
126
131
|
"cacheRead": 0.2,
|
|
127
132
|
"contextWindow": 1000000,
|
|
128
133
|
"billing": "plan",
|
|
@@ -156,14 +161,18 @@ frontmatter of `skills/<tier>/SKILL.md`. A test makes sure that they agree.
|
|
|
156
161
|
|
|
157
162
|
### models
|
|
158
163
|
|
|
159
|
-
| Alias | ID | Input | Cache Read | Window | Max output | Billing | Efforts |
|
|
160
|
-
| -------- | ------------------ | ----- | ---------- | ------ | ---------- | ------- | ------------- |
|
|
161
|
-
| `opus` | `claude-opus-5-5` | $4 | $0.2 | 1M | as sent | plan | all |
|
|
162
|
-
| `sonnet` | `claude-sonnet-5` | $2 | $0.2 | 1M | as sent | plan | low–xhigh–max |
|
|
163
|
-
| `haiku` | `claude-haiku-4-5` | $1 | $0.1 | 200k | 64k | plan | none |
|
|
164
|
-
|
|
165
|
-
`id` is the model id that the gateway sends to Anthropic. `input`
|
|
166
|
-
`cacheRead` are list prices in USD per million tokens. `
|
|
164
|
+
| Alias | ID | Input | Output | Cache Read | Window | Max output | Billing | Efforts |
|
|
165
|
+
| -------- | ------------------ | ----- | ------ | ---------- | ------ | ---------- | ------- | ------------- |
|
|
166
|
+
| `opus` | `claude-opus-5-5` | $4 | $20 | $0.2 | 1M | as sent | plan | all |
|
|
167
|
+
| `sonnet` | `claude-sonnet-5` | $2 | $10 | $0.2 | 1M | as sent | plan | low–xhigh–max |
|
|
168
|
+
| `haiku` | `claude-haiku-4-5` | $1 | $5 | $0.1 | 200k | 64k | plan | none |
|
|
169
|
+
|
|
170
|
+
`id` is the model id that the gateway sends to Anthropic. `input`, `output`
|
|
171
|
+
and `cacheRead` are list prices in USD per million tokens. `output` is
|
|
172
|
+
optional and feeds only the `shadow` estimate in `decisions.jsonl`; the policy
|
|
173
|
+
does not read it. The defaults match `test/fixtures/list-prices.json`, which
|
|
174
|
+
names its source and the date it was checked; a test fails when the two
|
|
175
|
+
differ. `contextWindow` is the
|
|
167
176
|
size of the context window in tokens. `maxOutput`, when set, caps the
|
|
168
177
|
`max_tokens` that Claude Code sends; the API rejects a request above the
|
|
169
178
|
model's output limit. `billing` is `plan` for models that use
|
|
@@ -180,12 +189,12 @@ effort and thinking from the request.
|
|
|
180
189
|
| `gateway.auxiliaryTier` | The tier for side requests, for example session titles. |
|
|
181
190
|
| `gateway.idleShutdownMs` | The gateway exits after this long without requests, when no turn waits for a tool result. Two hours by default; `0` keeps it running. |
|
|
182
191
|
| `upgradeVotes` | The number of consecutive votes above the current tier before an upgrade of one tier. |
|
|
183
|
-
| `upgradeBase`, `upgradeSlope`, `upgradePivotUsd` | The required probability mass: `base + slope * tax / (tax + pivot)`. The `tax` is the extra input cost to read the context on the new model. |
|
|
192
|
+
| `upgradeBase`, `upgradeSlope`, `upgradePivotUsd` | The required probability mass: `base + slope * tax / (tax + pivot)`. The `tax` is the extra input cost to read the context on the new route; the cache is per model and effort. |
|
|
184
193
|
| `jumpConfidence` | The mass that lets a jump of two tiers skip the vote delay. |
|
|
185
194
|
| `downgradeVotes`, `downgradeMass` | The number of consecutive votes, and the mass at or below the candidate, for a downgrade. |
|
|
186
195
|
| `continuationMass` | The Jev probability for "this prompt continues the task" that keeps the current route. |
|
|
187
196
|
| `escalationHoldTurns` | The number of turns to hold one tier up after two failed repairs of the same error. |
|
|
188
|
-
| `cashCapUsd` | The cold cache
|
|
197
|
+
| `cashCapUsd` | The cold-write guard: the estimated first cache write above which the gateway refuses an automatic route to a `credits` model whose cache is not warm. Not a budget for the turn: output is not counted. |
|
|
189
198
|
|
|
190
199
|
## Environment variables
|
|
191
200
|
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 720 500" font-family="system-ui, -apple-system, 'Segoe UI', sans-serif">
|
|
2
|
+
<rect width="720" height="500" fill="#fcfcfb"/>
|
|
3
|
+
<text x="24" y="34" font-size="18" font-weight="700" fill="#0b0b0b">Where the router sent the calls</text>
|
|
4
|
+
<text x="24" y="56" font-size="13" fill="#898781">one developer, 31 sessions, one day of dogfooding (2026-09-23) — 1578 routed requests</text>
|
|
5
|
+
|
|
6
|
+
<text x="24" y="74" font-size="12" font-weight="600" fill="#52514e">SHARE OF REQUESTS BY TIER</text>
|
|
7
|
+
|
|
8
|
+
<text x="196" y="111" text-anchor="end" font-size="13" fill="#52514e">micro · Haiku</text>
|
|
9
|
+
<rect x="210" y="96" width="50.07299270072993" height="22" rx="4" fill="#86b6ef"/>
|
|
10
|
+
<text x="268.0729927007299" y="111"
|
|
11
|
+
text-anchor="start" font-size="13" font-weight="600"
|
|
12
|
+
fill="#0b0b0b">9.8%</text>
|
|
13
|
+
|
|
14
|
+
<text x="196" y="157" text-anchor="end" font-size="13" fill="#52514e">low · Sonnet</text>
|
|
15
|
+
<rect x="210" y="142" width="420" height="22" rx="4" fill="#3987e5"/>
|
|
16
|
+
<text x="620" y="157"
|
|
17
|
+
text-anchor="end" font-size="13" font-weight="600"
|
|
18
|
+
fill="#ffffff">82.2%</text>
|
|
19
|
+
|
|
20
|
+
<text x="196" y="203" text-anchor="end" font-size="13" fill="#52514e">medium · Opus (high)</text>
|
|
21
|
+
<rect x="210" y="188" width="3" height="22" rx="4" fill="#1c5cab"/>
|
|
22
|
+
<text x="221" y="203"
|
|
23
|
+
text-anchor="start" font-size="13" font-weight="600"
|
|
24
|
+
fill="#0b0b0b">0.1%</text>
|
|
25
|
+
|
|
26
|
+
<text x="196" y="249" text-anchor="end" font-size="13" fill="#52514e">high · Opus (xhigh)</text>
|
|
27
|
+
<rect x="210" y="234" width="40.87591240875912" height="22" rx="4" fill="#0d366b"/>
|
|
28
|
+
<text x="258.8759124087591" y="249"
|
|
29
|
+
text-anchor="start" font-size="13" font-weight="600"
|
|
30
|
+
fill="#0b0b0b">8%</text>
|
|
31
|
+
|
|
32
|
+
<line x1="24" y1="300" x2="696" y2="300" stroke="#e1e0d9" stroke-width="1"/>
|
|
33
|
+
<text x="24" y="320" font-size="12" font-weight="600" fill="#52514e">INPUT-TOKEN COST, LIST PRICE — SAME TRAFFIC, REPRICED AT OPUS RATES</text>
|
|
34
|
+
|
|
35
|
+
<text x="196" y="355" text-anchor="end" font-size="13" fill="#52514e">router (actual)</text>
|
|
36
|
+
<rect x="210" y="340" width="341.043219076006" height="22" rx="4" fill="#2a78d6"/>
|
|
37
|
+
<text x="559.0432190760059" y="355" font-size="13" font-weight="600" fill="#0b0b0b">$114.42</text>
|
|
38
|
+
|
|
39
|
+
<text x="196" y="401" text-anchor="end" font-size="13" fill="#52514e">same traffic, Opus prices</text>
|
|
40
|
+
<rect x="210" y="386" width="420" height="22" rx="4" fill="#c3c2b7"/>
|
|
41
|
+
<text x="638" y="401" font-size="13" font-weight="600" fill="#0b0b0b">$140.91</text>
|
|
42
|
+
<text x="24" y="448" font-size="12" fill="#898781">−18.8% on input tokens alone. Output-token price isn't in the model, so real savings are likely larger, not smaller.</text>
|
|
43
|
+
<text x="24" y="468" font-size="12" fill="#898781">Not measured: answer quality. One machine, one day — a dogfood snapshot, not a benchmark.</text>
|
|
44
|
+
</svg>
|
package/docs/user-guide.md
CHANGED
|
@@ -29,13 +29,20 @@ ANTHROPIC_BASE_URL=http://127.0.0.1:43170 claude --plugin-dir . --model jev-rout
|
|
|
29
29
|
route stays there for two turns. A repair is an edit between the two errors.
|
|
30
30
|
- An upgrade needs two consecutive votes for a higher tier. A jump of two
|
|
31
31
|
tiers with high confidence happens at once. The required confidence goes up
|
|
32
|
-
with the cost to read the context again on the new model.
|
|
32
|
+
with the cost to read the context again on the new model. Opus at `high` and
|
|
33
|
+
Opus at `xhigh` are two caches: a change of effort rewrites the cached
|
|
34
|
+
conversation, so `medium` to `high` is not free.
|
|
33
35
|
- A downgrade needs two consecutive confident votes.
|
|
34
|
-
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
36
|
+
- When the conversation gets shorter, after a compaction or a rewind, the
|
|
37
|
+
gateway forgets the cached prefixes, the pending votes and the escalation
|
|
38
|
+
hold. They were about turns that are no longer in the conversation.
|
|
39
|
+
- The cold-write guard: an automatic switch to a model that bills usage
|
|
40
|
+
credits is refused when that model's cache is not warm and the first cache
|
|
41
|
+
write would cost more than `policy.cashCapUsd`. Then the strongest plan tier
|
|
42
|
+
serves the turn. The guard limits that one estimated write, not the cost of
|
|
43
|
+
the turn: output is not counted. Behind the gateway, Claude Code does not
|
|
44
|
+
show its consent prompt for these credits. No default model bills credits;
|
|
45
|
+
this applies once you add one in `router.json`.
|
|
39
46
|
- When Jev fails or times out, or when there is no key, the baseline tier
|
|
40
47
|
serves the turn. After three failures in a row, new turns skip Jev for one
|
|
41
48
|
minute and then try it once, so an outage costs one slow turn a minute, not
|
|
@@ -64,12 +71,19 @@ The gateway sends the real model id unchanged.
|
|
|
64
71
|
## See the current route
|
|
65
72
|
|
|
66
73
|
`/router:status` shows the gateway, the routes, and the model, effort and
|
|
67
|
-
reason of the last turn in this session.
|
|
74
|
+
reason of the last turn in this session. It also says why in words, and for
|
|
75
|
+
a vote it shows the estimate: the probability mass, the bar it had to reach,
|
|
76
|
+
the switching tax and whether each cache was `warm`, `expired` or `unknown`.
|
|
77
|
+
`unknown` means no response for that cache since the session started or the
|
|
78
|
+
conversation got shorter. The dollars are list prices. For plan models they
|
|
79
|
+
are a list-price equivalent, not a charge.
|
|
68
80
|
|
|
69
81
|
The status line wrapper from `/router:setup` adds one line to your status line
|
|
70
82
|
while the session uses the router:
|
|
71
83
|
|
|
72
|
-
- `jev-router ▸ opus-5-5 · xhigh`: the model
|
|
84
|
+
- `jev-router ▸ opus-5-5 · xhigh · upgrade`: the model, the effort and the
|
|
85
|
+
reason of the last turn. `upgrade-pending` means that Jev asked for a higher
|
|
86
|
+
tier and the policy held the route back.
|
|
73
87
|
- `router: no turn yet`: the session has no routed turn.
|
|
74
88
|
- `router: gateway off, the next prompt starts it`: the gateway stopped after
|
|
75
89
|
idle time, or it crashed. Either way the next prompt starts it.
|
|
@@ -84,7 +98,7 @@ the plugin data directory. The directory is
|
|
|
84
98
|
by hand, the directory is `$TMPDIR/router/`.
|
|
85
99
|
|
|
86
100
|
```sh
|
|
87
|
-
tail -n 20 ~/.claude/plugins/data/router-*/decisions.jsonl | jq -c '{tier, reason, estimate, observed}'
|
|
101
|
+
tail -n 20 ~/.claude/plugins/data/router-*/decisions.jsonl | jq -c '{tier, reason, estimate, shadow, observed}'
|
|
88
102
|
```
|
|
89
103
|
|
|
90
104
|
`tier` is the selected tier. `reason` is the rule that decided. The `observed`
|
|
@@ -92,6 +106,12 @@ lines carry the model that answered, the context tokens and the cache reads
|
|
|
92
106
|
from the response. Compare the estimated and the observed cache reads to tune
|
|
93
107
|
the thresholds.
|
|
94
108
|
|
|
109
|
+
`shadow` is what following Jev's choice instead of the current route would
|
|
110
|
+
cost, at list prices: the next turn, each later turn, and the number of turns
|
|
111
|
+
until a cheaper route repays its cache write. The policy does not read it.
|
|
112
|
+
`historyBreak` lines mark a compaction or a rewind; `cacheReset` lines mark a
|
|
113
|
+
context that shrank by more than a fifth.
|
|
114
|
+
|
|
95
115
|
`scripts/transcript-models.sh <transcript.jsonl>` shows the model for each
|
|
96
116
|
assistant message in a Claude Code transcript.
|
|
97
117
|
|
|
@@ -115,3 +135,6 @@ assistant message in a Claude Code transcript.
|
|
|
115
135
|
- Rate limits and overload errors from Anthropic (429, 529) reach Claude Code
|
|
116
136
|
unchanged. Claude Code waits and retries; the gateway does not add a second
|
|
117
137
|
layer of retries.
|
|
138
|
+
- Updating from 0.3.0 or earlier: those gateways cannot hand over to a newer
|
|
139
|
+
one. Stop the old one once with `pkill -f scripts/gateway.mjs`; the next
|
|
140
|
+
prompt starts the new version.
|
package/lib/config.mjs
CHANGED
|
@@ -20,12 +20,14 @@ export const DEFAULTS = {
|
|
|
20
20
|
micro: { model: 'haiku' },
|
|
21
21
|
},
|
|
22
22
|
// `id` is sent upstream verbatim. List prices in USD per million tokens; `cacheRead` is absolute,
|
|
23
|
-
// not a multiplier (Opus 5.5 reads at 0.05x input, the rest at the standard 0.1x
|
|
23
|
+
// not a multiplier (Opus 5.5 reads at 0.05x input, the rest at the standard 0.1x). A price change must also change
|
|
24
|
+
// test/fixtures/list-prices.json, with its source and date. `output` feeds only the shadow estimate in the log.
|
|
24
25
|
// `efforts` lists what the model accepts; an empty list means no effort field and no adaptive thinking.
|
|
25
26
|
models: {
|
|
26
27
|
opus: {
|
|
27
28
|
id: 'claude-opus-5-5',
|
|
28
29
|
input: 4,
|
|
30
|
+
output: 20,
|
|
29
31
|
cacheRead: 0.2,
|
|
30
32
|
contextWindow: 1_000_000,
|
|
31
33
|
billing: 'plan',
|
|
@@ -34,6 +36,7 @@ export const DEFAULTS = {
|
|
|
34
36
|
sonnet: {
|
|
35
37
|
id: 'claude-sonnet-5',
|
|
36
38
|
input: 2,
|
|
39
|
+
output: 10,
|
|
37
40
|
cacheRead: 0.2,
|
|
38
41
|
contextWindow: 1_000_000,
|
|
39
42
|
billing: 'plan',
|
|
@@ -43,6 +46,7 @@ export const DEFAULTS = {
|
|
|
43
46
|
haiku: {
|
|
44
47
|
id: 'claude-haiku-4-5',
|
|
45
48
|
input: 1,
|
|
49
|
+
output: 5,
|
|
46
50
|
cacheRead: 0.1,
|
|
47
51
|
contextWindow: 200_000,
|
|
48
52
|
maxOutput: 64_000,
|
|
@@ -116,6 +120,8 @@ function validate(config) {
|
|
|
116
120
|
if (!Number.isFinite(model[field]) || model[field] < 0)
|
|
117
121
|
throw new Error(`models.${alias}.${field} must be a non-negative number`);
|
|
118
122
|
}
|
|
123
|
+
if (model.output !== undefined && !(Number.isFinite(model.output) && model.output >= 0))
|
|
124
|
+
throw new Error(`models.${alias}.output must be a non-negative number`);
|
|
119
125
|
if (typeof model.id !== 'string' || !model.id) throw new Error(`models.${alias}.id is required`);
|
|
120
126
|
if (model.maxOutput !== undefined && !(Number.isInteger(model.maxOutput) && model.maxOutput > 0))
|
|
121
127
|
throw new Error(`models.${alias}.maxOutput must be a positive integer`);
|
package/lib/cost.mjs
CHANGED
|
@@ -1,22 +1,46 @@
|
|
|
1
|
-
// Cache-aware
|
|
1
|
+
// Cache-aware cost of the next request. Pure arithmetic over configured prices and the gateway's memory.
|
|
2
|
+
import { clampEffort } from './rewrite.mjs';
|
|
3
|
+
|
|
4
|
+
// The prompt cache belongs to a model, and its messages part to the effort too: a top-level effort change rewrites
|
|
5
|
+
// the messages cache. Opus at `high` and Opus at `xhigh` are two caches, not one.
|
|
6
|
+
export function cacheKey(modelId, effort) {
|
|
7
|
+
return effort ? `${modelId}@${effort}` : modelId;
|
|
8
|
+
}
|
|
9
|
+
|
|
10
|
+
// The cache a request routed to `tier` uses. `sentEffort` is what Claude Code sent, for a route that keeps it.
|
|
11
|
+
export function routeCacheKey(config, tier, sentEffort) {
|
|
12
|
+
const model = modelOf(config, tier);
|
|
13
|
+
return cacheKey(model.id, clampEffort(config.routes[tier].effort ?? sentEffort, model.efforts));
|
|
14
|
+
}
|
|
2
15
|
|
|
3
16
|
export function isWarm(modelState, now, cache) {
|
|
4
17
|
if (!modelState) return false;
|
|
5
18
|
return now < modelState.lastAt + cache.ttlMs[modelState.ttl] - cache.warmMarginMs;
|
|
6
19
|
}
|
|
7
20
|
|
|
21
|
+
// For the logs: `unknown` means no response for this cache since the session started or its history broke. The cost
|
|
22
|
+
// arithmetic treats `unknown` and `expired` alike, as a full write.
|
|
23
|
+
export function cacheState(modelState, now, cache) {
|
|
24
|
+
if (!modelState) return 'unknown';
|
|
25
|
+
return isWarm(modelState, now, cache) ? 'warm' : 'expired';
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
function modelOf(config, tier) {
|
|
29
|
+
return config.models[config.routes[tier].model];
|
|
30
|
+
}
|
|
31
|
+
|
|
8
32
|
function ttlFor(config, alias, facts) {
|
|
9
33
|
if (config.models[alias].billing === 'credits') return '5m';
|
|
10
34
|
return facts.lastRequest?.ttl ?? '5m';
|
|
11
35
|
}
|
|
12
36
|
|
|
13
|
-
// USD for the input side of one request
|
|
14
|
-
export function inputCostUsd(config,
|
|
37
|
+
// USD for the input side of one request routed to `tier` with `tokens` of context.
|
|
38
|
+
export function inputCostUsd(config, tier, tokens, facts, now) {
|
|
39
|
+
const alias = config.routes[tier].model;
|
|
15
40
|
const model = config.models[alias];
|
|
16
|
-
const state = facts.models[
|
|
41
|
+
const state = facts.models[routeCacheKey(config, tier, facts.effort)];
|
|
17
42
|
const reusable = isWarm(state, now, config.cache) ? Math.min(state.prefixTokens, tokens) : 0;
|
|
18
|
-
const
|
|
19
|
-
const write = model.input * config.cache.writeMultiplier[ttl];
|
|
43
|
+
const write = model.input * config.cache.writeMultiplier[ttlFor(config, alias, facts)];
|
|
20
44
|
return (model.cacheRead * reusable + write * (tokens - reusable)) / 1e6;
|
|
21
45
|
}
|
|
22
46
|
|
|
@@ -31,9 +55,30 @@ export function nextContextTokens(facts) {
|
|
|
31
55
|
return last ? last.tokens + last.outputTokens : 0;
|
|
32
56
|
}
|
|
33
57
|
|
|
34
|
-
export function switchingTaxUsd(config,
|
|
58
|
+
export function switchingTaxUsd(config, candidateTier, incumbentTier, facts, now) {
|
|
35
59
|
const tokens = nextContextTokens(facts);
|
|
36
60
|
return (
|
|
37
|
-
inputCostUsd(config,
|
|
61
|
+
inputCostUsd(config, candidateTier, tokens, facts, now) - inputCostUsd(config, incumbentTier, tokens, facts, now)
|
|
38
62
|
);
|
|
39
63
|
}
|
|
64
|
+
|
|
65
|
+
// Shadow estimate for the decision log; the policy does not read it. At list prices, so for `plan` models it is a
|
|
66
|
+
// list-price equivalent, not a charge. Output uses the last observed output size for both routes, although effort
|
|
67
|
+
// changes how much a model writes.
|
|
68
|
+
// - nextTurnUsd: candidate minus incumbent for the next request, input at the current cache state plus output.
|
|
69
|
+
// - laterTurnUsd: the same difference for each later turn, once both caches are warm.
|
|
70
|
+
// - paybackTurns: 0 when the switch is cheaper at once, n when later turns repay it after n turns, null when never.
|
|
71
|
+
export function shadowEconomics(config, candidateTier, incumbentTier, facts, now) {
|
|
72
|
+
const candidate = modelOf(config, candidateTier);
|
|
73
|
+
const incumbent = modelOf(config, incumbentTier);
|
|
74
|
+
if (candidate.output === undefined || incumbent.output === undefined) return null;
|
|
75
|
+
const tokens = nextContextTokens(facts);
|
|
76
|
+
const outputTokens = facts.lastRequest?.outputTokens ?? 0;
|
|
77
|
+
const outputUsd = ((candidate.output - incumbent.output) * outputTokens) / 1e6;
|
|
78
|
+
const nextTurnUsd = switchingTaxUsd(config, candidateTier, incumbentTier, facts, now) + outputUsd;
|
|
79
|
+
const laterTurnUsd = ((candidate.cacheRead - incumbent.cacheRead) * tokens) / 1e6 + outputUsd;
|
|
80
|
+
let paybackTurns = null;
|
|
81
|
+
if (nextTurnUsd <= 0 && laterTurnUsd <= 0) paybackTurns = 0;
|
|
82
|
+
else if (nextTurnUsd > 0 && laterTurnUsd < 0) paybackTurns = Math.ceil(nextTurnUsd / -laterTurnUsd);
|
|
83
|
+
return { nextTurnUsd, laterTurnUsd, paybackTurns, outputTokens };
|
|
84
|
+
}
|
package/lib/facts.mjs
CHANGED
|
@@ -32,6 +32,8 @@ export function factsFromRequest(body, memory, { recentTurns, maxTextChars }) {
|
|
|
32
32
|
// The same history length and the same last message: Claude Code resent the request (429, 529, a dropped stream).
|
|
33
33
|
turnKey: last?.role === 'user' ? `${messages.length}:${hash(JSON.stringify(last.content))}` : null,
|
|
34
34
|
failure: repeatedFailure(errors, edits, messages.length - 1),
|
|
35
|
+
// The effort Claude Code sent: a route without its own effort keeps it, and it is part of the cache key.
|
|
36
|
+
effort: body.output_config?.effort ?? null,
|
|
35
37
|
lastRoute: memory.lastRoute,
|
|
36
38
|
lastRequest: memory.lastRequest,
|
|
37
39
|
models: memory.models,
|
package/lib/gateway.mjs
CHANGED
|
@@ -113,7 +113,7 @@ export function createGateway({
|
|
|
113
113
|
if (reader.stopReason === 'tool_use') activity.turnPaused(session);
|
|
114
114
|
if (!routed || routed.auxiliary) return;
|
|
115
115
|
try {
|
|
116
|
-
router.recordResponse(session, routed.tier, usage);
|
|
116
|
+
router.recordResponse(session, routed.tier, usage, routed.effort);
|
|
117
117
|
} catch (recordError) {
|
|
118
118
|
onError(recordError);
|
|
119
119
|
}
|
package/lib/policy.mjs
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
// Switching policy v0 (docs/
|
|
1
|
+
// Switching policy v0 (docs/architecture.md): stickiness, escalation floor, cost-gated votes.
|
|
2
2
|
import { rank, TIERS } from './config.mjs';
|
|
3
|
-
import { coldWriteUsd, isWarm, nextContextTokens, switchingTaxUsd } from './cost.mjs';
|
|
3
|
+
import { cacheState, coldWriteUsd, isWarm, nextContextTokens, routeCacheKey, switchingTaxUsd } from './cost.mjs';
|
|
4
4
|
|
|
5
5
|
export function initialState() {
|
|
6
6
|
return { turn: 0, votes: [], holdUntilTurn: 0, escalatedSignature: null };
|
|
@@ -34,15 +34,13 @@ export function decide({ config, facts, advice, state, baseline, now }) {
|
|
|
34
34
|
if (rank(gated.fallback) <= rank(incumbent)) return stay('cash-gate', gated.estimate);
|
|
35
35
|
return result(gated.fallback, 'cash-gate', next, gated.estimate);
|
|
36
36
|
}
|
|
37
|
-
const tax = Math.max(
|
|
38
|
-
0,
|
|
39
|
-
switchingTaxUsd(config, config.routes[choice].model, config.routes[incumbent].model, facts, now),
|
|
40
|
-
);
|
|
37
|
+
const tax = Math.max(0, switchingTaxUsd(config, choice, incumbent, facts, now));
|
|
41
38
|
const threshold =
|
|
42
39
|
config.policy.upgradeBase + config.policy.upgradeSlope * (tax / (tax + config.policy.upgradePivotUsd));
|
|
43
40
|
const jump = rank(choice) - rank(incumbent) >= 2 && upgrade >= config.policy.jumpConfidence;
|
|
44
41
|
const streak = trailing(next.votes, (v) => rank(v.tier) > rank(incumbent));
|
|
45
|
-
const
|
|
42
|
+
const cache = { candidate: cacheOf(config, choice, facts, now), incumbent: cacheOf(config, incumbent, facts, now) };
|
|
43
|
+
const estimate = { taxUsd: tax, threshold, upgradeMass: upgrade, streak, cache };
|
|
46
44
|
if (jump || (streak >= config.policy.upgradeVotes && upgrade >= threshold))
|
|
47
45
|
return result(choice, jump ? 'jump' : 'upgrade', next, estimate);
|
|
48
46
|
return stay('upgrade-pending', estimate);
|
|
@@ -73,16 +71,22 @@ export function fitTier(config, tier, tokens) {
|
|
|
73
71
|
);
|
|
74
72
|
}
|
|
75
73
|
|
|
76
|
-
//
|
|
74
|
+
// Cold-write guard: an automatic route to a credits-billed model whose cache is not warm must not start with a cache
|
|
75
|
+
// write above `cashCapUsd`. It bounds that one estimated write, not the spend of the turn: a warm cache passes, and
|
|
76
|
+
// output is not counted.
|
|
77
77
|
function cashGate(config, tier, facts, now) {
|
|
78
78
|
const alias = config.routes[tier].model;
|
|
79
|
-
|
|
80
|
-
if (
|
|
81
|
-
if (isWarm(facts.models[model.id], now, config.cache)) return { blocked: false };
|
|
79
|
+
if (config.models[alias].billing !== 'credits') return { blocked: false };
|
|
80
|
+
if (isWarm(facts.models[routeCacheKey(config, tier, facts.effort)], now, config.cache)) return { blocked: false };
|
|
82
81
|
const cold = coldWriteUsd(config, alias, nextContextTokens(facts), facts);
|
|
83
82
|
if (cold <= config.policy.cashCapUsd) return { blocked: false };
|
|
84
83
|
const fallback = [...TIERS].reverse().find((t) => config.models[config.routes[t].model].billing === 'plan');
|
|
85
|
-
|
|
84
|
+
const estimate = { coldUsd: cold, cap: config.policy.cashCapUsd, cache: cacheOf(config, tier, facts, now) };
|
|
85
|
+
return { blocked: true, fallback, estimate };
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
function cacheOf(config, tier, facts, now) {
|
|
89
|
+
return cacheState(facts.models[routeCacheKey(config, tier, facts.effort)], now, config.cache);
|
|
86
90
|
}
|
|
87
91
|
|
|
88
92
|
function trailing(votes, predicate) {
|
package/lib/router.mjs
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
// Per-request orchestration: facts -> advice -> policy -> rewritten body. Knows nothing about HTTP.
|
|
2
|
-
import { LEGACY_ALIAS } from './config.mjs';
|
|
3
|
-
import { nextContextTokens } from './cost.mjs';
|
|
2
|
+
import { LEGACY_ALIAS, TIERS } from './config.mjs';
|
|
3
|
+
import { cacheKey, nextContextTokens, shadowEconomics } from './cost.mjs';
|
|
4
4
|
import { factsFromRequest } from './facts.mjs';
|
|
5
5
|
import { askJev } from './jev.mjs';
|
|
6
6
|
import { decide, fitTier, initialState } from './policy.mjs';
|
|
@@ -20,14 +20,23 @@ export function emptyMemory() {
|
|
|
20
20
|
return {
|
|
21
21
|
lastRoute: null,
|
|
22
22
|
lastReason: null,
|
|
23
|
+
lastEstimate: null,
|
|
23
24
|
lastEffort: null,
|
|
24
25
|
lastRequest: null,
|
|
25
26
|
lastTurnKey: null,
|
|
27
|
+
lastMessageCount: null,
|
|
26
28
|
models: {},
|
|
27
29
|
state: null,
|
|
28
30
|
};
|
|
29
31
|
}
|
|
30
32
|
|
|
33
|
+
// The history the gateway saw is gone: a compaction or a rewind. The cached prefixes are unknown, and the votes and
|
|
34
|
+
// the escalation hold were about turns that are no longer in it. The route stays until the next decision.
|
|
35
|
+
function restartHistory(memory) {
|
|
36
|
+
memory.models = {};
|
|
37
|
+
if (memory.state) memory.state = { ...memory.state, votes: [], holdUntilTurn: 0, escalatedSignature: null };
|
|
38
|
+
}
|
|
39
|
+
|
|
31
40
|
export class Router {
|
|
32
41
|
constructor({ config, fetchFn, dataDir, now = Date.now, onError = () => {} }) {
|
|
33
42
|
this.config = config;
|
|
@@ -49,10 +58,16 @@ export class Router {
|
|
|
49
58
|
// The hints come from Claude Code's gateway headers (CLAUDE_CODE_GATEWAY_HINT_HEADERS=1); each may be missing.
|
|
50
59
|
async route(body, { sessionId, requestClass = null, agentType = null, contextCompacted = null }) {
|
|
51
60
|
const memory = this.memory(sessionId);
|
|
52
|
-
// The compaction rewrote the conversation: no model has its prefix cached any more.
|
|
53
|
-
if (contextCompacted) memory.models = {};
|
|
54
|
-
const facts = factsFromRequest(body, memory, this.config.context);
|
|
55
61
|
const auxiliary = requestClass ? SIDE_REQUEST_CLASSES.has(requestClass) : isAuxiliaryShape(body);
|
|
62
|
+
// Messages only grow within one history. Fewer than the last main request: Claude Code compacted the conversation
|
|
63
|
+
// or the user rewound it. Context editing clears tool results but keeps the messages, so it is not a break.
|
|
64
|
+
const messageCount = Array.isArray(body.messages) ? body.messages.length : 0;
|
|
65
|
+
const rewound = !auxiliary && memory.lastMessageCount !== null && messageCount < memory.lastMessageCount;
|
|
66
|
+
if (contextCompacted || rewound) {
|
|
67
|
+
restartHistory(memory);
|
|
68
|
+
this.log({ session: sessionId, historyBreak: contextCompacted ? 'compaction' : 'shorter-history' });
|
|
69
|
+
}
|
|
70
|
+
const facts = factsFromRequest(body, memory, this.config.context);
|
|
56
71
|
let decision;
|
|
57
72
|
if (auxiliary) decision = { tier: this.config.gateway.auxiliaryTier, reason: 'auxiliary', state: memory.state };
|
|
58
73
|
else if (facts.continuation && memory.lastRoute)
|
|
@@ -66,11 +81,16 @@ export class Router {
|
|
|
66
81
|
if (tier !== decision.tier) decision = { ...decision, tier, reason: 'context-fit' };
|
|
67
82
|
}
|
|
68
83
|
const rewritten = rewriteRequest(body, decision.tier, this.config);
|
|
84
|
+
const effort = rewritten.output_config?.effort ?? null;
|
|
69
85
|
if (!auxiliary) {
|
|
70
86
|
memory.lastRoute = decision.tier;
|
|
71
|
-
memory.lastEffort =
|
|
87
|
+
memory.lastEffort = effort;
|
|
72
88
|
memory.lastTurnKey = facts.turnKey;
|
|
73
|
-
|
|
89
|
+
memory.lastMessageCount = messageCount;
|
|
90
|
+
if (!REUSED_ROUTE.has(decision.reason)) {
|
|
91
|
+
memory.lastReason = decision.reason;
|
|
92
|
+
memory.lastEstimate = decision.estimate ?? null;
|
|
93
|
+
}
|
|
74
94
|
memory.state = decision.state;
|
|
75
95
|
this.persist(sessionId, memory);
|
|
76
96
|
}
|
|
@@ -81,6 +101,7 @@ export class Router {
|
|
|
81
101
|
tier: decision.tier,
|
|
82
102
|
reason: decision.reason,
|
|
83
103
|
estimate: decision.estimate ?? null,
|
|
104
|
+
shadow: decision.shadow ?? null,
|
|
84
105
|
advice: decision.advice ?? null,
|
|
85
106
|
adviceError: decision.adviceError ?? null,
|
|
86
107
|
contextTokens: memory.lastRequest?.tokens ?? 0,
|
|
@@ -88,6 +109,7 @@ export class Router {
|
|
|
88
109
|
return {
|
|
89
110
|
body: rewritten,
|
|
90
111
|
tier: decision.tier,
|
|
112
|
+
effort,
|
|
91
113
|
reason: decision.reason,
|
|
92
114
|
auxiliary,
|
|
93
115
|
};
|
|
@@ -130,15 +152,15 @@ export class Router {
|
|
|
130
152
|
}
|
|
131
153
|
}
|
|
132
154
|
}
|
|
133
|
-
const
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
return { ...decision, advice, adviceError };
|
|
155
|
+
const now = this.now();
|
|
156
|
+
const decision = decide({ config: this.config, facts, advice, state, baseline: gateway.baselineTier, now });
|
|
157
|
+
// Shadow only: what following Jev's choice instead of staying would cost at list prices. The policy ignores it.
|
|
158
|
+
const incumbent = TIERS.includes(facts.lastRoute) ? facts.lastRoute : gateway.baselineTier;
|
|
159
|
+
const shadow =
|
|
160
|
+
TIERS.includes(advice?.choice) && advice.choice !== incumbent
|
|
161
|
+
? shadowEconomics(this.config, advice.choice, incumbent, facts, now)
|
|
162
|
+
: null;
|
|
163
|
+
return { ...decision, advice, adviceError, shadow };
|
|
142
164
|
}
|
|
143
165
|
|
|
144
166
|
jevSucceeded() {
|
|
@@ -160,15 +182,19 @@ export class Router {
|
|
|
160
182
|
);
|
|
161
183
|
}
|
|
162
184
|
|
|
163
|
-
// Called with the usage the gateway read from a forwarded main-conversation response
|
|
164
|
-
|
|
185
|
+
// Called with the usage the gateway read from a forwarded main-conversation response, and the effort the gateway
|
|
186
|
+
// sent with that request: the cache is keyed by both.
|
|
187
|
+
recordResponse(sessionId, tier, usage, effort = null) {
|
|
165
188
|
if (!usage) return;
|
|
166
189
|
const memory = this.memory(sessionId);
|
|
167
190
|
const modelId = usage.model ?? this.config.models[this.config.routes[tier].model].id;
|
|
168
191
|
const at = this.now();
|
|
169
|
-
// A context that shrank
|
|
170
|
-
// the
|
|
171
|
-
if (memory.lastRequest && usage.tokens < memory.lastRequest.tokens * COMPACTION_SHRINK)
|
|
192
|
+
// A context that shrank by more than a fifth: a compaction, or context editing that cleared old tool results.
|
|
193
|
+
// Either way the cached prefixes no longer match. The votes stay: a shorter history in messages resets them.
|
|
194
|
+
if (memory.lastRequest && usage.tokens < memory.lastRequest.tokens * COMPACTION_SHRINK) {
|
|
195
|
+
memory.models = {};
|
|
196
|
+
this.log({ session: sessionId, cacheReset: 'context-shrink' });
|
|
197
|
+
}
|
|
172
198
|
memory.lastRequest = {
|
|
173
199
|
model: modelId,
|
|
174
200
|
tokens: usage.tokens,
|
|
@@ -177,7 +203,11 @@ export class Router {
|
|
|
177
203
|
ttl: usage.ttl,
|
|
178
204
|
at,
|
|
179
205
|
};
|
|
180
|
-
memory.models[modelId] = {
|
|
206
|
+
memory.models[cacheKey(modelId, effort)] = {
|
|
207
|
+
lastAt: at,
|
|
208
|
+
prefixTokens: usage.tokens + usage.outputTokens,
|
|
209
|
+
ttl: usage.ttl,
|
|
210
|
+
};
|
|
181
211
|
this.persist(sessionId, memory);
|
|
182
212
|
this.log({ session: sessionId, observed: { ...usage, model: modelId, tier } });
|
|
183
213
|
}
|
package/lib/status.mjs
CHANGED
|
@@ -34,6 +34,7 @@ export function statusSnapshot(config, memory) {
|
|
|
34
34
|
? {
|
|
35
35
|
tier: memory.lastRoute,
|
|
36
36
|
reason: memory.lastReason ?? null,
|
|
37
|
+
estimate: memory.lastEstimate ?? null,
|
|
37
38
|
effort: memory.lastEffort ?? null,
|
|
38
39
|
model: memory.lastRequest?.model ?? null,
|
|
39
40
|
tokens: memory.lastRequest?.tokens ?? null,
|
|
@@ -51,14 +52,32 @@ function routeRow(config, tier) {
|
|
|
51
52
|
return { tier, model: model.id, effort };
|
|
52
53
|
}
|
|
53
54
|
|
|
54
|
-
//
|
|
55
|
+
// Why the route of the last turn is what it is, for /router:status.
|
|
56
|
+
const REASONS = {
|
|
57
|
+
upgrade: 'Jev voted above the current tier often enough, with enough mass for the switching tax',
|
|
58
|
+
jump: 'Jev was confident enough to skip a tier and the vote delay',
|
|
59
|
+
downgrade: 'Jev voted for a lower tier often enough, with enough mass',
|
|
60
|
+
'upgrade-pending': 'Jev asked for a higher tier; the route stays until the votes and the mass are enough',
|
|
61
|
+
'downgrade-pending': 'Jev asked for a lower tier; the route stays until the votes and the mass are enough',
|
|
62
|
+
'same-tier': 'Jev agreed with the current tier',
|
|
63
|
+
continuation: 'the prompt continues the task, so the route stays',
|
|
64
|
+
uncertain: 'Jev abstained, so the route stays',
|
|
65
|
+
'no-advice': 'no Jev answer (no key, a failure or a pause), so the route stays',
|
|
66
|
+
escalation: 'the same error came back after an edit: one tier up',
|
|
67
|
+
hold: 'the tier stays up for a few turns after an escalation',
|
|
68
|
+
'cash-gate':
|
|
69
|
+
'cold-write guard: the first cache write on a credits model would cost more than policy.cashCapUsd, so the strongest plan tier served',
|
|
70
|
+
'context-fit': "the chosen model's window does not hold the context",
|
|
71
|
+
forced: 'ROUTER_FORCE_TIER is set',
|
|
72
|
+
};
|
|
73
|
+
|
|
74
|
+
// One status-line segment, e.g. "router ▸ opus-5-5 · xhigh · upgrade".
|
|
55
75
|
export function statusSegment(status) {
|
|
56
76
|
if (!status) return 'router: gateway off, the next prompt starts it';
|
|
57
77
|
const last = status.session;
|
|
58
78
|
if (!last) return `${status.alias}: no turn yet`;
|
|
59
79
|
const model = shortModel(last.model ?? status.routes.find((r) => r.tier === last.tier)?.model);
|
|
60
|
-
|
|
61
|
-
return `${status.alias} ▸ ${model}${effort}`;
|
|
80
|
+
return [`${status.alias} ▸ ${model}`, last.effort, last.reason].filter(Boolean).join(' · ');
|
|
62
81
|
}
|
|
63
82
|
|
|
64
83
|
// Markdown for /router:status.
|
|
@@ -82,10 +101,32 @@ export function statusReport(status) {
|
|
|
82
101
|
const effort = last.effort ? ` at ${last.effort}` : '';
|
|
83
102
|
const context = last.tokens ? `, context ${last.tokens} tokens, cache reads ${last.cacheReadTokens}` : '';
|
|
84
103
|
lines.push(`Last turn: ${last.tier} → ${model}${effort}, reason ${last.reason ?? 'unknown'}${context}.`);
|
|
104
|
+
if (REASONS[last.reason]) lines.push(`Why: ${REASONS[last.reason]}.`);
|
|
105
|
+
const estimate = describeEstimate(last.estimate);
|
|
106
|
+
if (estimate) lines.push(`Estimate: ${estimate}.`);
|
|
85
107
|
}
|
|
86
108
|
return lines.join('\n');
|
|
87
109
|
}
|
|
88
110
|
|
|
111
|
+
// Dollars are list prices: for `plan` models a list-price equivalent, not a charge. `unknown` cache: no response for
|
|
112
|
+
// that cache since the session started or its history broke.
|
|
113
|
+
function describeEstimate(e) {
|
|
114
|
+
if (!e) return null;
|
|
115
|
+
const usd = (n) => `$${n.toFixed(2)}`;
|
|
116
|
+
const parts = [];
|
|
117
|
+
if (e.upgradeMass !== undefined)
|
|
118
|
+
parts.push(
|
|
119
|
+
`upgrade mass ${e.upgradeMass.toFixed(2)} against a bar of ${e.threshold.toFixed(2)}`,
|
|
120
|
+
`switching tax ${usd(e.taxUsd)} at list prices`,
|
|
121
|
+
);
|
|
122
|
+
if (e.downgradeMass !== undefined) parts.push(`downgrade mass ${e.downgradeMass.toFixed(2)}`);
|
|
123
|
+
if (e.coldUsd !== undefined) parts.push(`cold write ${usd(e.coldUsd)} against the cap of ${usd(e.cap)}`);
|
|
124
|
+
if (e.streak !== undefined) parts.push(`${e.streak} vote(s) in a row`);
|
|
125
|
+
if (e.cache?.candidate) parts.push(`cache: candidate ${e.cache.candidate}, current ${e.cache.incumbent}`);
|
|
126
|
+
else if (typeof e.cache === 'string') parts.push(`cache ${e.cache}`);
|
|
127
|
+
return parts.join(', ');
|
|
128
|
+
}
|
|
129
|
+
|
|
89
130
|
// null when the gateway does not answer in time.
|
|
90
131
|
export async function fetchStatus(port, sessionId) {
|
|
91
132
|
const query = sessionId ? `?session=${encodeURIComponent(sessionId)}` : '';
|
package/package.json
CHANGED