rightmodeler 0.3.0 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -40,6 +40,12 @@ function isObject(value) {
40
40
  return typeof value === "object" && value !== null && !Array.isArray(value);
41
41
  }
42
42
 
43
+ // The same 1e-12 snap as roundUsd in ../budget.ts, which this runtime ships without:
44
+ // lease sums then do not depend on the order requests settle in.
45
+ function roundUsd(value) {
46
+ return Number(value.toFixed(12));
47
+ }
48
+
43
49
  function parseConfig() {
44
50
  const rawSwapPolicy = jsonEnv("RM_SWAP_POLICY");
45
51
  if (
@@ -227,11 +233,11 @@ function loadState(spoolPath, checkpointPath) {
227
233
  state.lastAttemptGroup = Math.max(state.lastAttemptGroup, row.attemptGroup);
228
234
  state.groups.set(row.logicalCallId, row.attemptGroup);
229
235
  state.attemptIds.add(row.attemptId);
230
- state.spentUsd += row.costUsd;
236
+ state.spentUsd = roundUsd(state.spentUsd + row.costUsd);
231
237
  }
232
238
 
233
239
  for (const reservation of state.reservations.values()) {
234
- state.spentUsd += reservation.reservedUsd;
240
+ state.spentUsd = roundUsd(state.spentUsd + reservation.reservedUsd);
235
241
  }
236
242
 
237
243
  return state;
@@ -736,9 +742,10 @@ async function main() {
736
742
  const estimatedInputTokens = forwardedBody.length;
737
743
  const estimatedWorstCaseUsd =
738
744
  estimatedInputTokens * pricing.input + maxTokens * pricing.output;
739
- const requiredLeaseUsd =
740
- state.spentUsd + reservedUsd + estimatedWorstCaseUsd;
741
- if (requiredLeaseUsd > config.lease.maxUsd) {
745
+ const requiredLeaseUsd = roundUsd(
746
+ state.spentUsd + reservedUsd + estimatedWorstCaseUsd,
747
+ );
748
+ if (requiredLeaseUsd > roundUsd(config.lease.maxUsd)) {
742
749
  appendRow(spoolPath, {
743
750
  kind: "blocked",
744
751
  runId: config.runId,
@@ -763,7 +770,7 @@ async function main() {
763
770
  return;
764
771
  }
765
772
 
766
- reservedUsd += estimatedWorstCaseUsd;
773
+ reservedUsd = roundUsd(reservedUsd + estimatedWorstCaseUsd);
767
774
  const attemptId = nextAttemptId(state);
768
775
  const responseSpoolPath = join(streamDirectory, `${attemptId}.txt`);
769
776
  appendRow(spoolPath, {
@@ -892,10 +899,10 @@ async function main() {
892
899
  endedAt: new Date().toISOString(),
893
900
  });
894
901
  state.reservations.delete(attemptId);
895
- state.spentUsd += leaseChargeUsd;
902
+ state.spentUsd = roundUsd(state.spentUsd + leaseChargeUsd);
896
903
  if (!outgoing.destroyed && !outgoing.writableEnded) outgoing.end();
897
904
  } finally {
898
- reservedUsd -= estimatedWorstCaseUsd;
905
+ reservedUsd = roundUsd(reservedUsd - estimatedWorstCaseUsd);
899
906
  }
900
907
  }
901
908
 
package/docs/commands.md CHANGED
@@ -61,6 +61,16 @@ Options:
61
61
  --base-url <url> OpenAI-compatible provider base URL
62
62
  --api-key-env <name> environment variable containing the
63
63
  provider API key
64
+ --route <kind> where candidate replays run: api (the
65
+ --base-url endpoint), claude-login or
66
+ codex-login (the claude or codex CLI
67
+ signed in on this machine); a plan route
68
+ needs --judge-route (choices: "api",
69
+ "claude-login", "codex-login")
70
+ --judge-route <kind> where the built-in judge runs: api,
71
+ claude-login or codex-login; use a vendor
72
+ other than the candidates' (choices:
73
+ "api", "claude-login", "codex-login")
64
74
  --max-cost-usd <amount> optional hard spend cap in USD; omit to
65
75
  run uncapped so every case and judge cell
66
76
  completes
@@ -99,8 +109,8 @@ Options:
99
109
  "aggregate", "confirm", "report")
100
110
  --code-graph <path> Graphify graph.json for static code
101
111
  context in the report; never evidence
102
- --yes accept the newest discovered trace without
103
- prompting
112
+ --yes accept the newest discovered trace and the
113
+ saved model route without prompting
104
114
  -h, --help display help for command
105
115
  ```
106
116
 
@@ -121,6 +131,16 @@ Options:
121
131
  --base-url <url> OpenAI-compatible provider base URL
122
132
  --api-key-env <name> environment variable containing the
123
133
  provider API key
134
+ --route <kind> where candidate replays run: api (the
135
+ --base-url endpoint), claude-login or
136
+ codex-login (the claude or codex CLI
137
+ signed in on this machine); a plan route
138
+ needs --judge-route (choices: "api",
139
+ "claude-login", "codex-login")
140
+ --judge-route <kind> where the built-in judge runs: api,
141
+ claude-login or codex-login; use a vendor
142
+ other than the candidates' (choices:
143
+ "api", "claude-login", "codex-login")
124
144
  --max-cost-usd <amount> optional hard spend cap in USD; omit to
125
145
  run uncapped so every case and judge cell
126
146
  completes
@@ -152,8 +172,8 @@ Options:
152
172
  --evaluator-gate-metric <name> scorer metric used for release gates
153
173
  --evaluator-gate-threshold <value> fallback pass threshold when the evaluator
154
174
  omits a pass decision
155
- --yes accept the newest discovered trace without
156
- prompting
175
+ --yes accept the newest discovered trace and the
176
+ saved model route without prompting
157
177
  --approved-run <digest> scope projection to one merged approved
158
178
  swap
159
179
  -h, --help display help for command
@@ -251,6 +271,16 @@ Options:
251
271
  --base-url <url> OpenAI-compatible provider base URL
252
272
  --api-key-env <name> environment variable containing the
253
273
  provider API key
274
+ --route <kind> where candidate replays run: api (the
275
+ --base-url endpoint), claude-login or
276
+ codex-login (the claude or codex CLI
277
+ signed in on this machine); a plan route
278
+ needs --judge-route (choices: "api",
279
+ "claude-login", "codex-login")
280
+ --judge-route <kind> where the built-in judge runs: api,
281
+ claude-login or codex-login; use a vendor
282
+ other than the candidates' (choices:
283
+ "api", "claude-login", "codex-login")
254
284
  --max-cost-usd <amount> optional hard spend cap in USD; omit to
255
285
  run uncapped so every case and judge cell
256
286
  completes
@@ -320,6 +350,16 @@ Options:
320
350
  --base-url <url> OpenAI-compatible provider base URL
321
351
  --api-key-env <name> environment variable containing the
322
352
  provider API key
353
+ --route <kind> where candidate replays run: api (the
354
+ --base-url endpoint), claude-login or
355
+ codex-login (the claude or codex CLI
356
+ signed in on this machine); a plan route
357
+ needs --judge-route (choices: "api",
358
+ "claude-login", "codex-login")
359
+ --judge-route <kind> where the built-in judge runs: api,
360
+ claude-login or codex-login; use a vendor
361
+ other than the candidates' (choices:
362
+ "api", "claude-login", "codex-login")
323
363
  --max-cost-usd <amount> optional hard spend cap in USD; omit to
324
364
  run uncapped so every case and judge cell
325
365
  completes
@@ -585,7 +625,7 @@ print documentation packaged with this CLI
585
625
  Arguments:
586
626
  name packaged document name (choices: "commands", "evaluators",
587
627
  "exit-codes", "gateways", "getting-started", "github",
588
- "github-actions", "modeb")
628
+ "github-actions", "modeb", "model-routes")
589
629
 
590
630
  Options:
591
631
  -h, --help display help for command
@@ -1,6 +1,6 @@
1
1
  # Evaluators
2
2
 
3
- The default evaluator is the built-in judge selected from the configured provider catalog. Candidate and reference families are excluded when choosing the judge. The exclusion applies per replayed call site: when the call sites of one traced family use models of different vendors (for example `openai/...` and `anthropic/...`), each call site's cases are judged by a model from neither the candidate's model family nor that call site's. Confirmation judges the recorded final output, so its judge also comes from outside the final step's model family.
3
+ The default evaluator is the built-in judge selected from the configured provider catalog. Candidate and reference families are excluded when choosing the judge. The exclusion applies per replayed call site: when the call sites of one traced family use models of different vendors (for example `openai/...` and `anthropic/...`), each call site's cases are judged by a model from neither the candidate's model family nor that call site's. Confirmation judges the recorded final output, so its judge also comes from outside the final step's model family. In replay, when the recorded answer is one assistant message of text, the judge compares the candidate's answer with that text; any other recorded answer, such as one that calls tools, has parts other than text or holds several messages, is shown to the judge as recorded.
4
4
 
5
5
  An external evaluator is requested with `--evaluator <provider>` on `init`, `estimate`, `replay`, and `confirm`, where the provider is `braintrust`, `langfuse`, `langsmith`, or `promptfoo`. Every other `--evaluator-*` option is a usage error without `--evaluator`.
6
6
 
@@ -6,7 +6,7 @@ Rightmodeler reserves exit codes `0` through `3` for machine-readable outcomes.
6
6
 
7
7
  - `0`: the command completed and no recommendation is being reported. Planning and partial `--through` runs also return `0` when successful.
8
8
  - `1`: a complete `init` or `report` found an actionable recommendation.
9
- - `2`: the run needs input at a resumable boundary, such as missing traces, a cancelled trace prompt, provider configuration, or required confirmation configuration.
9
+ - `2`: the run needs input at a resumable boundary, such as missing traces, a cancelled trace prompt, provider configuration, required confirmation configuration, or a plan's usage limit (rerun after it resets).
10
10
  - `3`: the cost budget was reached at a resumable boundary.
11
11
  - `10` or greater: command-line or runtime failure.
12
12
 
@@ -54,13 +54,18 @@ Use `--output json` for one result object or `--output jsonl` for stage events f
54
54
  - `invalid_option` (exit `2`): correct the option and rerun; use `rightmodeler <command> --help` for accepted values.
55
55
  - `invalid_policy_file` (exit `2`): fix the named field in the `--policy` file and rerun; `qualityFloor` must be greater than 0.8 and less than 1, `shortlistTop` a positive integer, `allowModels` and `denyModels` arrays of model ids.
56
56
  - `invalid_pricing_file` (exit `2`): fix `--pricing-file` to map each model id to non-negative `input` and `output` USD per token and, optionally, a positive integer `maxOutputTokens`, then rerun.
57
+ - `judge_family_unknown` (exit `2`): the catalog's model ids name no vendor, so the built-in judge could share a vendor with the candidate or the recorded model; use a gateway whose ids carry their vendor (`vendor/model`), or grade with `--evaluator`.
57
58
  - `missing_provider_configuration` (exit `2`): pass `--base-url <url>` and, if needed, `--api-key-env <environment-variable-name>` naming a populated variable.
58
59
  - `missing_traces_path` (exit `2`): pass `--traces <path>` pointing to an existing trace file or directory.
59
60
  - `mixed_trace_formats` (exit `2`): split the directory so every file uses the same trace format, or pass one file with `--traces`.
60
61
  - `modeb_cloud_unavailable` (exit `2`): install the optional sandbox SDK and set its credentials, or set `"backend": "docker"` in the `--modeb-config` file, then rerun.
62
+ - `no_neutral_judge` (exit `2`): the catalog has no priced model from a vendor other than both the candidate's and the recorded model's, which the built-in judge needs; list or price one (a multi-vendor gateway, `--catalog-reference` or `--pricing-file`), run the judge through another vendor's CLI with `--judge-route`, or grade with `--evaluator`.
61
63
  - `no_priced_candidates` (exit `2`): point `--base-url` at a catalog that publishes per-token pricing, pass `--catalog-reference <url>`, expose priced LiteLLM `GET /model/info`, or pass `--pricing-file <path>`, then rerun.
62
64
  - `no_replayable_call_sites` (exit `2`): point `--repo` at a service with plain text completions, or add a matcher for a text call site, then rerun.
63
65
  - `not_git_repository` (exit `2`): run the command again from a Git repository with at least one commit.
66
+ - `plan_cli_unavailable` (exit `2`): the CLI a plan route runs is missing, older than the verified version, or changed an output shape, or `CI` is set; install or update it (`claude update`, or `npm install -g @openai/codex@latest` for `codex`), unset `CI` on your own machine, or use an API route with `--base-url`.
67
+ - `plan_login_required` (exit `2`): sign the CLI in to your plan (`claude auth login`, or `codex login` with ChatGPT) and remove any API key setting it would use, or use an API route with `--base-url`; finished calls are kept.
68
+ - `plan_usage_limit` (exit `2`): a plan you are signed in to reached its usage limit; rerun the same command after the reset time in the message, and completed replay and judge calls are kept and not repeated; or choose a route that does not use this plan with `--route` or `--judge-route`.
64
69
  - `stage_not_completed` (exit `2`): run `rightmodeler init --through <stage>` first, then rerun the command.
65
70
  - `unusable_trace_input` (exit `2`): the selected discovered trace could not be adapted; rerun and choose a different trace file.
66
71
  - `usage_error` (exit `10`): the command line is invalid; `message` carries the parser text.
package/docs/gateways.md CHANGED
@@ -69,7 +69,7 @@ On Kubernetes (Kubernetes 1.32 or newer, Envoy Gateway 1.8.1 or newer, Helm char
69
69
 
70
70
  ## Bifrost
71
71
 
72
- Verified on the open-source Bifrost gateway transports/v2.2.1 (Apache 2.0, `maximhq/bifrost:v2.2.1`); pin the image, because releases arrive weekly. Enterprise features are a separate image and are not covered here.
72
+ Verified on the open-source Bifrost gateway transports/v2.2.1 (Apache 2.0, `maximhq/bifrost:v2.2.1`); pin the image, because releases arrive weekly. Enterprise features ship in a separate image; this guide uses the open-source one.
73
73
 
74
74
  As a replay route:
75
75
 
@@ -7,7 +7,13 @@ Rightmodeler analyzes recorded model calls, replays them against cheaper candida
7
7
  - Node.js 24 or newer.
8
8
  - A Git repository to analyze.
9
9
  - Trace input in a supported format.
10
- - An OpenAI-compatible provider base URL and the name of an environment variable containing its API key before replay begins. Its `/v1/models` catalog should publish per-token pricing. OpenRouter and Vercel AI Gateway do. For a LiteLLM endpoint, Rightmodeler can fall back to `GET /model/info`; for a gateway that lists bare model ids, pass `--catalog-reference`; for bare OpenAI or another unpriced endpoint, pass `--pricing-file`.
10
+ - An OpenAI-compatible provider base URL and the name of an environment variable containing its API key before replay begins. Its `/v1/models` catalog should publish per-token pricing. OpenRouter and Vercel AI Gateway do. For a LiteLLM endpoint, Rightmodeler can fall back to `GET /model/info`; for a gateway that lists bare model ids or a direct OpenAI or Anthropic key, pass `--catalog-reference`; for another unpriced endpoint, pass `--pricing-file`.
11
+
12
+ Instead of an API key, candidates or the judge can run through the `claude`
13
+ CLI you are signed in to on this machine, under your Claude plan:
14
+ `--route claude-login` with a `--judge-route` from another vendor. Run
15
+ `rightmodeler docs model-routes` for what a plan route measures, what it costs
16
+ your plan, and its safeguards.
11
17
 
12
18
  Supported trace sources are OTel GenAI, AI SDK telemetry, OpenAI JSONL,
13
19
  Langfuse, Braintrust, LangSmith, OpenInference, Helicone, Bifrost, W&B Weave,
@@ -21,6 +27,13 @@ Run this from the repository you want to analyze:
21
27
  npx rightmodeler init
22
28
  ```
23
29
 
30
+ Before it looks for traces, an interactive `init` or `estimate` that will reach
31
+ replay asks how to call models: through the `claude` or `codex` CLI signed in on
32
+ this machine, or through an API key for OpenRouter, Vercel AI Gateway, OpenAI,
33
+ Anthropic or another OpenAI-compatible endpoint. It prints the equivalent flags
34
+ and saves the answer as the next default, which `--yes` applies without asking.
35
+ `npx rightmodeler docs model-routes` describes the choice.
36
+
24
37
  Rightmodeler checks conventional local trace files, Claude Code transcripts for
25
38
  the repository, and Codex sessions whose recorded working directory matches the
26
39
  repository. In an interactive terminal it lists matches newest-first with an
@@ -121,9 +134,8 @@ Graph edges are never replay trials, runtime proof, or quality evidence, and the
121
134
 
122
135
  Rightmodeler reads per-token pricing from the model catalog. When every catalog
123
136
  entry has null pricing and no `--pricing-file` is set, it requests LiteLLM
124
- `GET /model/info` on the same host. Use `--pricing-file` for bare OpenAI
125
- endpoints or when `/model/info` has no usable per-token costs; file entries
126
- override provider pricing.
137
+ `GET /model/info` on the same host. Use `--pricing-file` when `/model/info` has
138
+ no usable per-token costs; file entries override provider pricing.
127
139
 
128
140
  Some gateways answer `/v1/models` with bare model ids: Envoy AI Gateway lists
129
141
  the models a route declares, and Bifrost lists custom providers with ids and
@@ -169,6 +181,33 @@ Without usable pricing from the catalog, a catalog reference, LiteLLM
169
181
  instead of reporting zero cost. The judge must be priced too, so price at least
170
182
  one model from a family other than the current model's and the candidate's.
171
183
 
184
+ ### Direct OpenAI and Anthropic keys
185
+
186
+ Point `--base-url` at `https://api.openai.com/v1` or
187
+ `https://api.anthropic.com/v1` and name the variable that holds the key with
188
+ `--api-key-env`. Neither vendor's model list publishes prices, so also pass
189
+ `--catalog-reference https://ai-gateway.vercel.sh/v1/models`. Rightmodeler takes
190
+ the vendor from these two hosts: their bare ids are that vendor's models and
191
+ join the reference as `openai/<id>` or `anthropic/<id>`. Dated and dotted forms
192
+ of a name match within one vendor, in the reference and in your traces, so
193
+ `claude-haiku-4-5-20251001` matches `anthropic/claude-haiku-4.5`. For Anthropic,
194
+ rightmodeler sends `anthropic-version: 2023-06-01` (a
195
+ `--header 'anthropic-version: ...'` value wins), reads every page of the model
196
+ list, and never asks for structured output, which Anthropic's OpenAI-compatible
197
+ endpoint ignores. For OpenAI, rightmodeler caps a reply's length with
198
+ `max_completion_tokens`, which OpenAI's chat reference names in place of the
199
+ deprecated `max_tokens`; every other host still gets `max_tokens`.
200
+
201
+ The built-in judge must come from a vendor other than both the candidate's and
202
+ the recorded model's, and rightmodeler checks this before any paid call, also
203
+ when an unreachable `--evaluator` falls back to the built-in judge. One vendor's
204
+ key lists only that vendor's models, so a run on it alone stops with
205
+ `no_neutral_judge` unless the judge runs through the other vendor's CLI you are
206
+ signed in to (`--judge-route`, see `rightmodeler docs model-routes`) or your own
207
+ evaluator grades the replays. Bare ids from any other host name no vendor, and
208
+ a run on them stops with `judge_family_unknown`; use a gateway whose ids carry
209
+ their vendor (`vendor/model`), or `--evaluator`.
210
+
172
211
  ## Which model answered
173
212
 
174
213
  Rightmodeler checks every replay and judge response, and in Mode B every response to a step whose model it sets, before it counts. It records the model the response names, and it leaves a response out of the evidence when that model is not the one it asked for, when a gateway reports that it answered from its cache (Portkey's `x-portkey-cache-status: HIT`, Bifrost's `cache_debug.cache_hit`), or when a gateway reports that it changed the request (a Portkey hook with `transformed: true`, Bifrost's compat plugin dropping parameters). Such a response is recorded with `attribution: "substituted"`, is never graded, and counts as `attribution_substituted`; more than 5% of a family's replays substituted abstains the family. A `replay_responses_substituted` warning counts them.
@@ -48,7 +48,7 @@ concurrency:
48
48
  cancel-in-progress: false
49
49
 
50
50
  env:
51
- RIGHTMODELER_VERSION: "0.3.0"
51
+ RIGHTMODELER_VERSION: "0.4.0"
52
52
  RIGHTMODELER_TRACES: traces
53
53
  RIGHTMODELER_MAX_COST_USD: "5"
54
54
  RM_ANNOTATE: |
@@ -305,7 +305,7 @@ jobs:
305
305
 
306
306
  ## Optional: a GitHub App token
307
307
 
308
- Not tested live: rightmodeler's acceptance runs of this workflow use `GITHUB_TOKEN` only. A GitHub App installation token lets the draft's `pull_request` workflows start without approval, and the draft is authored by `<app-slug>[bot]`. To use one:
308
+ The workflow above runs on the built-in `GITHUB_TOKEN`, the supported default. A GitHub App installation token is an optional upgrade: the draft's `pull_request` workflows start without approval, and the draft is authored by `<app-slug>[bot]`. To use one:
309
309
 
310
310
  1. Create a GitHub App with the permissions in [GitHub](github.md) and install it on the repository.
311
311
  2. Store its client ID in the repository variable `RIGHTMODELER_APP_CLIENT_ID` and its private key in the repository secret `RIGHTMODELER_APP_PRIVATE_KEY`.
@@ -0,0 +1,111 @@
1
+ # Model routes
2
+
3
+ Replay sends each recorded request to cheaper candidate models, and the built-in judge grades every answer against the recorded one. Candidates and the judge each run on a route: an API endpoint with a key, or a plan you are signed in to on this machine through its command-line tool.
4
+
5
+ ## Routes
6
+
7
+ - `--route api` sends candidate calls to the `--base-url` endpoint with the key named by `--api-key-env`. It is the default when `--base-url` is given, so commands from earlier releases behave as before.
8
+ - `--route claude-login` runs candidate calls through the `claude` CLI you are signed in to, under your Claude plan.
9
+ - `--route codex-login` runs candidate calls through the `codex` CLI you are signed in to, under your ChatGPT plan.
10
+ - `--judge-route api`, `--judge-route claude-login` or `--judge-route codex-login` picks where the built-in judge runs. With `--base-url` and no route flag, the judge uses the API route too.
11
+
12
+ The judge must come from a vendor other than both the candidate's and the recorded model's. A plan route serves one vendor's models, so a plan `--route` needs an explicit `--judge-route`; without one, rightmodeler stops with `invalid_option`. Candidates from the judge route's own vendor are left out with the warning `judge_vendor_candidates_dropped`, and the run stops before any model call with `no_neutral_judge` when no candidate is left or the recorded model comes from the judge's vendor.
13
+
14
+ `--base-url`, `--api-key-env` and `--header` configure the API route, so they are refused when both `--route` and `--judge-route` name a plan route. `--detach` and `--modeb-config` are refused when either role uses a plan route: detached replay and Mode B confirmation call models only through an API endpoint. `--evaluator` works with every route; the built-in judge on `--judge-route` grades when the evaluator is unreachable.
15
+
16
+ ```sh
17
+ npx rightmodeler init --traces traces.jsonl --route claude-login --judge-route api --base-url https://openrouter.ai/api/v1 --api-key-env OPENROUTER_API_KEY
18
+ npx rightmodeler init --traces traces.jsonl --base-url https://api.openai.com/v1 --api-key-env OPENAI_API_KEY --catalog-reference https://ai-gateway.vercel.sh/v1/models --judge-route claude-login
19
+ npx rightmodeler init --traces traces.jsonl --route codex-login --judge-route claude-login
20
+ ```
21
+
22
+ ## Choosing at the start
23
+
24
+ Run in a terminal, `init` and `estimate` first ask how to call models, before the trace question and before any stage runs.
25
+
26
+ - **Your plans:** rightmodeler lists the `claude` and `codex` CLIs on this machine, each with its status and, when it cannot be used, how to fix it. The status comes from each CLI's version and login commands, run with the same API key variables kept away as for a call, and with the variable a saved answer names kept away too. You choose the CLI that replays candidates (choose the vendor your app calls today) and where the judge runs: the other CLI, or an API key for OpenRouter, Vercel AI Gateway or another OpenAI-compatible endpoint. The first time a plan route is chosen, rightmodeler shows what it sends and asks `[y/N]`; any other answer than `y` continues with no route.
27
+ - **An API key:** OpenRouter, Vercel AI Gateway, OpenAI, Anthropic, or another OpenAI-compatible endpoint, then the name of the environment variable that holds the key. The question checks only whether that variable is set, never its value. It accepts only a name made of capital letters, digits and underscores, so a pasted key is refused and never repeated or saved, and it refuses a base URL that carries a user name, a password or a query string.
28
+
29
+ The answer is saved in the store as `project/setup/model-route.json`: route names, a base URL and a variable name, never a key. The next interactive run shows it as flags and keeps it when you press Enter; type `c` to choose again. `--yes` applies it without asking. The question and the saved answer are skipped with `--output json` or `jsonl`, without a terminal, when `--route`, `--judge-route`, `--base-url`, `--api-key-env` or `--header` is passed, with `init --plan`, and with `--through` before `replay`. Scripts pass the flags the question prints. With `--modeb-config`, only API routes through a multi-vendor endpoint are offered, because Mode B confirmation runs only through an API key. Ctrl-C or Ctrl-D at a question continues with no route: the free stages run and replay stops with `missing_provider_configuration`.
30
+
31
+ ## API routes
32
+
33
+ | Choice | `--base-url` | Key variable, by default | Also passed |
34
+ | ----------------- | --------------------------------- | ------------------------ | --------------------------------------------------------------------------------------------------- |
35
+ | OpenRouter | `https://openrouter.ai/api/v1` | `OPENROUTER_API_KEY` | nothing |
36
+ | Vercel AI Gateway | `https://ai-gateway.vercel.sh/v1` | `AI_GATEWAY_API_KEY` | nothing |
37
+ | OpenAI | `https://api.openai.com/v1` | `OPENAI_API_KEY` | `--route api --judge-route claude-login --catalog-reference https://ai-gateway.vercel.sh/v1/models` |
38
+ | Anthropic | `https://api.anthropic.com/v1` | `ANTHROPIC_API_KEY` | `--route api --judge-route codex-login --catalog-reference https://ai-gateway.vercel.sh/v1/models` |
39
+ | Another endpoint | the URL you type | `RIGHTMODELER_API_KEY` | nothing |
40
+
41
+ OpenAI's and Anthropic's APIs serve only their own models and list no prices, so the judge runs through the other vendor's CLI signed in on this machine and prices come from the public list; without that CLI, choose a gateway. A variable that is not set yet is still saved: set it in your own shell before replay.
42
+
43
+ ## Use your Claude plan
44
+
45
+ `--route claude-login` and `--judge-route claude-login` run the `claude` CLI (Claude Code) already installed and signed in on this machine, version 2.1.282 or newer. Rightmodeler runs your own unmodified binary as a child process under the login it already holds. It never reads, stores or forwards a token, and never opens `~/.claude`, the macOS Keychain or a credential file.
46
+
47
+ Before any model call, rightmodeler runs `claude --version` and `claude auth status`. It accepts only a Claude plan login: signed in, through Anthropic directly, with a claude.ai login or a `claude setup-token` token. Anything else, such as an API key, Amazon Bedrock or Google Vertex, stops with `plan_login_required`.
48
+
49
+ Each call runs in a fresh, empty temporary directory with no tools, no MCP servers, no settings files, no skills, no auto memory, no session file and one turn: `claude -p --model <id> --system-prompt-file <file> --tools "" --strict-mcp-config --disable-slash-commands --setting-sources "" --no-session-persistence --max-turns 1 --settings {"switchModelsOnFlag":false} --output-format stream-json --verbose`, with the recorded user message on standard input. Because no session file is written, replays never appear in the Claude Code transcripts rightmodeler reads as traces.
50
+
51
+ When `claude` runs the built-in judge, rightmodeler adds `--json-schema` with the verdict schema and sets `--max-turns 3`. `claude` returns the verdict through its StructuredOutput tool, checks it against the schema and asks the model again when it does not match; rightmodeler reads the verdict from the `structured_output` field. A verdict that still does not match after two corrections, or a call that ends without `structured_output`, counts as a malformed judge answer, as on the API route. Candidate replays get no schema and keep `--max-turns 1`. In rightmodeler's check on `claude` 2.1.283 on 2026-09-26, `--max-turns 3` allowed the first answer and two corrections, and the schema added about 540 input tokens to each Opus 5 judge call; cost estimates use the request's own tokens.
52
+
53
+ An API key variable would make `claude` bill the key instead of your plan: in non-interactive mode the key is always used when present. Rightmodeler removes every `ANTHROPIC_*` and `OPENAI_*` variable, `CODEX_API_KEY`, the run's `--api-key-env` variable, the other key variables the first-run question offers (`OPENROUTER_API_KEY`, `AI_GATEWAY_API_KEY` and `RIGHTMODELER_API_KEY`), and the variables of a parent Claude Code session from the child's environment, and warns with `plan_route_key_withheld` when `ANTHROPIC_API_KEY` or `ANTHROPIC_AUTH_TOKEN` was set, naming the variable and never its value. It keeps `CLAUDE_CODE_OAUTH_TOKEN` and `CLAUDE_CONFIG_DIR`. If `claude` still reports that a call would be paid by a key, rightmodeler stops that call at once with `plan_login_required`.
54
+
55
+ Rightmodeler runs at most `--max-concurrency` `claude` processes at a time (default 2) and stops any still running when it exits. It refuses plan routes when the `CI` environment variable is set: a plan is for your own machine, not for continuous integration.
56
+
57
+ ## Use your ChatGPT plan through Codex
58
+
59
+ `--route codex-login` and `--judge-route codex-login` run the `codex` CLI already installed and signed in on this machine, version 0.153.3 or newer. As with Claude, rightmodeler runs your own unmodified binary under the login it already holds.
60
+
61
+ - What runs: `codex exec --json --ephemeral --ignore-user-config --ignore-rules --strict-config --skip-git-repo-check --sandbox read-only`, in a fresh, empty temporary directory, with the recorded system and developer messages in a temporary instructions file (`model_instructions_file`), the recorded user message on standard input, and settings that turn off the shell and other tools, web search, apps, plugins, skills, memories, the injected permission, app and environment context, project instructions and history. Code execution is off too (`features.code_mode_host=false`). A case with no system or developer message runs with `instructions=""` instead of the file, because Codex refuses an empty instructions file and would otherwise add its own default instructions. `--ephemeral` keeps the session file out of `~/.codex/sessions`, where rightmodeler reads Codex traces, and `--ignore-user-config` keeps your `config.toml`, MCP servers and plugins out of the call. `--strict-config` makes a newer `codex` that renamed one of these settings stop with `plan_cli_unavailable` instead of quietly turning a tool back on. When `codex` runs the built-in judge, rightmodeler also writes the verdict schema to that directory and passes `--output-schema`, so Codex's final message is the verdict as JSON.
62
+ - What it never touches: `auth.json`, the keyring or any token. Rightmodeler picks the credential store by name (`file`, else `keyring`) from what `codex login status` reports, and passes that name to each call. `CODEX_API_KEY`, every `OPENAI_*` and every `ANTHROPIC_*` variable, and the key variables listed for Claude above are kept away from `codex`, and `plan_route_key_withheld` names a set `CODEX_API_KEY`, `OPENAI_API_KEY`, `OPENAI_FEDERATION_RULE_ID` or `OPENAI_IDENTITY_TOKEN_FILE`, never its value. `CODEX_HOME` and `CODEX_ACCESS_TOKEN` are passed through unread. A login with an API key, workload identity or Amazon Bedrock stops with `plan_login_required`, because it would not use your ChatGPT plan.
63
+ - What a Codex route measures: Codex adds about 2,000 input tokens of its own context to each call. That includes your global instructions file, `$CODEX_HOME/AGENTS.md` (or `AGENTS.override.md`), which Codex always adds and cannot be told to leave out; rightmodeler checks only that the file exists, never reads it, and warns once with `codex_global_instructions`. Cost estimates use the recorded input tokens, so this context does not change the savings. Codex cannot set temperature or an output limit, and each model runs at its default reasoning effort, as on the API route, which sends none. Codex reports no latency, so the p50 latency of a Codex answer reads n/a.
64
+ - What Codex does not report: which model answered. Rightmodeler records the requested model, and leaves a call out of the evidence when Codex reports that it rerouted the call to another model.
65
+ - Tools: Codex keeps a code tool and a patch tool registered that no setting removes, so a call can include a tool step its output does not show. A call where Codex reports a tool step is left out of the evidence. In rightmodeler's check on `codex-cli` 0.153.3 on 2026-09-25, a prompt asking the model to read a file with its code tool used 13,325 input tokens with code execution on and 3,826 with it off; in both runs the model answered that it could not run anything, did not return the file's contents, and the output showed no tool step.
66
+ - Models: the list comes from `codex debug models` of the same binary that makes the calls, so update Codex to see newer models. GPT-5.5 retires from ChatGPT sign-in on October 14, 2026; it stays on the OpenAI API.
67
+ - Limits: a limit on one model blocks that candidate, or moves the judge to its next model. An account usage limit, a workspace out of credits or a spend cap starts no new call and exits `2` with `plan_usage_limit`, quoting Codex's reset time as Codex wrote it. Rerun after the reset: finished calls are kept. `codex exec` reports no remaining allowance, so rightmodeler cannot warn before the limit.
68
+ - Concurrency: at most `--max-concurrency` `codex` processes run at a time (default 2), each stopped when rightmodeler exits. Plan routes are refused when `CI` is set.
69
+
70
+ ## What a plan route measures
71
+
72
+ A plan route measures the model inside a coding CLI, not the API request your application makes:
73
+
74
+ - Claude Code adds its own instructions to every call, even when rightmodeler replaces the system prompt: 380 to 539 input tokens in testing, including the signed-in account's email address and today's date.
75
+ - The CLI cannot set temperature or an output limit, so the recorded values are not applied, and an answer can be longer than on the API route.
76
+ - It sends one user turn. Recorded cases with an earlier assistant or tool turn, or more than one user message, are left out of the replay sample with the warning `plan_route_cases_left_out`.
77
+ - With `claude`, latency is the API time the CLI reports, without its start-up time.
78
+ - With `claude`, rightmodeler checks which model answered on every call, and records a substitution when `claude` answers with another model, takes more than one turn or calls a tool. A judge call is expected to use the StructuredOutput tool and to report up to four turns for the first answer and two corrections, so it counts as a substitution only when `claude` calls another tool or takes more than four turns.
79
+
80
+ The report's "Model routes" section, and the pull request that `apply` opens, say when a result was measured through a plan.
81
+
82
+ With `claude`, your recorded prompts are sent to Anthropic under your plan account's data settings. Anthropic's data-usage page says: "We will train new models using data from Free, Pro, and Max accounts when this setting is on (including when you use Claude Code from these accounts)" (`https://code.claude.com/docs/en/data-usage`).
83
+
84
+ ## Costs, the cap and usage limits
85
+
86
+ Calls through a plan are not billed in dollars: they use your plan's usage allowance; with `claude`, the same 5-hour and weekly limits as your own Claude Code sessions. To budget and compare them, rightmodeler prices each call at API list prices from a public price list, `https://ai-gateway.vercel.sh/v1/models`, read without a key. `--catalog-reference <url-or-path>` replaces that list, including with a local file when the public list is unreachable. `--pricing-file` only overrides the prices of the ids it names.
87
+
88
+ A call's cost is the recorded request's input tokens plus the output tokens the CLI reports, at list price, and is always marked as an estimate. The CLI's added instructions stay out of it, so savings compare with the recorded calls. The report and `estimate` label these amounts as list-price equivalents and show dollars billed through an API route separately.
89
+
90
+ Plan routes have no default limit on calls or spend: a run makes as many calls as it needs, and the vendor's own usage limit is the only stop (the run resumes after the reset). `--max-cost-usd` is optional; when you set it on a plan route it caps the list-price equivalent, as a soft cap: the CLI sets no output limit, so a call can cost more than its reservation.
91
+
92
+ When `claude` reports your plan near its limit, rightmodeler warns once with `plan_usage_warning`, giving the share used and the reset time. When the plan reaches its limit, or further calls would bill usage credits (`credits_required`, or `claude` reporting overage), rightmodeler starts no new call and exits `2` with `plan_usage_limit`, quoting the reset time. Calls already running finish and are kept, so up to `--max-concurrency` calls can still use credits when overage begins. Rerun the same command after the reset: finished replay and judge calls are reused.
93
+
94
+ Anthropic announced, then paused, a change that would take `claude -p` usage off your plan's limits and onto a monthly credit, then usage credits (`https://support.claude.com/en/articles/15036540-use-the-claude-agent-sdk-with-your-claude-plan`). For now, the page says, `claude -p` still draws from your subscription's usage limits. If the change takes effect, rightmodeler stops at `credits_required` or overage as described above.
95
+
96
+ Rightmodeler leaves out Claude Fable models, because in non-interactive mode "When a Fable request there would bill to usage credits, Claude Code bills it without asking" (`https://code.claude.com/docs/en/model-config`), and `[1m]` variants, whose 1M context can require usage credits. Models the price list does not price are left out with `plan_model_unpriced`.
97
+
98
+ ## Errors
99
+
100
+ - `plan_cli_unavailable` (exit `2`): `claude` is missing or older than 2.1.282, `codex` is missing or older than 0.153.3, `CI` is set, the CLI changed an output shape rightmodeler relies on, or `codex` rejected one of rightmodeler's isolation settings.
101
+ - `plan_login_required` (exit `2`): `claude` is not signed in with a Claude plan, `codex` is not signed in with ChatGPT, the CLI would use an API key, or it lost its login during the run.
102
+ - `plan_usage_limit` (exit `2`): the plan reached its usage limit.
103
+ - `no_neutral_judge` and `judge_family_unknown` (exit `2`): no judge from a third vendor is available. See [Exit codes](exit-codes.md).
104
+
105
+ ## Terms
106
+
107
+ Anthropic's legal page says OAuth authentication "is designed to support ordinary use of Claude Code and other native Anthropic applications", that third-party developers may not "route requests through Free, Pro, or Max plan credentials on behalf of their users", and that this does not "prevent an end user from signing in to the unmodified Claude Code binary with their own Claude subscription" (`https://code.claude.com/docs/en/legal-and-compliance`). The Agent SDK page adds: "Unless previously approved, Anthropic does not allow third party developers to offer claude.ai login or rate limits for their products" (`https://code.claude.com/docs/en/agent-sdk/overview`). Anthropic's Consumer Terms restrict access "through automated or non-human means, whether through a bot, script, or otherwise" except "where we otherwise explicitly permit it" (`https://www.anthropic.com/legal/consumer-terms`).
108
+
109
+ OpenAI's non-interactive guide says "`codex exec` reuses saved CLI authentication by default." and "API keys are the right default for automation because they are simpler to provision and rotate. Use this path only if you specifically need to run as your Codex account." (`https://learn.chatgpt.com/docs/non-interactive-mode`). OpenAI's pricing page lists "Codex SDK, `codex exec`, and scriptable workflows" as available on Plus, Pro, Business and Enterprise (`https://learn.chatgpt.com/docs/pricing`). OpenAI's Terms of Use list "Automatically or programmatically extract data or Output" among what you may not do (`https://openai.com/policies/terms-of-use/`). OpenAI's CI guide says "Do not use this workflow for public or open-source repositories" about seeding `auth.json` on CI runners (`https://learn.chatgpt.com/docs/auth/ci-cd-auth`).
110
+
111
+ Refusing plan routes when `CI` is set is rightmodeler's own product choice: a plan route is local and opt-in. Rightmodeler runs on your machine, for you, with your own unmodified `claude` or `codex` and your own login, and never handles a credential. Whether a replay fits your plan's terms is your decision; the API route is always available.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "rightmodeler",
3
- "version": "0.3.0",
3
+ "version": "0.4.0",
4
4
  "description": "Find and prove safe model substitutions from the agent traces you already have.",
5
5
  "homepage": "https://www.rightmodeler.com",
6
6
  "type": "module",