rightmodeler 0.2.1 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,328 @@
1
+ # GitHub Actions
2
+
3
+ This workflow runs rightmodeler in GitHub Actions with the job's built-in `GITHUB_TOKEN`. It needs no GitHub App and no personal access token.
4
+
5
+ ## What it does
6
+
7
+ - Runs `init` every Monday and whenever a push to `main` changes `traces/`.
8
+ - Opens the draft pull request only when a person runs the workflow with `command=apply`.
9
+ - Runs `watch` every 6 hours on each swap pull request that is still being watched.
10
+ - Never merges. A person reviews and merges the draft.
11
+
12
+ ## Setup
13
+
14
+ 1. Let GitHub Actions open pull requests. In the repository settings, under Actions, General, Workflow permissions, turn on "Allow GitHub Actions to create and approve pull requests", or run `gh api -X PUT repos/<owner>/<repo>/actions/permissions/workflow -F can_approve_pull_request_reviews=true`. This is the only repository setting the workflow needs. In a repository owned by an organization, the organization must allow it first.
15
+ 2. Set the repository variable `RIGHTMODELER_PROVIDER_BASE_URL` and the repository secret `RIGHTMODELER_PROVIDER_API_KEY`, for example with `gh variable set` and `gh secret set`.
16
+ 3. Commit traces under `traces/`, or change `RIGHTMODELER_TRACES` and the `paths` filter together.
17
+ 4. Change `main` if the default branch has another name, and adjust `RIGHTMODELER_MAX_COST_USD`.
18
+ 5. Save the workflow below as `.github/workflows/rightmodeler.yml`.
19
+
20
+ Each job grants `GITHUB_TOKEN` only what its command needs: `apply` gets `contents: write` and `pull-requests: write`, and `watch` gets `contents: read`, `pull-requests: write`, `checks: read` and `statuses: read`. See [GitHub](github.md) for what each command does with them.
21
+
22
+ ## The workflow
23
+
24
+ ```yaml
25
+ name: rightmodeler
26
+
27
+ on:
28
+ schedule:
29
+ - cron: "17 5 * * 1"
30
+ - cron: "43 */6 * * *"
31
+ push:
32
+ branches: [main]
33
+ paths:
34
+ - "traces/**"
35
+ workflow_dispatch:
36
+ inputs:
37
+ command:
38
+ description: Which rightmodeler command to run
39
+ type: choice
40
+ options: [init, apply, watch]
41
+ default: init
42
+
43
+ permissions:
44
+ contents: read
45
+
46
+ concurrency:
47
+ group: rightmodeler-store
48
+ cancel-in-progress: false
49
+
50
+ env:
51
+ RIGHTMODELER_VERSION: "0.4.0"
52
+ RIGHTMODELER_TRACES: traces
53
+ RIGHTMODELER_MAX_COST_USD: "5"
54
+ RM_ANNOTATE: |
55
+ const { readFileSync } = require("node:fs");
56
+ const data = (s) => String(s).replace(/%/g, "%25").replace(/\r/g, "%0D").replace(/\n/g, "%0A");
57
+ const prop = (s) => data(s).replace(/:/g, "%3A").replace(/,/g, "%2C");
58
+ for (const file of process.argv.slice(1)) {
59
+ let text = "";
60
+ try { text = readFileSync(file, "utf8"); } catch { continue; }
61
+ for (const line of text.split("\n")) {
62
+ let value;
63
+ try { value = JSON.parse(line); } catch { continue; }
64
+ if (value === null || typeof value !== "object") continue;
65
+ if (value.event === "result") value = value.result;
66
+ if (value.event === "warning") {
67
+ console.log(`::warning title=${prop(`rightmodeler ${value.code}`)}::${data(value.message)}`);
68
+ } else if (value.status === "refused" && Array.isArray(value.reasons)) {
69
+ for (const reason of value.reasons) {
70
+ console.log(`::error title=${prop(`rightmodeler ${reason.code}`)}::${data(reason.message)}`);
71
+ }
72
+ } else if (typeof value.code === "string" && typeof value.message === "string") {
73
+ console.log(`::error title=${prop(`rightmodeler ${value.code}`)}::${data(`${value.message} Remedy: ${value.remedy ?? ""}`)}`);
74
+ }
75
+ }
76
+ }
77
+
78
+ jobs:
79
+ init:
80
+ if: >-
81
+ github.event_name == 'push' ||
82
+ (github.event_name == 'schedule' && github.event.schedule == '17 5 * * 1') ||
83
+ (github.event_name == 'workflow_dispatch' && inputs.command == 'init')
84
+ runs-on: ubuntu-latest
85
+ timeout-minutes: 60
86
+ steps:
87
+ - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
88
+ with:
89
+ fetch-depth: 0
90
+ persist-credentials: false
91
+ - uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0
92
+ with:
93
+ node-version: 24
94
+ - uses: actions/cache/restore@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0
95
+ with:
96
+ path: .rightmodeler
97
+ key: rightmodeler-store-${{ github.run_id }}-${{ github.run_attempt }}
98
+ restore-keys: rightmodeler-store-
99
+ - id: init
100
+ name: Find and prove cheaper models
101
+ env:
102
+ RIGHTMODELER_PROVIDER_BASE_URL: ${{ vars.RIGHTMODELER_PROVIDER_BASE_URL }}
103
+ RIGHTMODELER_PROVIDER_API_KEY: ${{ secrets.RIGHTMODELER_PROVIDER_API_KEY }}
104
+ run: |
105
+ out="$RUNNER_TEMP/rightmodeler"
106
+ mkdir -p "$out"
107
+ npx --yes "rightmodeler@${RIGHTMODELER_VERSION}" --version
108
+ set +e
109
+ npx --yes "rightmodeler@${RIGHTMODELER_VERSION}" init \
110
+ --traces "$RIGHTMODELER_TRACES" \
111
+ --base-url "$RIGHTMODELER_PROVIDER_BASE_URL" \
112
+ --api-key-env RIGHTMODELER_PROVIDER_API_KEY \
113
+ --max-cost-usd "$RIGHTMODELER_MAX_COST_USD" \
114
+ --output jsonl --repo "$GITHUB_WORKSPACE" \
115
+ >"$out/init.jsonl" 2>"$out/init.err"
116
+ code=$?
117
+ set -e
118
+ node -e "$RM_ANNOTATE" "$out/init.jsonl" "$out/init.err"
119
+ report="$GITHUB_WORKSPACE/.rightmodeler/project/reports/report.md"
120
+ if [ -f "$report" ]; then
121
+ cp "$report" "$out/report.md"
122
+ cat "$report" >>"$GITHUB_STEP_SUMMARY"
123
+ fi
124
+ case "$code" in
125
+ 0) echo "recommendation=false" >>"$GITHUB_OUTPUT" ;;
126
+ 1)
127
+ echo "recommendation=true" >>"$GITHUB_OUTPUT"
128
+ echo "::notice title=rightmodeler::A proven swap is ready. Run this workflow with command=apply to open the draft pull request."
129
+ ;;
130
+ *)
131
+ echo "::error title=rightmodeler::init exited $code; the annotations above name the cause and the fix."
132
+ exit 1
133
+ ;;
134
+ esac
135
+ - if: always()
136
+ uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
137
+ with:
138
+ name: rightmodeler-init
139
+ path: ${{ runner.temp }}/rightmodeler/
140
+ if-no-files-found: ignore
141
+ - if: always()
142
+ uses: actions/cache/save@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0
143
+ with:
144
+ path: .rightmodeler
145
+ key: rightmodeler-store-${{ github.run_id }}-${{ github.run_attempt }}
146
+
147
+ apply:
148
+ if: github.event_name == 'workflow_dispatch' && inputs.command == 'apply'
149
+ runs-on: ubuntu-latest
150
+ timeout-minutes: 30
151
+ permissions:
152
+ contents: write
153
+ pull-requests: write
154
+ steps:
155
+ - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
156
+ with:
157
+ fetch-depth: 0
158
+ persist-credentials: false
159
+ - uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0
160
+ with:
161
+ node-version: 24
162
+ - uses: actions/cache/restore@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0
163
+ with:
164
+ path: .rightmodeler
165
+ key: rightmodeler-store-${{ github.run_id }}-${{ github.run_attempt }}
166
+ restore-keys: rightmodeler-store-
167
+ - id: apply
168
+ name: Open the draft pull request
169
+ env:
170
+ RIGHTMODELER_GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
171
+ run: |
172
+ out="$RUNNER_TEMP/rightmodeler"
173
+ mkdir -p "$out"
174
+ npx --yes "rightmodeler@${RIGHTMODELER_VERSION}" --version
175
+ run_apply() {
176
+ npx --yes "rightmodeler@${RIGHTMODELER_VERSION}" apply "$@" \
177
+ --owner "$GITHUB_REPOSITORY_OWNER" \
178
+ --github-repo "${GITHUB_REPOSITORY#*/}" \
179
+ --github-base-url "$GITHUB_API_URL" \
180
+ --github-token-env RIGHTMODELER_GITHUB_TOKEN \
181
+ --output json --repo "$GITHUB_WORKSPACE"
182
+ }
183
+ for mode in dry-run apply; do
184
+ set +e
185
+ if [ "$mode" = "dry-run" ]; then
186
+ run_apply --dry-run >"$out/$mode.json" 2>"$out/$mode.err"
187
+ else
188
+ run_apply >"$out/$mode.json" 2>"$out/$mode.err"
189
+ fi
190
+ code=$?
191
+ set -e
192
+ node -e "$RM_ANNOTATE" "$out/$mode.json" "$out/$mode.err"
193
+ if [ "$code" -ne 0 ]; then
194
+ echo "::error title=rightmodeler::apply ($mode) exited $code; the annotations above name the cause and the fix."
195
+ exit 1
196
+ fi
197
+ done
198
+ pr="$(node -p 'JSON.parse(require("node:fs").readFileSync(process.argv[1], "utf8")).prNumber' "$out/apply.json")"
199
+ echo "pr-number=$pr" >>"$GITHUB_OUTPUT"
200
+ echo "Draft pull request #$pr is open for review. rightmodeler never merges it." >>"$GITHUB_STEP_SUMMARY"
201
+ - if: always()
202
+ uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
203
+ with:
204
+ name: rightmodeler-apply
205
+ path: ${{ runner.temp }}/rightmodeler/
206
+ if-no-files-found: ignore
207
+ - if: always()
208
+ uses: actions/cache/save@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0
209
+ with:
210
+ path: .rightmodeler
211
+ key: rightmodeler-store-${{ github.run_id }}-${{ github.run_attempt }}
212
+
213
+ watch:
214
+ if: >-
215
+ (github.event_name == 'schedule' && github.event.schedule == '43 */6 * * *') ||
216
+ (github.event_name == 'workflow_dispatch' && inputs.command == 'watch')
217
+ runs-on: ubuntu-latest
218
+ timeout-minutes: 30
219
+ permissions:
220
+ contents: read
221
+ pull-requests: write
222
+ checks: read
223
+ statuses: read
224
+ steps:
225
+ - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
226
+ with:
227
+ fetch-depth: 0
228
+ persist-credentials: false
229
+ - uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0
230
+ with:
231
+ node-version: 24
232
+ - uses: actions/cache/restore@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0
233
+ with:
234
+ path: .rightmodeler
235
+ key: rightmodeler-store-${{ github.run_id }}-${{ github.run_attempt }}
236
+ restore-keys: rightmodeler-store-
237
+ - id: watch
238
+ name: Reconcile open swap pull requests
239
+ env:
240
+ RIGHTMODELER_GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
241
+ run: |
242
+ out="$RUNNER_TEMP/rightmodeler"
243
+ mkdir -p "$out"
244
+ npx --yes "rightmodeler@${RIGHTMODELER_VERSION}" --version
245
+ set +e
246
+ npx --yes "rightmodeler@${RIGHTMODELER_VERSION}" status --output json \
247
+ --repo "$GITHUB_WORKSPACE" >"$out/status.json" 2>"$out/status.err"
248
+ code=$?
249
+ set -e
250
+ node -e "$RM_ANNOTATE" "$out/status.err"
251
+ if [ "$code" -ne 0 ]; then
252
+ echo "::error title=rightmodeler::status exited $code"
253
+ exit 1
254
+ fi
255
+ failed=0
256
+ for pr in $(node -p 'JSON.parse(require("node:fs").readFileSync(process.argv[1], "utf8")).pullRequests.map((entry) => entry.prNumber).join(" ")' "$out/status.json"); do
257
+ set +e
258
+ npx --yes "rightmodeler@${RIGHTMODELER_VERSION}" watch --pr "$pr" \
259
+ --owner "$GITHUB_REPOSITORY_OWNER" \
260
+ --github-repo "${GITHUB_REPOSITORY#*/}" \
261
+ --github-base-url "$GITHUB_API_URL" \
262
+ --github-token-env RIGHTMODELER_GITHUB_TOKEN \
263
+ --output json --repo "$GITHUB_WORKSPACE" \
264
+ >"$out/watch-$pr.json" 2>"$out/watch-$pr.err"
265
+ code=$?
266
+ set -e
267
+ node -e "$RM_ANNOTATE" "$out/watch-$pr.json" "$out/watch-$pr.err"
268
+ case "$code" in
269
+ 0) ;;
270
+ 1) echo "::notice title=rightmodeler::watch acted on pull request #$pr" ;;
271
+ 2)
272
+ held="$(node -p 'try { JSON.parse(require("node:fs").readFileSync(process.argv[1], "utf8")).status } catch { "" }' "$out/watch-$pr.json")"
273
+ if [ "$held" = "lock_held" ]; then
274
+ echo "::warning title=rightmodeler::another watcher holds the lock for pull request #$pr; the next run retries"
275
+ else
276
+ failed=1
277
+ fi
278
+ ;;
279
+ *) failed=1 ;;
280
+ esac
281
+ done
282
+ exit "$failed"
283
+ - if: always()
284
+ uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
285
+ with:
286
+ name: rightmodeler-watch
287
+ path: ${{ runner.temp }}/rightmodeler/
288
+ if-no-files-found: ignore
289
+ - if: always()
290
+ uses: actions/cache/save@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0
291
+ with:
292
+ path: .rightmodeler
293
+ key: rightmodeler-store-${{ github.run_id }}-${{ github.run_attempt }}
294
+ ```
295
+
296
+ ## How it behaves
297
+
298
+ - **The store.** `.rightmodeler/` lives in the Actions cache. Each run saves it under a new key and restores the newest `rightmodeler-store-` entry. GitHub removes entries unused for 7 days, and the 6-hourly watch keeps the store in use. Anyone with read access to the repository can read cache contents and uploaded artifacts, so use this workflow in private repositories.
299
+ - **One run at a time.** All runs share one concurrency group. If a run is already waiting, a newly queued run replaces it, so dispatch again if yours was replaced.
300
+ - **Annotations.** Every error and warning the CLI prints becomes an annotation that names its code, and errors also carry the remedy. `init` adds the report to the job summary, and every job uploads its output files as an artifact.
301
+ - **Stale evidence.** If `main` moves between `init` and `apply`, `apply` refuses with `stale_evidence`. Run `init` again.
302
+ - **Other branches.** A dispatch from another branch works on that branch. It starts from the default branch's store, saves its own copy that only that branch's runs restore, and `apply` opens the draft against that branch.
303
+ - **The token.** The draft is authored by `github-actions[bot]`, and the owners of the swapped files are requested as reviewers. GitHub starts no workflow for a push made with `GITHUB_TOKEN`. For a pull request that `GITHUB_TOKEN` opens, GitHub creates the `pull_request` workflow runs in an approval-required state, and a person with write access starts them with "Approve workflows to run" on the pull request.
304
+ - **Schedules.** GitHub can delay scheduled runs at busy times, and disables schedules in public repositories after 60 days without activity.
305
+
306
+ ## Optional: a GitHub App token
307
+
308
+ The workflow above runs on the built-in `GITHUB_TOKEN`, the supported default. A GitHub App installation token is an optional upgrade: the draft's `pull_request` workflows start without approval, and the draft is authored by `<app-slug>[bot]`. To use one:
309
+
310
+ 1. Create a GitHub App with the permissions in [GitHub](github.md) and install it on the repository.
311
+ 2. Store its client ID in the repository variable `RIGHTMODELER_APP_CLIENT_ID` and its private key in the repository secret `RIGHTMODELER_APP_PRIVATE_KEY`.
312
+ 3. In the `apply` and `watch` jobs, add this step before the rightmodeler step, and change that step's `RIGHTMODELER_GITHUB_TOKEN` to `${{ steps.app-token.outputs.token }}`:
313
+
314
+ ```
315
+ - id: app-token
316
+ uses: actions/create-github-app-token@bcd2ba49218906704ab6c1aa796996da409d3eb1 # v3.2.0
317
+ with:
318
+ client-id: ${{ vars.RIGHTMODELER_APP_CLIENT_ID }}
319
+ private-key: ${{ secrets.RIGHTMODELER_APP_PRIVATE_KEY }}
320
+ ```
321
+
322
+ ## Cloud Mode B
323
+
324
+ To confirm in Vercel Sandbox, add `--modeb-config <file>` to the `init` command and map the `VERCEL_TOKEN`, `VERCEL_TEAM_ID` and `VERCEL_PROJECT_ID` secrets into that step's `env`. See [Mode B](modeb.md).
325
+
326
+ ## Upgrading
327
+
328
+ Change `RIGHTMODELER_VERSION`. Each rightmodeler step first runs the CLI with `--version`, so a version npm cannot install fails the step with npm's error in its log instead of being mistaken for a rightmodeler exit code. Each release's copy of this guide pins that release, and `rightmodeler docs github-actions` prints the copy for the installed version.
package/docs/github.md ADDED
@@ -0,0 +1,110 @@
1
+ # GitHub
2
+
3
+ Three commands talk to GitHub. `apply` opens a draft pull request that changes model identifiers only and carries an evidence table. `watch` reconciles one of those pull requests per run. `rollback` opens a draft pull request that restores a merged swap. The CLI never merges a pull request: a person reviews and merges.
4
+
5
+ ## Tokens
6
+
7
+ Pass the name of the environment variable that holds the token with `--github-token-env`, never the token itself.
8
+
9
+ | Token | What works |
10
+ | ------------------------------------------------------------------------ | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
11
+ | GitHub App installation token (recommended) | `apply` and `watch` fully. |
12
+ | Classic personal access token, or `gh auth token`, with the `repo` scope | `apply` and `watch` fully. |
13
+ | Fine-grained personal access token | `apply` fully. `watch` works but cannot read check runs, so each pass prints the `github_checks_unavailable` warning and uses commit statuses only. |
14
+ | GitHub Actions `GITHUB_TOKEN` | `apply` and `watch` when the job grants `contents: write`, `pull-requests: write` and `statuses: read` (add `checks: read` for check runs) and the repository allows GitHub Actions to create pull requests. CI on the pull requests it opens waits for approval, which an App installation token avoids. |
15
+
16
+ Give a GitHub App these repository permissions: Contents read and write, Pull requests read and write, Checks read, Commit statuses read, and Metadata read. Pull requests it opens are authored by `<app-slug>[bot]`.
17
+
18
+ Give a fine-grained personal access token access to the repository with Contents read and write, Pull requests read and write, Commit statuses read, and Metadata read. GitHub offers no check-run permission for these tokens.
19
+
20
+ Commenting on a pull request needs only Pull requests write, so neither token needs the Issues permission.
21
+
22
+ For a ready workflow that uses `GITHUB_TOKEN`, see [GitHub Actions](github-actions.md).
23
+
24
+ ## API host
25
+
26
+ `--github-base-url` defaults to `https://api.github.com`. For GitHub Enterprise Server, pass its API URL, such as `https://github.example.com/api/v3`. Enterprise Server 3.21 or newer is needed, because the CLI sends REST API version `2026-03-10` and older servers answer every call with HTTP 400.
27
+
28
+ `--github-repo` defaults to the name of the directory given to `--repo` on `apply`, `watch` and `rollback`. Pass it when the directory name differs from the repository name on GitHub.
29
+
30
+ ## Apply
31
+
32
+ `apply` needs:
33
+
34
+ - a completed `init` run whose report recommends at least one swap;
35
+ - local `HEAD` on a branch and equal to the evidence revision, and that branch on GitHub at the same revision (the branch becomes the pull request base);
36
+ - no uncommitted change to any file the swap touches.
37
+
38
+ `--dry-run` runs every check and reads GitHub but writes nothing. Its result lists the branch, title, body, files and reviewers it would use. The reviewer list is shown before the pull request author is removed, because the author is known only once the pull request exists.
39
+
40
+ Reviewers:
41
+
42
+ - Owners of each swapped file come from the first of `.github/CODEOWNERS`, `CODEOWNERS` and `docs/CODEOWNERS` that exists, and the last matching rule wins.
43
+ - A file with no matching rule falls back to its three most recent `git blame` authors, matched to GitHub users by commit email. Blame needs full history, so fetch with depth 0 in CI.
44
+ - At most five users and five teams are requested, and never the pull request author.
45
+ - If GitHub rejects the batch with HTTP 422, each reviewer is requested alone and the rejected ones are dropped.
46
+ - Team reviewers need a repository owned by an organization, and every reviewer needs access to the repository.
47
+ - GitHub does not request code owners on draft pull requests by itself, so every review request on the draft comes from `apply`.
48
+
49
+ The branch is `<prefix>swap-<family>-<digest prefix>`, where the prefix is the one most common among the repository's recent branches, or `rightmodeler/` when no branch has one. The title is `perf(models): swap <families>` when the repository uses conventional commits, and `Swap <families> models` otherwise. The body is the repository's pull request template, if it has one, followed by the `## Rightmodeler evidence` table: revision, corpus version, and per family the decision, evaluators, cascade status, worst-case bound, models, cost per case, latency, caps and case IDs. Case IDs are SHA-256 digests of the replayed cases, never prompts. With `--code-graph <path>`, a `## Code context (Graphify)` section for the swapped call sites follows the table; the owners it lists are never requested as reviewers.
50
+
51
+ Rerunning `apply` for the same evidence returns the same open pull request with `status: "existing"` and requests reviewers again only if the first attempt left no record of them.
52
+
53
+ ## Refusal codes
54
+
55
+ A refusal exits `1` and prints a result with `status: "refused"` and one or more `reasons`, each with a `code`, a `message` and a `detail`.
56
+
57
+ Apply:
58
+
59
+ - `no_confirmed_recommendation`: no recommended family has confirmed or not-required cascade evidence.
60
+ - `release_gate_failed`: at least one release gate is not green.
61
+ - `inconsistent_evidence`: the selected families do not share one evidence revision, corpus and gate policy.
62
+ - `previously_rejected`: this evidence and swap set was previously rejected and needs new evidence before it can be proposed again.
63
+ - `stale_evidence`: the evidence revision does not match the repository `HEAD`, or the remote base moved beyond it; re-prove before applying.
64
+ - `detached_head`: the repository has no current branch to use as the pull request base.
65
+ - `dirty_worktree`: a file the swap touches has uncommitted changes; commit or stash them before applying.
66
+ - `stale_location`: a proposed swap no longer matches its scan-time file digest or no longer has one fresh source location.
67
+ - `diff_lint_failed`: the proposed diff changes something other than model identifiers.
68
+ - `formatter_blocked`: the repository's formatter changed content outside the proposed swap.
69
+ - `host_conventions_unreadable`: one or more repository instructions could not be read unambiguously.
70
+ - `invalid_repository_revision`: the evidence revision is not a 40- or 64-character lowercase hex object ID.
71
+ - `apply_branch_unowned`: the swap branch exists without a recorded apply start.
72
+ - `apply_branch_scope_mismatch`: the existing swap branch changes files outside its declared scope.
73
+ - `apply_resume_state_mismatch`: the recorded pre-apply state does not match the resumed change.
74
+ - `apply_restore_failed`: after a failed apply, a file on the swap branch could not be restored.
75
+
76
+ Rollback:
77
+
78
+ - `missing_remediation_evidence`: the pull request has no valid recorded apply evidence.
79
+ - `original_pr_not_merged`: only a merged swap pull request can be rolled back.
80
+ - `pre_apply_revision_unavailable`: the recorded pre-apply file revision is unavailable or does not match its digest.
81
+ - `post_apply_digest_mismatch`: the affected files no longer match the recorded post-apply state.
82
+ - `rollback_branch_unowned`: the rollback branch exists without a recorded rollback start or base revision.
83
+ - `rollback_branch_scope_mismatch`: the existing rollback branch changes files outside the recorded scope.
84
+ - `rollback_restore_mismatch`: the rollback did not restore the recorded pre-apply state.
85
+ - `rollback_restore_failed`: after a failed rollback, a file could not be restored to the base state.
86
+
87
+ ## Watch
88
+
89
+ Each `watch` run makes one pass over one pull request under a lock kept in the store, so overlapping runs do not act twice. Run it on a schedule or from repository events. A pass:
90
+
91
+ - records a merge and ends watching;
92
+ - records a close without merge as a rejection and ends watching, after which `apply` refuses that swap with `previously_rejected`;
93
+ - answers each human review and comment once with the stored evidence for the families it names, else for the families in the file it comments on, else for every family in the pull request;
94
+ - marks families for re-proof when a reviewer requests changes, or when the base branch changes a swapped file, and says so in a comment; the re-proof itself happens on the next pipeline run;
95
+ - on a failing check run or commit status, comments once, and if a check with the same name fails again under a new run on the same head commit, closes the pull request.
96
+
97
+ Exit codes:
98
+
99
+ - `0`: nothing needed doing.
100
+ - `1`: the pass took an action, listed in the result's `actions`.
101
+ - `2`: another watcher holds the lock, or the store has no completed run. With `--output json` or `jsonl`, a held lock prints a result with `"status":"lock_held"` on standard output, while a missing run prints an error with code `stage_not_completed` on standard error and nothing on standard output.
102
+ - `10` or greater: runtime failure.
103
+
104
+ When GitHub refuses to list check runs with HTTP 403, the pass still reconciles reviews, comments, commit statuses, merges and base-branch changes, and prints the `github_checks_unavailable` warning. Fine-grained personal access tokens always cause it. To include check runs, use a GitHub App installation token with Checks read or a classic token with the `repo` scope.
105
+
106
+ ## Rollback
107
+
108
+ `rollback --pr <number>` opens a draft pull request that restores the pre-apply contents of the files a merged swap changed. It refuses unless the original pull request merged, and a rerun returns the same rollback pull request.
109
+
110
+ See [Exit codes](exit-codes.md) and [Commands](commands.md) for every option.
package/docs/modeb.md CHANGED
@@ -6,6 +6,7 @@ Mode B runs confirmation cases inside a container when a recommendation can affe
6
6
  {
7
7
  "version": "1",
8
8
  "image": "my-agent:latest",
9
+ "backend": "docker",
9
10
  "appSpec": {
10
11
  "mountPath": ".",
11
12
  "command": ["node", "/rightmodeler/app/driver.mjs", "{caseFile}"],
@@ -26,6 +27,39 @@ Mode B runs confirmation cases inside a container when a recommendation can affe
26
27
  - `appSpec.command` is a non-empty array of non-empty arguments. At least one argument must contain `{caseFile}`; the harness replaces every occurrence with the in-container case file path.
27
28
  - `appSpec.installCommand` is optional. When present, it is a non-empty array of non-empty arguments run before the workload.
28
29
  - `stepMap` maps at least one canonical scanner step ID to the runtime step header emitted by the application. Runtime headers must be unique.
30
+ - `backend` is optional and is either `"docker"` (the default) or `"cloud"`. The cloud backend runs each case in a remote sandbox, so `image` must name an image that sandbox platform can pull, and the run fails before any case starts when the sandbox SDK or its credentials are absent.
29
31
  - `confirmMaxRunSets` is optional and must be a non-negative integer.
30
32
 
33
+ ## Runtime contract
34
+
35
+ - The container receives `OPENAI_BASE_URL=http://127.0.0.1:8787/v1`, a placeholder `OPENAI_API_KEY`, and the `RM_RUN_ID`, `RM_CASE_ID`, and `RM_EXECUTION_ID` identifiers.
36
+ - `/rightmodeler/scratch/driver/case.json` contains `{ "caseId", "input", "headers"? }`. The `headers` field is omitted when the case has no headers.
37
+ - Every model request must carry `x-rm-step` exactly once. Its value is the runtime step header selected through `stepMap`.
38
+ - Every model request must carry `x-rm-call`. Use one ID per logical call and reuse that ID when the SDK retries the call.
39
+ - The request body must be a JSON object with a non-empty `model` string.
40
+ - On a step the candidate does not replace, the request must name that step's current model by the id the provider catalog lists for it. The proxy holds prices only for the models the run's steps call, so a request for any other model is recorded as lost with `missing_pricing` and never forwarded.
41
+ - `max_completion_tokens` or `max_tokens` is accepted when present. When both are absent, the proxy reserves against the catalog maximum output when known, or 4096 tokens otherwise, and does not add a limit to the forwarded body.
42
+ - Streamed requests are sent with `stream_options.include_usage: true` and metered from the trailing usage chunk. A stream without a usage chunk is charged at its reservation.
43
+ - The last non-empty stdout line must be `{ "runId", "caseId", "executionId", "finalOutput" }`.
44
+ - A request is recorded as lost and never forwarded for `missing_correlation`, `duplicate_step_correlation`, `request_too_large` at 10 MiB, `malformed_json`, `invalid_request`, or `missing_pricing`.
45
+ - A case with any lost request is a lost execution. Lost reason counts are reported on the Mode B result.
46
+ - A request that the case lease cannot cover receives HTTP 402. The case is blocked on budget without an execution fact and is retried on rerun.
47
+ - If the host cannot observe a container exit within the configured timeout plus ten seconds, it force-removes the container and records the case as lost under `container_lifecycle`.
48
+ - The `docker` CLI must reach the Docker daemon. A missing daemon, a failed egress listener, or a failed container launch blocks affected cases with a named reason instead of recording executions, so a rerun retries them.
49
+ - The in-container proxy runs on both backends. The workload always reaches it at `OPENAI_BASE_URL`, and it meters the case lease and records every attempt. Only the hop after it differs: the Docker backend forwards to a host listener over `host.docker.internal`, while the cloud backend forwards straight to the provider and the sandbox platform's egress firewall attaches the model credential in flight. The credential never enters the sandbox on either backend.
50
+ - The workload is killed at the configured timeout by both the host and an in-container deadline.
51
+
52
+ ## Cloud backend
53
+
54
+ - The cloud backend runs each confirmation case in a short-lived Vercel Sandbox microVM in your own Vercel project. It is separate from the Vercel AI Gateway: the `VERCEL_*` variables only authorize creating sandboxes, and the model credential is always the variable named by `--api-key-env`, whatever the provider.
55
+ - Install: `npx rightmodeler` installs `@vercel/sandbox` as an optional dependency. Installing with `--omit=optional` leaves it out; the CLI still runs and the cloud backend reports `modeb_cloud_unavailable`.
56
+ - Credentials: either `VERCEL_OIDC_TOKEN` (from `vercel link` then `vercel env pull`; it expires after 12 hours) or all of `VERCEL_TOKEN`, `VERCEL_TEAM_ID` and `VERCEL_PROJECT_ID`. Use the access token in CI.
57
+ - Image: `image` must be a Vercel Container Registry reference or a managed image such as `vercel/sandbox/node:24`; a Docker Hub name does not work, and a custom image's entrypoint and command do not run. Commands run as the image's default user, which is not root (`ubuntu` on `vercel/sandbox/node:24`). An image the platform cannot start blocks its case with `launch-failed`, so a rerun retries it.
58
+ - Provider: the base URL must be HTTPS, because the sandbox firewall matches the provider host by TLS server name before it attaches the credential. Its path is kept, so a base URL such as `https://openrouter.ai/api/v1` reaches the same endpoints as on the Docker backend.
59
+ - Credential handling: the firewall adds the model credential to requests for the provider host only, and its policy lets every other host through, so this is credential brokering, not an egress allowlist.
60
+ - Metering: the in-sandbox proxy asks the provider for uncompressed responses so it can read usage from every answer. There is no host listener to mark answers of its own, so every HTTP answer is attributed to the provider, including one the platform firewall produces itself; a connection that fails is attributed to the network path and its case is recorded as a lost execution.
61
+ - Timing: the 60 second case deadline starts when the workload command starts and includes `appSpec.installCommand`; every case starts a fresh microVM, so bake dependencies into the image. The sandbox itself lives for the deadline plus 60 seconds, which also bounds how long an interrupted run can leave one running.
62
+ - Cleanup: sandboxes are created non-persistent and deleted after each case.
63
+ - Scoring: sandboxes never score. The host judges every case; sandbox output is only the candidate text.
64
+
31
65
  See [Commands](commands.md) for where `--modeb-config` is accepted and [Exit codes](exit-codes.md) for blocked or failed runs.
@@ -0,0 +1,111 @@
1
+ # Model routes
2
+
3
+ Replay sends each recorded request to cheaper candidate models, and the built-in judge grades every answer against the recorded one. Candidates and the judge each run on a route: an API endpoint with a key, or a plan you are signed in to on this machine through its command-line tool.
4
+
5
+ ## Routes
6
+
7
+ - `--route api` sends candidate calls to the `--base-url` endpoint with the key named by `--api-key-env`. It is the default when `--base-url` is given, so commands from earlier releases behave as before.
8
+ - `--route claude-login` runs candidate calls through the `claude` CLI you are signed in to, under your Claude plan.
9
+ - `--route codex-login` runs candidate calls through the `codex` CLI you are signed in to, under your ChatGPT plan.
10
+ - `--judge-route api`, `--judge-route claude-login` or `--judge-route codex-login` picks where the built-in judge runs. With `--base-url` and no route flag, the judge uses the API route too.
11
+
12
+ The judge must come from a vendor other than both the candidate's and the recorded model's. A plan route serves one vendor's models, so a plan `--route` needs an explicit `--judge-route`; without one, rightmodeler stops with `invalid_option`. Candidates from the judge route's own vendor are left out with the warning `judge_vendor_candidates_dropped`, and the run stops before any model call with `no_neutral_judge` when no candidate is left or the recorded model comes from the judge's vendor.
13
+
14
+ `--base-url`, `--api-key-env` and `--header` configure the API route, so they are refused when both `--route` and `--judge-route` name a plan route. `--detach` and `--modeb-config` are refused when either role uses a plan route: detached replay and Mode B confirmation call models only through an API endpoint. `--evaluator` works with every route; the built-in judge on `--judge-route` grades when the evaluator is unreachable.
15
+
16
+ ```sh
17
+ npx rightmodeler init --traces traces.jsonl --route claude-login --judge-route api --base-url https://openrouter.ai/api/v1 --api-key-env OPENROUTER_API_KEY
18
+ npx rightmodeler init --traces traces.jsonl --base-url https://api.openai.com/v1 --api-key-env OPENAI_API_KEY --catalog-reference https://ai-gateway.vercel.sh/v1/models --judge-route claude-login
19
+ npx rightmodeler init --traces traces.jsonl --route codex-login --judge-route claude-login
20
+ ```
21
+
22
+ ## Choosing at the start
23
+
24
+ Run in a terminal, `init` and `estimate` first ask how to call models, before the trace question and before any stage runs.
25
+
26
+ - **Your plans:** rightmodeler lists the `claude` and `codex` CLIs on this machine, each with its status and, when it cannot be used, how to fix it. The status comes from each CLI's version and login commands, run with the same API key variables kept away as for a call, and with the variable a saved answer names kept away too. You choose the CLI that replays candidates (choose the vendor your app calls today) and where the judge runs: the other CLI, or an API key for OpenRouter, Vercel AI Gateway or another OpenAI-compatible endpoint. The first time a plan route is chosen, rightmodeler shows what it sends and asks `[y/N]`; any other answer than `y` continues with no route.
27
+ - **An API key:** OpenRouter, Vercel AI Gateway, OpenAI, Anthropic, or another OpenAI-compatible endpoint, then the name of the environment variable that holds the key. The question checks only whether that variable is set, never its value. It accepts only a name made of capital letters, digits and underscores, so a pasted key is refused and never repeated or saved, and it refuses a base URL that carries a user name, a password or a query string.
28
+
29
+ The answer is saved in the store as `project/setup/model-route.json`: route names, a base URL and a variable name, never a key. The next interactive run shows it as flags and keeps it when you press Enter; type `c` to choose again. `--yes` applies it without asking. The question and the saved answer are skipped with `--output json` or `jsonl`, without a terminal, when `--route`, `--judge-route`, `--base-url`, `--api-key-env` or `--header` is passed, with `init --plan`, and with `--through` before `replay`. Scripts pass the flags the question prints. With `--modeb-config`, only API routes through a multi-vendor endpoint are offered, because Mode B confirmation runs only through an API key. Ctrl-C or Ctrl-D at a question continues with no route: the free stages run and replay stops with `missing_provider_configuration`.
30
+
31
+ ## API routes
32
+
33
+ | Choice | `--base-url` | Key variable, by default | Also passed |
34
+ | ----------------- | --------------------------------- | ------------------------ | --------------------------------------------------------------------------------------------------- |
35
+ | OpenRouter | `https://openrouter.ai/api/v1` | `OPENROUTER_API_KEY` | nothing |
36
+ | Vercel AI Gateway | `https://ai-gateway.vercel.sh/v1` | `AI_GATEWAY_API_KEY` | nothing |
37
+ | OpenAI | `https://api.openai.com/v1` | `OPENAI_API_KEY` | `--route api --judge-route claude-login --catalog-reference https://ai-gateway.vercel.sh/v1/models` |
38
+ | Anthropic | `https://api.anthropic.com/v1` | `ANTHROPIC_API_KEY` | `--route api --judge-route codex-login --catalog-reference https://ai-gateway.vercel.sh/v1/models` |
39
+ | Another endpoint | the URL you type | `RIGHTMODELER_API_KEY` | nothing |
40
+
41
+ OpenAI's and Anthropic's APIs serve only their own models and list no prices, so the judge runs through the other vendor's CLI signed in on this machine and prices come from the public list; without that CLI, choose a gateway. A variable that is not set yet is still saved: set it in your own shell before replay.
42
+
43
+ ## Use your Claude plan
44
+
45
+ `--route claude-login` and `--judge-route claude-login` run the `claude` CLI (Claude Code) already installed and signed in on this machine, version 2.1.282 or newer. Rightmodeler runs your own unmodified binary as a child process under the login it already holds. It never reads, stores or forwards a token, and never opens `~/.claude`, the macOS Keychain or a credential file.
46
+
47
+ Before any model call, rightmodeler runs `claude --version` and `claude auth status`. It accepts only a Claude plan login: signed in, through Anthropic directly, with a claude.ai login or a `claude setup-token` token. Anything else, such as an API key, Amazon Bedrock or Google Vertex, stops with `plan_login_required`.
48
+
49
+ Each call runs in a fresh, empty temporary directory with no tools, no MCP servers, no settings files, no skills, no auto memory, no session file and one turn: `claude -p --model <id> --system-prompt-file <file> --tools "" --strict-mcp-config --disable-slash-commands --setting-sources "" --no-session-persistence --max-turns 1 --settings {"switchModelsOnFlag":false} --output-format stream-json --verbose`, with the recorded user message on standard input. Because no session file is written, replays never appear in the Claude Code transcripts rightmodeler reads as traces.
50
+
51
+ When `claude` runs the built-in judge, rightmodeler adds `--json-schema` with the verdict schema and sets `--max-turns 3`. `claude` returns the verdict through its StructuredOutput tool, checks it against the schema and asks the model again when it does not match; rightmodeler reads the verdict from the `structured_output` field. A verdict that still does not match after two corrections, or a call that ends without `structured_output`, counts as a malformed judge answer, as on the API route. Candidate replays get no schema and keep `--max-turns 1`. In rightmodeler's check on `claude` 2.1.283 on 2026-09-26, `--max-turns 3` allowed the first answer and two corrections, and the schema added about 540 input tokens to each Opus 5 judge call; cost estimates use the request's own tokens.
52
+
53
+ An API key variable would make `claude` bill the key instead of your plan: in non-interactive mode the key is always used when present. Rightmodeler removes every `ANTHROPIC_*` and `OPENAI_*` variable, `CODEX_API_KEY`, the run's `--api-key-env` variable, the other key variables the first-run question offers (`OPENROUTER_API_KEY`, `AI_GATEWAY_API_KEY` and `RIGHTMODELER_API_KEY`), and the variables of a parent Claude Code session from the child's environment, and warns with `plan_route_key_withheld` when `ANTHROPIC_API_KEY` or `ANTHROPIC_AUTH_TOKEN` was set, naming the variable and never its value. It keeps `CLAUDE_CODE_OAUTH_TOKEN` and `CLAUDE_CONFIG_DIR`. If `claude` still reports that a call would be paid by a key, rightmodeler stops that call at once with `plan_login_required`.
54
+
55
+ Rightmodeler runs at most `--max-concurrency` `claude` processes at a time (default 2) and stops any still running when it exits. It refuses plan routes when the `CI` environment variable is set: a plan is for your own machine, not for continuous integration.
56
+
57
+ ## Use your ChatGPT plan through Codex
58
+
59
+ `--route codex-login` and `--judge-route codex-login` run the `codex` CLI already installed and signed in on this machine, version 0.153.3 or newer. As with Claude, rightmodeler runs your own unmodified binary under the login it already holds.
60
+
61
+ - What runs: `codex exec --json --ephemeral --ignore-user-config --ignore-rules --strict-config --skip-git-repo-check --sandbox read-only`, in a fresh, empty temporary directory, with the recorded system and developer messages in a temporary instructions file (`model_instructions_file`), the recorded user message on standard input, and settings that turn off the shell and other tools, web search, apps, plugins, skills, memories, the injected permission, app and environment context, project instructions and history. Code execution is off too (`features.code_mode_host=false`). A case with no system or developer message runs with `instructions=""` instead of the file, because Codex refuses an empty instructions file and would otherwise add its own default instructions. `--ephemeral` keeps the session file out of `~/.codex/sessions`, where rightmodeler reads Codex traces, and `--ignore-user-config` keeps your `config.toml`, MCP servers and plugins out of the call. `--strict-config` makes a newer `codex` that renamed one of these settings stop with `plan_cli_unavailable` instead of quietly turning a tool back on. When `codex` runs the built-in judge, rightmodeler also writes the verdict schema to that directory and passes `--output-schema`, so Codex's final message is the verdict as JSON.
62
+ - What it never touches: `auth.json`, the keyring or any token. Rightmodeler picks the credential store by name (`file`, else `keyring`) from what `codex login status` reports, and passes that name to each call. `CODEX_API_KEY`, every `OPENAI_*` and every `ANTHROPIC_*` variable, and the key variables listed for Claude above are kept away from `codex`, and `plan_route_key_withheld` names a set `CODEX_API_KEY`, `OPENAI_API_KEY`, `OPENAI_FEDERATION_RULE_ID` or `OPENAI_IDENTITY_TOKEN_FILE`, never its value. `CODEX_HOME` and `CODEX_ACCESS_TOKEN` are passed through unread. A login with an API key, workload identity or Amazon Bedrock stops with `plan_login_required`, because it would not use your ChatGPT plan.
63
+ - What a Codex route measures: Codex adds about 2,000 input tokens of its own context to each call. That includes your global instructions file, `$CODEX_HOME/AGENTS.md` (or `AGENTS.override.md`), which Codex always adds and cannot be told to leave out; rightmodeler checks only that the file exists, never reads it, and warns once with `codex_global_instructions`. Cost estimates use the recorded input tokens, so this context does not change the savings. Codex cannot set temperature or an output limit, and each model runs at its default reasoning effort, as on the API route, which sends none. Codex reports no latency, so the p50 latency of a Codex answer reads n/a.
64
+ - What Codex does not report: which model answered. Rightmodeler records the requested model, and leaves a call out of the evidence when Codex reports that it rerouted the call to another model.
65
+ - Tools: Codex keeps a code tool and a patch tool registered that no setting removes, so a call can include a tool step its output does not show. A call where Codex reports a tool step is left out of the evidence. In rightmodeler's check on `codex-cli` 0.153.3 on 2026-09-25, a prompt asking the model to read a file with its code tool used 13,325 input tokens with code execution on and 3,826 with it off; in both runs the model answered that it could not run anything, did not return the file's contents, and the output showed no tool step.
66
+ - Models: the list comes from `codex debug models` of the same binary that makes the calls, so update Codex to see newer models. GPT-5.5 retires from ChatGPT sign-in on October 14, 2026; it stays on the OpenAI API.
67
+ - Limits: a limit on one model blocks that candidate, or moves the judge to its next model. An account usage limit, a workspace out of credits or a spend cap starts no new call and exits `2` with `plan_usage_limit`, quoting Codex's reset time as Codex wrote it. Rerun after the reset: finished calls are kept. `codex exec` reports no remaining allowance, so rightmodeler cannot warn before the limit.
68
+ - Concurrency: at most `--max-concurrency` `codex` processes run at a time (default 2), each stopped when rightmodeler exits. Plan routes are refused when `CI` is set.
69
+
70
+ ## What a plan route measures
71
+
72
+ A plan route measures the model inside a coding CLI, not the API request your application makes:
73
+
74
+ - Claude Code adds its own instructions to every call, even when rightmodeler replaces the system prompt: 380 to 539 input tokens in testing, including the signed-in account's email address and today's date.
75
+ - The CLI cannot set temperature or an output limit, so the recorded values are not applied, and an answer can be longer than on the API route.
76
+ - It sends one user turn. Recorded cases with an earlier assistant or tool turn, or more than one user message, are left out of the replay sample with the warning `plan_route_cases_left_out`.
77
+ - With `claude`, latency is the API time the CLI reports, without its start-up time.
78
+ - With `claude`, rightmodeler checks which model answered on every call, and records a substitution when `claude` answers with another model, takes more than one turn or calls a tool. A judge call is expected to use the StructuredOutput tool and to report up to four turns for the first answer and two corrections, so it counts as a substitution only when `claude` calls another tool or takes more than four turns.
79
+
80
+ The report's "Model routes" section, and the pull request that `apply` opens, say when a result was measured through a plan.
81
+
82
+ With `claude`, your recorded prompts are sent to Anthropic under your plan account's data settings. Anthropic's data-usage page says: "We will train new models using data from Free, Pro, and Max accounts when this setting is on (including when you use Claude Code from these accounts)" (`https://code.claude.com/docs/en/data-usage`).
83
+
84
+ ## Costs, the cap and usage limits
85
+
86
+ Calls through a plan are not billed in dollars: they use your plan's usage allowance; with `claude`, the same 5-hour and weekly limits as your own Claude Code sessions. To budget and compare them, rightmodeler prices each call at API list prices from a public price list, `https://ai-gateway.vercel.sh/v1/models`, read without a key. `--catalog-reference <url-or-path>` replaces that list, including with a local file when the public list is unreachable. `--pricing-file` only overrides the prices of the ids it names.
87
+
88
+ A call's cost is the recorded request's input tokens plus the output tokens the CLI reports, at list price, and is always marked as an estimate. The CLI's added instructions stay out of it, so savings compare with the recorded calls. The report and `estimate` label these amounts as list-price equivalents and show dollars billed through an API route separately.
89
+
90
+ Plan routes have no default limit on calls or spend: a run makes as many calls as it needs, and the vendor's own usage limit is the only stop (the run resumes after the reset). `--max-cost-usd` is optional; when you set it on a plan route it caps the list-price equivalent, as a soft cap: the CLI sets no output limit, so a call can cost more than its reservation.
91
+
92
+ When `claude` reports your plan near its limit, rightmodeler warns once with `plan_usage_warning`, giving the share used and the reset time. When the plan reaches its limit, or further calls would bill usage credits (`credits_required`, or `claude` reporting overage), rightmodeler starts no new call and exits `2` with `plan_usage_limit`, quoting the reset time. Calls already running finish and are kept, so up to `--max-concurrency` calls can still use credits when overage begins. Rerun the same command after the reset: finished replay and judge calls are reused.
93
+
94
+ Anthropic announced, then paused, a change that would take `claude -p` usage off your plan's limits and onto a monthly credit, then usage credits (`https://support.claude.com/en/articles/15036540-use-the-claude-agent-sdk-with-your-claude-plan`). For now, the page says, `claude -p` still draws from your subscription's usage limits. If the change takes effect, rightmodeler stops at `credits_required` or overage as described above.
95
+
96
+ Rightmodeler leaves out Claude Fable models, because in non-interactive mode "When a Fable request there would bill to usage credits, Claude Code bills it without asking" (`https://code.claude.com/docs/en/model-config`), and `[1m]` variants, whose 1M context can require usage credits. Models the price list does not price are left out with `plan_model_unpriced`.
97
+
98
+ ## Errors
99
+
100
+ - `plan_cli_unavailable` (exit `2`): `claude` is missing or older than 2.1.282, `codex` is missing or older than 0.153.3, `CI` is set, the CLI changed an output shape rightmodeler relies on, or `codex` rejected one of rightmodeler's isolation settings.
101
+ - `plan_login_required` (exit `2`): `claude` is not signed in with a Claude plan, `codex` is not signed in with ChatGPT, the CLI would use an API key, or it lost its login during the run.
102
+ - `plan_usage_limit` (exit `2`): the plan reached its usage limit.
103
+ - `no_neutral_judge` and `judge_family_unknown` (exit `2`): no judge from a third vendor is available. See [Exit codes](exit-codes.md).
104
+
105
+ ## Terms
106
+
107
+ Anthropic's legal page says OAuth authentication "is designed to support ordinary use of Claude Code and other native Anthropic applications", that third-party developers may not "route requests through Free, Pro, or Max plan credentials on behalf of their users", and that this does not "prevent an end user from signing in to the unmodified Claude Code binary with their own Claude subscription" (`https://code.claude.com/docs/en/legal-and-compliance`). The Agent SDK page adds: "Unless previously approved, Anthropic does not allow third party developers to offer claude.ai login or rate limits for their products" (`https://code.claude.com/docs/en/agent-sdk/overview`). Anthropic's Consumer Terms restrict access "through automated or non-human means, whether through a bot, script, or otherwise" except "where we otherwise explicitly permit it" (`https://www.anthropic.com/legal/consumer-terms`).
108
+
109
+ OpenAI's non-interactive guide says "`codex exec` reuses saved CLI authentication by default." and "API keys are the right default for automation because they are simpler to provision and rotate. Use this path only if you specifically need to run as your Codex account." (`https://learn.chatgpt.com/docs/non-interactive-mode`). OpenAI's pricing page lists "Codex SDK, `codex exec`, and scriptable workflows" as available on Plus, Pro, Business and Enterprise (`https://learn.chatgpt.com/docs/pricing`). OpenAI's Terms of Use list "Automatically or programmatically extract data or Output" among what you may not do (`https://openai.com/policies/terms-of-use/`). OpenAI's CI guide says "Do not use this workflow for public or open-source repositories" about seeding `auth.json` on CI runners (`https://learn.chatgpt.com/docs/auth/ci-cd-auth`).
110
+
111
+ Refusing plan routes when `CI` is set is rightmodeler's own product choice: a plan route is local and opt-in. Rightmodeler runs on your machine, for you, with your own unmodified `claude` or `codex` and your own login, and never handles a credential. Whether a replay fits your plan's terms is your decision; the API route is always available.