@atmin.ai/review 0.1.0-alpha.2 → 0.1.0-alpha.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (93) hide show
  1. package/DEPENDENCIES.md +20 -0
  2. package/README.md +246 -18
  3. package/dist/ablation.d.ts +30 -0
  4. package/dist/ablation.js +90 -0
  5. package/dist/assessment.d.ts +4 -0
  6. package/dist/assessment.js +61 -5
  7. package/dist/callees.d.ts +14 -0
  8. package/dist/callees.js +73 -0
  9. package/dist/claim-result.d.ts +64 -0
  10. package/dist/claim-result.js +206 -0
  11. package/dist/claim-run.d.ts +38 -0
  12. package/dist/claim-run.js +227 -0
  13. package/dist/claim.d.ts +71 -0
  14. package/dist/claim.js +163 -0
  15. package/dist/cli.js +103 -11
  16. package/dist/code-review-runner.d.ts +25 -0
  17. package/dist/code-review-runner.js +280 -0
  18. package/dist/contracts.d.ts +395 -5
  19. package/dist/contracts.js +49 -11
  20. package/dist/cost-report.d.ts +4 -2
  21. package/dist/evidence.d.ts +34 -0
  22. package/dist/evidence.js +57 -0
  23. package/dist/failure-excerpt.d.ts +5 -0
  24. package/dist/failure-excerpt.js +63 -0
  25. package/dist/github/api.d.ts +40 -1
  26. package/dist/github/api.js +108 -16
  27. package/dist/github/billing.d.ts +32 -0
  28. package/dist/github/billing.js +215 -0
  29. package/dist/github/checks.js +11 -5
  30. package/dist/github/cli.js +76 -13
  31. package/dist/github/config.d.ts +4 -0
  32. package/dist/github/config.js +16 -1
  33. package/dist/github/dashboard-view.d.ts +205 -0
  34. package/dist/github/dashboard-view.js +261 -0
  35. package/dist/github/dashboard.d.ts +11 -0
  36. package/dist/github/dashboard.js +551 -0
  37. package/dist/github/inline.d.ts +11 -0
  38. package/dist/github/inline.js +107 -0
  39. package/dist/github/outcomes.d.ts +17 -0
  40. package/dist/github/outcomes.js +127 -0
  41. package/dist/github/repositories.d.ts +110 -0
  42. package/dist/github/repositories.js +263 -0
  43. package/dist/github/runner-api.d.ts +4 -0
  44. package/dist/github/runner-api.js +200 -0
  45. package/dist/github/runner.d.ts +16 -3
  46. package/dist/github/runner.js +108 -21
  47. package/dist/github/runners.d.ts +56 -0
  48. package/dist/github/runners.js +127 -0
  49. package/dist/github/settings.d.ts +34 -0
  50. package/dist/github/settings.js +76 -0
  51. package/dist/github/site.d.ts +2 -0
  52. package/dist/github/site.js +48 -0
  53. package/dist/github/store.d.ts +16 -3
  54. package/dist/github/store.js +97 -7
  55. package/dist/github/task.js +10 -4
  56. package/dist/github/webhook.d.ts +6 -2
  57. package/dist/github/webhook.js +78 -57
  58. package/dist/github/worker.d.ts +27 -5
  59. package/dist/github/worker.js +215 -51
  60. package/dist/investigation.d.ts +112 -6
  61. package/dist/investigation.js +268 -76
  62. package/dist/investigator.d.ts +278 -0
  63. package/dist/investigator.js +375 -0
  64. package/dist/jev.d.ts +34 -0
  65. package/dist/jev.js +224 -0
  66. package/dist/lifecycle.d.ts +41 -0
  67. package/dist/lifecycle.js +180 -0
  68. package/dist/models/claude-cli.d.ts +11 -0
  69. package/dist/models/claude-cli.js +107 -0
  70. package/dist/openai-model.js +7 -0
  71. package/dist/openrouter-model.js +48 -7
  72. package/dist/policy.d.ts +42 -0
  73. package/dist/policy.js +62 -0
  74. package/dist/rating.d.ts +21 -0
  75. package/dist/rating.js +65 -0
  76. package/dist/render-claim.d.ts +4 -0
  77. package/dist/render-claim.js +61 -0
  78. package/dist/render.d.ts +19 -3
  79. package/dist/render.js +95 -21
  80. package/dist/run.js +6 -3
  81. package/dist/runner-cli.d.ts +2 -0
  82. package/dist/runner-cli.js +9 -0
  83. package/dist/snapshot.d.ts +19 -3
  84. package/dist/snapshot.js +161 -13
  85. package/dist/symbolic.d.ts +37 -0
  86. package/dist/symbolic.js +360 -0
  87. package/dist/titles.d.ts +13 -0
  88. package/dist/titles.js +79 -0
  89. package/dist/trace.d.ts +8 -0
  90. package/dist/trace.js +87 -0
  91. package/dist/verification.d.ts +22 -0
  92. package/dist/verification.js +136 -0
  93. package/package.json +10 -6
package/DEPENDENCIES.md CHANGED
@@ -13,3 +13,23 @@ Runtime and development dependencies from the lockfile. Each retains its license
13
13
  | require-from-string | 2.0.2 | MIT | Runtime |
14
14
  | typescript | 5.9.3 | Apache-2.0 | Development |
15
15
  | undici-types | 7.18.2 | MIT | Development |
16
+
17
+ ## Dashboard (`web/`, served by the GitHub worker, not in the npm package)
18
+
19
+ Direct dependencies from `web/package-lock.json`. The bundle also includes the atmin
20
+ brand kit's customized shadcn/ui components (`web/LICENSE.shadcn`) and the Geist fonts
21
+ (`web/fonts/OFL.txt`).
22
+
23
+ | Package | Version | License | Use |
24
+ |---|---|---|---|
25
+ | class-variance-authority | 0.7.1 | Apache-2.0 | Runtime |
26
+ | clsx | 2.1.1 | MIT | Runtime |
27
+ | lucide-react | 1.45.0 | ISC | Runtime |
28
+ | radix-ui | 1.6.7 | MIT | Runtime |
29
+ | react | 19.2.4 | MIT | Runtime |
30
+ | react-dom | 19.2.4 | MIT | Runtime |
31
+ | tailwind-merge | 3.6.0 | MIT | Runtime |
32
+ | @tailwindcss/cli | 4.3.3 | MIT | Build |
33
+ | esbuild | 0.28.2 | MIT | Build |
34
+ | tailwindcss | 4.3.3 | MIT | Build |
35
+ | tw-animate-css | 1.4.0 | MIT | Build |
package/README.md CHANGED
@@ -6,11 +6,37 @@ alpha: model quality and severity calibration are still being measured.
6
6
 
7
7
  ## Install the alpha
8
8
 
9
+ ### Homebrew
10
+
11
+ On macOS or Linux:
12
+
13
+ ```sh
14
+ brew install atmin-inc/tap/atmin-review
15
+ gh auth login
16
+ cp "$(brew --prefix atmin-inc/tap/atmin-review)/share/atmin-review/profiles/smoke-openrouter-free.json" ./review-profile.json
17
+ ```
18
+
19
+ Set `OPENROUTER_API_KEY` in your environment, then run:
20
+
21
+ ```sh
22
+ atmin review https://github.com/OWNER/REPO/pull/123 \
23
+ --profile ./review-profile.json --out ./private-review
24
+ ```
25
+
26
+ Homebrew installs Node, Git, the GitHub CLI and the `atmin` command. `atmin <tool>`
27
+ runs the installed `atmin-<tool>` binary, so other atmin tools sit beside this one:
28
+ `atmin review …` runs `atmin-review`, and `atmin code-review-runner …` runs your own
29
+ review runner. All commands below are available as `atmin review` (or
30
+ `atmin-review`) without the `npx` prefix. If you installed the older
31
+ `atmin-inc/tap/atmin` formula, run `brew uninstall atmin` first.
32
+
33
+ ### npm
34
+
9
35
  Requires Node 24 or newer, Git, and an authenticated GitHub CLI (`gh auth login`).
10
36
  Install the versioned release in a fresh directory:
11
37
 
12
38
  ```sh
13
- npm install @atmin.ai/review@0.1.0-alpha.2
39
+ npm install @atmin.ai/review@0.1.0-alpha.4
14
40
  npx atmin-review --help
15
41
  cp node_modules/@atmin.ai/review/profiles/smoke-openrouter-free.json ./review-profile.json
16
42
  ```
@@ -28,6 +54,18 @@ availability and rate limits depend on the provider. The bundled paid DeepSeek
28
54
  profile caps a run at $2; copying it is an explicit choice to use paid inference.
29
55
  Direct OpenAI support uses `OPENAI_API_KEY` and an explicit profile.
30
56
 
57
+ `review` runs the claim pipeline described under [Current review direction](#current-review-direction)
58
+ below, the same run the GitHub worker publishes. Use `profiles/review-luna-openrouter.json`
59
+ (GPT-6 Luna through OpenRouter, capped at $2 a run, about $0.035 a run measured on benchmark
60
+ PRs). It is the measured `martian-luna-openrouter.json` with a larger input cap, so diffs up
61
+ to 512 KB fit with room to read; the benchmark profile keeps the cap it was measured with. Set `TYPESAFE_API_KEY` as well to turn on
62
+ the Jev rung, which is how it was measured; without it the report says the rung was off.
63
+ `profiles/review-luna-openai.json` runs the same model directly on OpenAI with
64
+ `OPENAI_API_KEY`. It has not been measured on a real review yet. Its cost is priced from
65
+ OpenAI's usage, including prompt tokens written to the cache (1.25x input) and the higher
66
+ rate for prompts over 272K tokens.
67
+ Confirmed findings the reviewer rated P3 are listed by location, not shown.
68
+
31
69
  Each run captures immutable commits, reads changed files and relevant callers,
32
70
  records anchored findings, and renders a report. It never executes repository
33
71
  scripts. Keep snapshot directories private: they contain repository source.
@@ -64,7 +102,8 @@ headline. A numerical average cannot cancel out a serious defect.
64
102
  Priority depends on a concrete trigger, impact, reachability and counterevidence.
65
103
  Findings cite immutable source. A model's reasoning is not proof of execution.
66
104
  The target commit's `.atmin/review.json` may enable optional suggestions and name
67
- required validation checks; a PR cannot weaken its own policy:
105
+ required validation checks; a PR cannot weaken its own policy. Without the file no
106
+ check is required, and the `atmin review` check reflects the review alone:
68
107
 
69
108
  ```json
70
109
  {"schemaVersion":1,"rubricVersion":"1","includeOptional":false,"requiredChecks":["change-validation"]}
@@ -72,13 +111,37 @@ required validation checks; a PR cannot weaken its own policy:
72
111
 
73
112
  ## GitHub App worker
74
113
 
114
+ The current source adds automatic CI refresh and inline findings. These changes
115
+ are not yet in the npm/Homebrew `0.1.0-alpha.2` package; use a source checkout
116
+ to operate this worker until the next packaged release.
117
+
75
118
  The worker receives signed webhooks, stores jobs in SQLite, updates one bot
76
- summary per PR, and publishes an `atmin review` check. Maintainers can comment
77
- `/atmin review` to rerun. Use a dedicated host user; the pilot worker is not a
78
- sandbox for executing repository code or an isolation boundary for many tenants.
119
+ summary per PR, and publishes an `atmin review` check. Each review is a claim-pipeline run
120
+ (see `review` above); point `profile` at `profiles/review-luna-openrouter.json` and set
121
+ `OPENROUTER_API_KEY` and, for the Jev rung, `TYPESAFE_API_KEY`. The first review of a PR
122
+ reads the whole change. Each later push is reviewed incrementally: only the commits since
123
+ the last completed review are read for new findings, and that review's findings are
124
+ re-checked against the new head. A force-push or a merge from the target branch gets a full
125
+ review. Automatic reviews pause after five reviewed heads of one PR; comment
126
+ `/atmin review` for a full review, which also restarts the count. Reviews are bounded by
127
+ `maxReviewsPerDay`. Diffs over 512 KB are not reviewed. Maintainers can comment
128
+ `/atmin review` to rerun. Use a dedicated host user. Source review does not execute repository code.
129
+ Optional [isolated checks](docs/isolated-checks.md) run selected commands and
130
+ verify proposed patches on a configured Linux worker. This private pilot is not
131
+ a hardened isolation boundary for many tenants. Each connected repository has its own
132
+ state directory, database and shared copy of its history. A repository over 2 GB on GitHub
133
+ does not connect; no review starts while the state disk has less than `minFreeDiskMb`
134
+ (default 2048) free; a shared copy over 3 GB is wiped before the next capture; and run
135
+ records older than 90 days are deleted, except each PR's latest completed review. Each
136
+ organization's run records and shared copies together are held to `maxInstallationDiskMb`
137
+ (default 5120), checked before each review, so one review can go past it. Over the limit,
138
+ the organization's shared copies are wiped first, since the next review fetches its own
139
+ again; if its run records alone are still over, the review does not start, the PR comment
140
+ says why, and nothing counts against the plan. `/admin` shows each organization's use.
79
141
 
80
- Register an App with repository Contents read, Pull requests write and Checks
81
- write. Subscribe to Pull request, Push and Issue comment events. Install it only
142
+ Register an App with repository Contents read, Issues read, Pull requests write and
143
+ Checks write. Issues read is what makes GitHub offer the Issue comment event, which carries
144
+ `/atmin review`. Subscribe to Pull request, Push, Issue comment and Check run events. Install it only
82
145
  on the intended repository. Set its webhook to your HTTPS proxy's
83
146
  `/webhooks/github`, forwarding to the worker on loopback port 8787.
84
147
 
@@ -100,18 +163,44 @@ and installation IDs in a local configuration:
100
163
  }
101
164
  ```
102
165
 
103
- `trustedChecks` is optional. Use the actual check name and producer App ID you
166
+ `trustedChecks`, `minFreeDiskMb` and `maxInstallationDiskMb` are optional. Use the actual check name and producer App ID you
104
167
  trust; check names must match the target policy's required checks. App 15368 is
105
168
  an illustrative configuration; verify the producer in your repository before
106
- using it. The worker queries GitHub on the exact reviewed head. Only a unique
107
- completed success counts as a pass. Missing, skipped, ambiguous, cancelled or
108
- unavailable checks remain unverified. A check on a different merge commit is
109
- not automatically treated as evidence for the head.
169
+ using it. The worker queries GitHub on the exact reviewed head. CI that runs on
170
+ both push and pull_request leaves one run per event there; a pass needs every run
171
+ of the check from that App completed and successful, and any failed run fails it.
172
+ Missing, skipped, cancelled, still-running or unavailable checks remain unverified. A check on a different merge commit is
173
+ not automatically treated as evidence for the head. The review does not wait for CI: while
174
+ unverified required checks are all that stand between a clean review and a pass, the
175
+ `atmin review` check stays pending (in progress) instead of failing, and a required check
176
+ that never passes keeps it pending.
110
177
 
111
178
  Protect CI workflow changes according to your repository's policy: trusting an
112
- App and check name is not verification of the workflow's code. This alpha reads
113
- CI at publication and explicit reconciliation; it does not subscribe to CI
114
- completion events or poll after publication.
179
+ App and check name is not verification of the workflow's code. Trusted Check run
180
+ creation and completion events refresh the saved report and check, including
181
+ events arriving during publication. Events are only wakeups: results are fetched
182
+ from GitHub again. Stale commits, unrelated checks and duplicate deliveries do
183
+ not start reviews. There is no background polling; use explicit reconciliation
184
+ if GitHub cannot deliver an event.
185
+
186
+ Findings also appear in one commit-bound advisory review per run, with up to 20
187
+ inline comments. Only anchors present in GitHub's diff are attached; the summary
188
+ retains every visible finding. Optional P4 findings follow target-branch policy.
189
+ CI refreshes reuse the batch; an explicit model rerun creates a new review run.
190
+ Unknown publication outcomes require reconciliation and never blindly repeat a
191
+ POST. Historical inline findings retain their original commits and run identity.
192
+
193
+ When a PR closes, the worker records what became of each finding the PR's completed
194
+ reviews published, first publication counted once. `changed` means the flagged line or a
195
+ line next to it was edited or removed by the PR's last head; `unchanged` means those three
196
+ lines are still in the file, ignoring indentation, wherever they moved to. An unrelated
197
+ edit to the same lines also counts as changed, so the changed share is at most the share of
198
+ findings acted on. It also records whether the PR merged, and the thumbs up, thumbs down
199
+ and human replies on the finding's inline comments (each carries a hidden
200
+ `<!-- atmin-finding:... -->` marker; comments posted before 2026-09-30 have none, so their
201
+ reactions are not read). Resolved conversations are not read, since GitHub reports them
202
+ only through GraphQL. A read that fails is recorded as `unknown` with its cause, and one
203
+ log line per closed PR gives the counts. `/admin` shows the totals per repository.
115
204
 
116
205
  ```sh
117
206
  npx atmin-review-github check ./pilot.json
@@ -126,9 +215,60 @@ npx atmin-review-github pause ./pilot.json
126
215
  Reconciliation refreshes the saved report and CI without new inference. Pausing
127
216
  cancels active work. Failed and cancelled starts count toward the rolling daily
128
217
  limit. Stop the service before backups; retain the database and spending receipts.
129
- Provide snapshot retention and disk limits before widening access. Capture
130
- fetches repository history, so large repositories can exceed the pilot's capacity.
131
- No hosted signup, billing, inline comments, or repository execution is included.
218
+ The worker keeps one copy of each connected repository's history in its state directory
219
+ (`source-cache.git`) and fetches only the commits it lacks. Each run's snapshot borrows
220
+ that copy and is deleted when the run ends; the JSON records stay. Disk limits per account
221
+ are still needed before widening access.
222
+ No repository execution is included.
223
+
224
+ With `REVIEW_DASHBOARD_CONFIG` pointing at a JSON file (`origin`, `clientId`, `models`,
225
+ and optionally `operators` and `appSlug`) and `GITHUB_OAUTH_CLIENT_SECRET` set, `serve`
226
+ also hosts the dashboard: the API, and the pages built into `web/dist` by
227
+ `npm run build:web` (pages answer 503 until that build exists). Anyone can install the
228
+ App and sign in with GitHub. Repository administrators connect up to ten repositories
229
+ per installation and pause or configure each one. Each installation is held to a
230
+ monthly plan: 20 free reviews per UTC calendar month by default. A review counts
231
+ once inference starts, failed runs included. Past the limit the PR gets a "review not
232
+ run" comment with the reason and no model call is made. The daily `maxReviewsPerDay`
233
+ cap still applies to every installation together.
234
+
235
+ With `STRIPE_SECRET_KEY` set, reviews past the free ones are paid from prepaid credit. An
236
+ administrator of a connected repository buys credit on the Billing page through Stripe
237
+ Checkout, $10, $25, $50 or $100 in US dollars, and it does not expire. Each review past the
238
+ free ones takes its price from the credit once its cost settles: one row per review in the
239
+ `credit` table, written once, so a later plan change never prices it again. Once the free
240
+ reviews are used, a review starts only while credit is above zero, so the last one can take it
241
+ slightly below and the next purchase covers that. At zero the PR comment says so and links to
242
+ the Billing page; each paid review's comment says what credit is left and warns below $2.
243
+ Without a key nobody can buy credit, and past the free reviews only credit an operator adds
244
+ pays. Reviews that started before 2026-10-01 are never charged.
245
+
246
+ A purchase is credited when the browser comes back from Checkout, once the worker has checked
247
+ with Stripe that this organization's customer paid the amount sold; if the buyer closes the
248
+ page first, the hourly check credits it. Paying saves the card. With auto top-up on, the
249
+ worker charges that card the chosen amount off-session whenever credit falls below $5,
250
+ checked after each review and hourly. A top-up's PaymentIntent is created unconfirmed under
251
+ an idempotency key, recorded in the `payments` table, then confirmed, so an interrupted
252
+ top-up is finished by reading it back, never by charging again. A card that declines, or
253
+ whose bank wants the card holder to approve the charge, stops auto top-up until an admin buys
254
+ credit or turns it on again. Stripe emails receipts to the address entered in Checkout when
255
+ Settings > Emails > Successful payments is on in the Stripe dashboard. The `invoices` table
256
+ of the earlier monthly-invoice release is left as it is.
257
+
258
+ `operators` lists GitHub user IDs, not logins, because a login can be renamed and taken
259
+ by someone else. Operators get `/admin`, which lists every installation of the App (read
260
+ with the App's credentials) with its repositories, reviews this month, model cost,
261
+ billing and margin, and changes a plan: `freeReviews`, `monthlyReviews` (0 turns
262
+ reviews off), `multiplier` and `minimumUsd`. Billing for each review past the free ones
263
+ is the larger of its cost times the multiplier and the minimum. An operator-set plan
264
+ applies as set; past its free reviews it is paid from credit like any other. Operators also
265
+ add credit to an organization, or take it away, with a note the organization does not see.
266
+ On OpenRouter, cost is the amount OpenRouter reports billing for each call, not a rate
267
+ card estimate; on OpenAI directly, it is priced from the call's usage, cache writes
268
+ included. A call whose charge or cache writes are not reported stays unsettled and is left
269
+ out of billing. OpenRouter's fee on credit purchases is not included. With
270
+ `appSlug` set, the dashboard offers the App's install link; set the App's Setup URL to
271
+ the dashboard origin so GitHub returns people there after installing.
132
272
 
133
273
  ## Develop
134
274
 
@@ -142,3 +282,91 @@ Tests use local repositories, fake provider responses and temporary databases;
142
282
  they do not consume model credits or establish model quality. Package verification
143
283
  installs the actual tarball in a clean directory and exercises both CLI entry
144
284
  points and snapshot rendering. Apache-2.0; see LICENSE and NOTICE.
285
+
286
+ ## Rating presets (current source)
287
+
288
+ Set `rating` inside target-branch `.atmin/review.json`:
289
+
290
+ ```json
291
+ {
292
+ "schemaVersion": 1,
293
+ "rubricVersion": "1",
294
+ "includeOptional": true,
295
+ "requiredChecks": ["change-validation"],
296
+ "rating": {
297
+ "preset": "strict-conventions",
298
+ "perfectRequires": { "noP3": true }
299
+ }
300
+ }
301
+ ```
302
+
303
+ Presets: `balanced` (default: fit, simplicity, appropriate verification and required
304
+ checks), `correctness-first` (5/5 for a complete current review without P0–P2), and
305
+ `strict-conventions` (Balanced plus documented rules). Overrides are booleans:
306
+ `codebaseFit`, `simplicity`, `verification`, `documentedConventions`, `passingChecks`,
307
+ `noP3`. A concern in a required criterion caps the score at 4; an assessed criterion
308
+ left unknown makes it unrated. P0/P1 cap at 1 and P2 at 3. A review with no quality
309
+ assessment, which is every claim-pipeline review, is scored by its findings alone, so a
310
+ clean one earns 5/5. Missing required check results are noted and do not withhold the
311
+ score; a failed one caps it at 4. No average, test-count quota,
312
+ or automatic penalty for optional P4 suggestions or unavailable patches.
313
+
314
+ Ratings are subjective and independent from finding severity and GitHub check
315
+ conclusions. An incomplete or stale review cannot be rated. Repository policy is
316
+ captured from the target branch; a PR cannot relax its own rules. These additions
317
+ are available in current source and await the next versioned package release.
318
+
319
+ ## Current review direction
320
+
321
+ The current design centres review on the **claim**: one falsifiable assertion
322
+ about one location, carried through investigation, verification and disposition.
323
+ Read the [claim-lifecycle design](docs/claim-lifecycle-design-2026-09-17.md) for
324
+ the claim schema, the ordered evidence ladder and the eval methodology.
325
+
326
+ The earlier [repository-state direction](docs/repository-state-direction.md)
327
+ stays published for context. Its versioned per-branch artifact is deferred by the
328
+ claim-lifecycle design in favour of a thin human-knowledge file, and one review
329
+ agent with separate investigation and verification phases carries forward. Both
330
+ are proposals, not claims about the current engine.
331
+
332
+ That lifecycle now runs end to end:
333
+
334
+ ```sh
335
+ npx atmin-review claim-review https://github.com/OWNER/REPO/pull/123 \
336
+ --profile ./review-profile.json
337
+ ```
338
+
339
+ A wide pass emits falsifiable claims, a separate pass settles each claim's
340
+ propositions against the frozen revision with none of the first pass's reasoning
341
+ in scope, and the verdict is composed from what survived. The report shows the
342
+ claims that died alongside the findings that lived: emitting widely is only
343
+ trustworthy when the discarding is visible. It accepts a prepared snapshot
344
+ directory in place of a URL, and writes `claims.json` and `verification.json`
345
+ beside the snapshot.
346
+
347
+ Whether a rung earns its place is a measurement, not an assumption:
348
+
349
+ ```sh
350
+ npx atmin-review claim-ablate ./private-review --rung cross_family_llm
351
+ npx atmin-review claim-ablate ./private-review --rung symbolic
352
+ ```
353
+
354
+ That re-verifies a finished run with one rung switched off, replaying recorded
355
+ model answers on both sides, so the difference between the two is that rung and
356
+ nothing else. It spends nothing and changes nothing. The report
357
+ counts what the rung added, what it took away, what it raised and what it called
358
+ into question, because a rung that only removes findings is still earning its
359
+ place when those findings were wrong.
360
+
361
+ Two limits are current, not permanent. The cross-family rung does not run, so no
362
+ claim reaches high confidence through agreement, and the report says so. Rung 1
363
+ knows three assertions — what a declaration contains, what a body contains, and
364
+ whether a symbol is referenced outside a file — each askable of the head or the
365
+ merge base.
366
+
367
+ The specs a v1 implementation targets are the [claim schema](spec/claim-schema.md),
368
+ the [evidence chain](spec/evidence-chain.md) and the
369
+ [verdict policy constraints](spec/verdict-policy.md). The
370
+ [regression harness layout](harness/README.md) and the
371
+ [pre-registered paired benchmark](bench/PLAN.md) describe how it gets measured.
372
+ Ship bar: 70% precision on the 15 Martian development cases.
@@ -0,0 +1,30 @@
1
+ import { type CrossFamilyAnswer, type CrossFamilyRung, type Verification } from './lifecycle.js';
2
+ import type { Revisions } from './symbolic.js';
3
+ import type { Claim } from './claim.js';
4
+ import type { Confidence, Rung, Verdict } from './evidence.js';
5
+ import type { Policy, ReviewVerdict } from './policy.js';
6
+ export declare function recordedRung(log: CrossFamilyAnswer[]): CrossFamilyRung;
7
+ export interface Tally {
8
+ claims: number;
9
+ verdicts: Record<Verdict, number>;
10
+ confidence: Record<Confidence, number>;
11
+ propositions: Record<Rung | 'unsettled', number>;
12
+ suspectChecks: number;
13
+ decision: ReviewVerdict;
14
+ rule: string;
15
+ }
16
+ export declare function tally(verification: Verification): Tally;
17
+ export interface RungContribution {
18
+ rung: Rung;
19
+ without: Tally;
20
+ with: Tally;
21
+ gained: string[];
22
+ lost: string[];
23
+ raised: string[];
24
+ refuted: string[];
25
+ suppressed: string[];
26
+ decisionChanged: boolean;
27
+ }
28
+ export declare function contributionOf(rung: Rung, claims: Claim[], revisions: Revisions, policy: Policy, log: CrossFamilyAnswer[]): RungContribution;
29
+ export declare function renderContribution(contribution: RungContribution): string;
30
+ export declare const contributionOfCrossFamily: (claims: Claim[], revisions: Revisions, policy: Policy, log: CrossFamilyAnswer[]) => RungContribution;
@@ -0,0 +1,90 @@
1
+ import { verifyClaims } from './lifecycle.js';
2
+ // Whether a rung earns its place is a measurement, not an assumption. Verification is
3
+ // deterministic and free once the claims exist, so the same claim set can be verified
4
+ // again with a rung switched off and the difference read directly. The expensive half —
5
+ // emission — is not repeated, and the model answers are replayed rather than re-asked,
6
+ // so the comparison is exact rather than a second sample of a noisy process.
7
+ // Replays what the cross-family rung answered in a recorded run. Questions outside the
8
+ // log return nothing, which is what a rung that cannot reach a proposition looks like.
9
+ // Keyed on the claim and the proposition's text, deliberately not on its revision. The
10
+ // revision is part of the question now, but two propositions of one claim that read
11
+ // identically are the same question written twice, not two questions — and leaving the
12
+ // key alone keeps recorded runs from before the revision was carried replayable exactly,
13
+ // which is what makes a past run a measurement rather than an anecdote.
14
+ export function recordedRung(log) {
15
+ const answers = new Map(log.map(entry => [`${entry.claimId}\u0000${entry.proposition}`, entry.evidence]));
16
+ return { settle: (proposition, claim) => answers.get(`${claim.claimId}\u0000${proposition}`) ?? [] };
17
+ }
18
+ const zero = (keys) => Object.fromEntries(keys.map(key => [key, 0]));
19
+ export function tally(verification) {
20
+ const counts = {
21
+ claims: verification.chains.length,
22
+ verdicts: zero(['confirmed', 'refuted', 'inconclusive', 'withheld']),
23
+ confidence: zero(['low', 'moderate', 'high']),
24
+ propositions: zero(['symbolic', 'ci_output', 'cross_family_llm', 'unsettled']),
25
+ suspectChecks: 0,
26
+ decision: verification.decision.verdict,
27
+ rule: verification.decision.rule,
28
+ };
29
+ for (const chain of verification.chains) {
30
+ counts.verdicts[chain.verdict]++;
31
+ if (chain.verdict === 'confirmed')
32
+ counts.confidence[chain.verifierConfidence]++;
33
+ counts.suspectChecks += chain.suspectChecks.length;
34
+ for (const record of chain.propositions)
35
+ counts.propositions[record.settledBy ?? 'unsettled']++;
36
+ }
37
+ return counts;
38
+ }
39
+ const shipped = (chains) => new Map(chains.filter(chain => chain.verdict === 'confirmed')
40
+ .map(chain => [chain.claimId, chain.verifierConfidence]));
41
+ const RANK = ['low', 'moderate', 'high'];
42
+ // A claim with its checks removed: every proposition survives, and none of them can be
43
+ // settled symbolically. That is what rung 1 being switched off looks like from the
44
+ // claim's side, and it leaves the claim itself untouched so the comparison stays paired.
45
+ const withoutChecks = (claim) => ({ ...claim, evidenceToCheck: claim.evidenceToCheck.map(({ proposition }) => ({ proposition })) });
46
+ // One rung's contribution over one claim set, measured against the same claims verified
47
+ // without it. Every other rung is held fixed — the model answers are replayed on both
48
+ // sides — so the difference is this rung and nothing else.
49
+ export function contributionOf(rung, claims, revisions, policy, log) {
50
+ if (rung === 'ci_output')
51
+ throw new Error('Rung 2 does not run in v1, so there is nothing to measure');
52
+ const replay = recordedRung(log);
53
+ const without = rung === 'symbolic'
54
+ ? verifyClaims(claims.map(withoutChecks), revisions, policy, replay)
55
+ : verifyClaims(claims, revisions, policy);
56
+ const withRung = verifyClaims(claims, revisions, policy, replay);
57
+ const before = shipped(without.chains);
58
+ const after = shipped(withRung.chains);
59
+ const verdictOf = (verification) => new Map(verification.chains.map(chain => [chain.claimId, chain.verdict]));
60
+ const priorVerdict = verdictOf(without);
61
+ return {
62
+ rung,
63
+ without: tally(without), with: tally(withRung),
64
+ gained: [...after.keys()].filter(id => !before.has(id)),
65
+ lost: [...before.keys()].filter(id => !after.has(id)),
66
+ raised: [...after].filter(([id, level]) => before.has(id)
67
+ && RANK.indexOf(level) > RANK.indexOf(before.get(id))).map(([id]) => id),
68
+ refuted: withRung.chains.filter(chain => chain.verdict === 'refuted'
69
+ && priorVerdict.get(chain.claimId) !== 'refuted').map(chain => chain.claimId),
70
+ suppressed: withRung.chains.filter(chain => chain.suspectChecks.length).map(chain => chain.claimId),
71
+ decisionChanged: without.decision.verdict !== withRung.decision.verdict,
72
+ };
73
+ }
74
+ export function renderContribution(contribution) {
75
+ const { without: a, with: b } = contribution;
76
+ const row = (label, left, right) => `| ${label} | ${left} | ${right} |`;
77
+ return ['| | without | with |', '| --- | --- | --- |',
78
+ row('confirmed', a.verdicts.confirmed, b.verdicts.confirmed),
79
+ row('refuted', a.verdicts.refuted, b.verdicts.refuted),
80
+ row('inconclusive', a.verdicts.inconclusive, b.verdicts.inconclusive),
81
+ row('propositions settled symbolically', a.propositions.symbolic, b.propositions.symbolic),
82
+ row('propositions settled by the model', a.propositions.cross_family_llm, b.propositions.cross_family_llm),
83
+ row('propositions unsettled', a.propositions.unsettled, b.propositions.unsettled),
84
+ row('findings at high confidence', a.confidence.high, b.confidence.high),
85
+ row('verdict', a.decision, b.decision), '',
86
+ `Claims that ship only with the rung: ${contribution.gained.length}. Claims it takes away: ${contribution.lost.length}.`,
87
+ `Claims it refuted outright: ${contribution.refuted.length}. Checks it called into question: ${b.suspectChecks}.`,
88
+ contribution.decisionChanged ? 'The rung changed the verdict.' : 'The rung did not change the verdict.', ''].join('\n');
89
+ }
90
+ export const contributionOfCrossFamily = (claims, revisions, policy, log) => contributionOf('cross_family_llm', claims, revisions, policy, log);
@@ -1,4 +1,6 @@
1
1
  import { type Packet, type Result, type Finding } from './contracts.js';
2
+ import { type Rating } from './rating.js';
3
+ export declare function validateFix(finding: Finding): void;
2
4
  export type Freshness = {
3
5
  status: 'current' | 'superseded' | 'unverified';
4
6
  reason: string;
@@ -12,6 +14,7 @@ export type ValidationCheck = {
12
14
  url?: string;
13
15
  };
14
16
  export interface Assessment {
17
+ rating: Rating;
15
18
  outcome: 'Changes needed' | 'Review incomplete' | 'Validation needed' | 'Superseded' | 'Unverified' | 'Suggestions' | 'No issues found';
16
19
  findingsVerdict: 'Changes needed' | 'Review incomplete' | 'Suggestions' | 'No issues found';
17
20
  scope: 'complete' | 'partial' | 'unavailable';
@@ -22,5 +25,6 @@ export interface Assessment {
22
25
  hiddenOptionalCount: number;
23
26
  reasons: string[];
24
27
  }
28
+ export declare function reviewSummary(assessment: Assessment): string;
25
29
  export declare function validateEvidence(packet: Packet, result: Result): void;
26
30
  export declare function assess(packet: Packet, result: Result, freshness?: Freshness, ci?: ValidationCheck[]): Assessment;
@@ -1,5 +1,28 @@
1
- import { parsePacket, parseResult, PRIORITIES } from './contracts.js';
1
+ import { ReviewInputError, parsePacket, parseResult, PRIORITIES } from './contracts.js';
2
+ import { rate } from './rating.js';
3
+ export function validateFix(finding) {
4
+ const fix = finding.fix;
5
+ if (!fix)
6
+ return;
7
+ if (finding.anchor.side !== 'head')
8
+ throw new ReviewInputError('Fixes must replace source in the head version of the finding file');
9
+ if (fix.endLine < fix.startLine || fix.endLine - fix.startLine >= 20
10
+ || fix.original.split('\n').length !== fix.endLine - fix.startLine + 1)
11
+ throw new ReviewInputError('The fix range must contain 1–20 existing source lines');
12
+ if (fix.replacement.split('\n').length > 20)
13
+ throw new ReviewInputError('The replacement must contain at most 20 lines');
14
+ if (fix.original === fix.replacement)
15
+ throw new ReviewInputError('The replacement is identical to the original source; no change was proposed');
16
+ if (fix.replacement.endsWith('\n'))
17
+ throw new ReviewInputError('Omit the terminating newline from replacement; GitHub adds the line ending');
18
+ if ([fix.original, fix.replacement].some(text => /[\u0000-\u0008\u000b-\u001f\u007f\u202a-\u202e\u2066-\u2069]|`{3}/.test(text))) {
19
+ throw new ReviewInputError('Control characters, CRLF and Markdown code fences are unsupported in a fix');
20
+ }
21
+ }
2
22
  export const unverified = () => ({ status: 'unverified', reason: 'Live PR state has not been checked for this rendering.', checkedAt: null });
23
+ export function reviewSummary(assessment) {
24
+ return `${assessment.outcome}. ${assessment.findings.length} finding(s) recorded. Review scope: ${assessment.scope}. Required validation: ${assessment.validation}. ${assessment.rating.score === null ? 'Not rated.' : `Rating: ${assessment.rating.score}/5.`} This is not merge approval.`;
25
+ }
3
26
  function unique(values, label) {
4
27
  if (new Set(values).size !== values.length)
5
28
  throw new Error(`Duplicate ${label}`);
@@ -28,6 +51,20 @@ export function validateEvidence(packet, result) {
28
51
  if (refs.some(id => !evidence.has(id)))
29
52
  throw new Error('Unknown evidence reference');
30
53
  };
54
+ if (result.quality) {
55
+ if (result.status === 'not-started')
56
+ throw new ReviewInputError('A not-started investigation cannot contain a quality assessment');
57
+ for (const criterion of Object.values(result.quality.criteria)) {
58
+ checkRefs(criterion.evidenceIds);
59
+ if (criterion.status !== 'unknown' && (!criterion.evidenceIds.length
60
+ || criterion.evidenceIds.some(id => evidence.get(id)?.provenance !== 'controller-captured'))) {
61
+ throw new ReviewInputError('Quality judgments must cite captured source reads; use unknown when evidence is missing');
62
+ }
63
+ }
64
+ if (result.quality.criteria.documentedConventions.status === 'concern' && !result.quality.conventionRules.length) {
65
+ throw new ReviewInputError('A documented-convention violation must quote an explicit target-branch rule');
66
+ }
67
+ }
31
68
  for (const coverage of result.coverage) {
32
69
  checkRefs(coverage.evidenceIds);
33
70
  if (coverage.status === 'reviewed') {
@@ -45,20 +82,38 @@ export function validateEvidence(packet, result) {
45
82
  }
46
83
  }
47
84
  for (const finding of result.findings) {
85
+ validateFix(finding);
48
86
  checkRefs(finding.evidenceIds);
49
- if (!inventory.has(finding.anchor.path))
50
- throw new Error('Finding must anchor to a changed path; callers belong in supporting evidence');
87
+ // A change can break code it did not touch: a caller left calling a function whose
88
+ // contract changed. Seen live 2026-09-24, seven confirmed P1s on callers that this rule
89
+ // could only report as limitations, under a check that read "no issues". Such a finding
90
+ // anchors to the head revision, where the broken caller is; loadReview checks the line.
91
+ if (!inventory.has(finding.anchor.path) && finding.anchor.side !== 'head')
92
+ throw new Error('A finding outside the changed paths must anchor to the head revision');
51
93
  if ((finding.priority === 'P4') !== (finding.kind === 'improvement'))
52
94
  throw new Error('P4 is an optional improvement; P0–P3 are defects');
53
95
  if (!finding.evidenceIds.some(id => evidence.get(id)?.anchors.some(a => a.path === finding.anchor.path && a.side === finding.anchor.side))) {
54
96
  throw new Error('Finding needs evidence anchored to its changed path and side');
55
97
  }
98
+ if (finding.fix && !finding.evidenceIds.some(id => {
99
+ const item = evidence.get(id);
100
+ return item?.provenance === 'controller-captured' && item.anchors[0]?.path === finding.anchor.path
101
+ && item.anchors[0]?.side === 'head' && item.capture.revision === packet.headSha
102
+ && item.capture.startLine <= finding.fix.startLine && item.capture.endLine >= finding.fix.endLine;
103
+ }))
104
+ throw new ReviewInputError('A fix must cite a captured head source read covering its entire replacement range');
56
105
  }
57
106
  }
58
107
  export function assess(packet, result, freshness = unverified(), ci = []) {
59
108
  validateEvidence(packet, result);
109
+ // Binary files are listed as not reviewed but do not hold the scope open: the model reads
110
+ // text, so a PR adding an icon or a font could never be rated (Lors chose this on
111
+ // 2026-09-28). A change made only of binary files had nothing reviewed, so it stays partial.
112
+ const kinds = new Map(packet.changedFiles.map(file => [file.path, file.kind]));
113
+ const readable = result.coverage.filter(c => kinds.get(c.path) !== 'binary');
60
114
  const scope = result.status === 'not-started' ? 'unavailable'
61
- : result.status === 'completed' && result.coverage.every(c => c.status === 'reviewed') ? 'complete' : 'partial';
115
+ : result.status === 'completed' && readable.every(c => c.status === 'reviewed')
116
+ && (readable.length > 0 || result.coverage.length === 0) ? 'complete' : 'partial';
62
117
  unique(ci.map(c => c.name), 'CI check name');
63
118
  const checks = packet.policy.requiredChecks.map(name => ci.find(c => c.name === name)
64
119
  ?? result.validation.find(c => c.name === name)
@@ -84,5 +139,6 @@ export function assess(packet, result, freshness = unverified(), ci = []) {
84
139
  reasons.push('Investigation is not complete across the captured changed-file inventory.');
85
140
  if (validation === 'failed' || validation === 'missing')
86
141
  reasons.push(`Required validation is ${validation}.`);
87
- return { outcome, findingsVerdict, scope, validation, freshness, validationChecks: checks, findings, hiddenOptionalCount: result.findings.length - findings.length, reasons };
142
+ return { outcome, findingsVerdict, scope, validation, freshness, validationChecks: checks, findings, hiddenOptionalCount: result.findings.length - findings.length, reasons,
143
+ rating: rate(packet, result, { scope, freshness, validation }) };
88
144
  }
@@ -0,0 +1,14 @@
1
+ import { type Revision } from './symbolic.js';
2
+ export declare const MAX_CALLED_CODE_BYTES: number;
3
+ export interface CalledCode {
4
+ symbol: string;
5
+ path: string;
6
+ line: number;
7
+ text: string;
8
+ capped: boolean;
9
+ }
10
+ export declare const TEST: RegExp;
11
+ export declare function calledCode(diff: string, head: Revision): {
12
+ called: CalledCode[];
13
+ omitted: string[];
14
+ };