assertledger 1.0.0 → 1.1.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (93) hide show
  1. package/README.fr.md +3 -3
  2. package/README.md +3 -3
  3. package/SECURITY.md +11 -6
  4. package/conformance/schema-extensions.json +36 -0
  5. package/dist/build-info.d.ts +16 -0
  6. package/dist/build-info.d.ts.map +1 -0
  7. package/dist/build-info.js +18 -0
  8. package/dist/build-info.js.map +1 -0
  9. package/dist/build-info.json +7 -0
  10. package/dist/cli.d.ts.map +1 -1
  11. package/dist/cli.js +117 -19
  12. package/dist/cli.js.map +1 -1
  13. package/dist/contracts/index.d.ts +943 -0
  14. package/dist/contracts/index.d.ts.map +1 -1
  15. package/dist/contracts/index.js +561 -21
  16. package/dist/contracts/index.js.map +1 -1
  17. package/dist/core/index.d.ts +10 -1
  18. package/dist/core/index.d.ts.map +1 -1
  19. package/dist/core/index.js +484 -8
  20. package/dist/core/index.js.map +1 -1
  21. package/dist/diagnostics.d.ts.map +1 -1
  22. package/dist/diagnostics.js +42 -2
  23. package/dist/diagnostics.js.map +1 -1
  24. package/dist/engine/adapters/node-test-runtime.d.ts +17 -0
  25. package/dist/engine/adapters/node-test-runtime.d.ts.map +1 -1
  26. package/dist/engine/adapters/node-test-runtime.js +45 -23
  27. package/dist/engine/adapters/node-test-runtime.js.map +1 -1
  28. package/dist/engine/container.d.ts +54 -0
  29. package/dist/engine/container.d.ts.map +1 -0
  30. package/dist/engine/container.js +464 -0
  31. package/dist/engine/container.js.map +1 -0
  32. package/dist/engine/git-regression.d.ts +12 -2
  33. package/dist/engine/git-regression.d.ts.map +1 -1
  34. package/dist/engine/git-regression.js +63 -16
  35. package/dist/engine/git-regression.js.map +1 -1
  36. package/dist/engine/index.d.ts +7 -1
  37. package/dist/engine/index.d.ts.map +1 -1
  38. package/dist/engine/index.js +300 -74
  39. package/dist/engine/index.js.map +1 -1
  40. package/dist/mcp/index.d.ts.map +1 -1
  41. package/dist/mcp/index.js +53 -1
  42. package/dist/mcp/index.js.map +1 -1
  43. package/dist/proof-planner/carry-over.d.ts +21 -0
  44. package/dist/proof-planner/carry-over.d.ts.map +1 -0
  45. package/dist/proof-planner/carry-over.js +212 -0
  46. package/dist/proof-planner/carry-over.js.map +1 -0
  47. package/dist/proof-planner/index.d.ts +11 -0
  48. package/dist/proof-planner/index.d.ts.map +1 -0
  49. package/dist/proof-planner/index.js +11 -0
  50. package/dist/proof-planner/index.js.map +1 -0
  51. package/dist/proof-planner/model.d.ts +272 -0
  52. package/dist/proof-planner/model.d.ts.map +1 -0
  53. package/dist/proof-planner/model.js +215 -0
  54. package/dist/proof-planner/model.js.map +1 -0
  55. package/dist/proof-planner/plan.d.ts +140 -0
  56. package/dist/proof-planner/plan.d.ts.map +1 -0
  57. package/dist/proof-planner/plan.js +1124 -0
  58. package/dist/proof-planner/plan.js.map +1 -0
  59. package/dist/proof-planner/policy.d.ts +100 -0
  60. package/dist/proof-planner/policy.d.ts.map +1 -0
  61. package/dist/proof-planner/policy.js +407 -0
  62. package/dist/proof-planner/policy.js.map +1 -0
  63. package/dist/proof-planner/render.d.ts +4 -0
  64. package/dist/proof-planner/render.d.ts.map +1 -0
  65. package/dist/proof-planner/render.js +73 -0
  66. package/dist/proof-planner/render.js.map +1 -0
  67. package/dist/sdk/index.d.ts +16 -6
  68. package/dist/sdk/index.d.ts.map +1 -1
  69. package/dist/sdk/index.js +63 -5
  70. package/dist/sdk/index.js.map +1 -1
  71. package/docs/agentic-test-profile.md +5 -0
  72. package/docs/architecture.md +15 -3
  73. package/docs/ci.md +7 -1
  74. package/docs/conformance-v1.md +11 -2
  75. package/docs/container-isolation.md +156 -0
  76. package/docs/evidence-export.md +146 -0
  77. package/docs/git-regression.md +7 -3
  78. package/docs/migration-timeout-discovery-inconclusive.md +126 -0
  79. package/docs/migration-verification-v2.md +55 -0
  80. package/docs/project-intent.md +3 -2
  81. package/docs/proof-model.md +28 -6
  82. package/docs/proof-planner.md +368 -0
  83. package/docs/reference.md +59 -12
  84. package/docs/roadmap.md +12 -4
  85. package/examples/agentic-profile/profile-manifest.mjs +5 -1
  86. package/examples/evidence-export/consumer.mjs +35 -0
  87. package/package.json +5 -5
  88. package/schemas/evidence-export-replay-result.v1.json +49 -0
  89. package/schemas/evidence-export-request.v1.json +681 -0
  90. package/schemas/evidence-export.v1.json +1479 -0
  91. package/schemas/evidence-manifest.v2.json +886 -0
  92. package/schemas/evidence-provider-manifest.v1.json +269 -0
  93. package/schemas/verification-request.v2.json +457 -0
@@ -0,0 +1,368 @@
1
+ # Proof planner (internal module)
2
+
3
+ The proof planner answers one question: **given a change, the claims it reaches, their criticality
4
+ and an explicit assurance policy, which evidence is proportionate?** It lives in
5
+ `src/proof-planner/`, is pure and deterministic, and is not exported from the package entry points.
6
+
7
+ ```text
8
+ impact provider (e.g. a semantic context tool) describes what MAY be affected
9
+ │ ChangeImpact (provider-neutral)
10
+ ▼
11
+ proof planner decides what MUST be proved
12
+ │ AssurancePlan
13
+ ▼
14
+ AssertLedger decides whether required evidence exists and is
15
+ valid: sufficiency, provenance, freshness, identity
16
+ ```
17
+
18
+ The planner never claims that evidence exists. AssertLedger never decides what is proportionate.
19
+ V1 implements the planner only; the AssertLedger-side satisfaction check is future work.
20
+
21
+ ## 1. Audit of the existing model
22
+
23
+ | Concept | Present | Where | What exists / what is missing |
24
+ | --- | --- | --- | --- |
25
+ | Claims | No | `src/contracts/index.ts` (profile `consistency.claim`) | The only `claim` field is a stability label. No claim id, criticality or claim-to-evidence link. The implicit claim of a campaign is "this candidate test detects the declared fault". |
26
+ | Evidence | Yes | `EvidenceObservation`, gates, `EvidenceManifest` v1/v2, evidence export | One kind only: repeated test executions across REFERENCE/TARGET/NEUTRAL worlds plus candidate-free controls, scoped to one campaign. Typecheck, CI jobs, reviews or corpora are not AssertLedger evidence. |
27
+ | Provenance | Partial | world `provenance` string, `assertledger-git-regression/1`, provider `sourceRevision` | Declared, unauthenticated (`authenticity: UNAUTHENTICATED`). Git commit/tree only inside a world provenance string. |
28
+ | Freshness | No | export `cost.execution.freshness: "UNKNOWN"`, capability `EXECUTION_FRESHNESS: UNSUPPORTED` | No timestamp, expiry or validity window; the core forbids the clock. The only usable proxy is digest identity. |
29
+ | Candidate identity | Partial | `repositoryDigest` (content digest of the snapshot), git-regression revisions | "Candidate" means a candidate **test**, not a candidate change. No first-class commit or tree of the change under proof. |
30
+ | Invalidation | No | replay checks integrity only | Nothing marks evidence stale when the repository changes. Outputs are append-only. |
31
+ | Confidence | Partial | export `confidence.level: "REPLAY_CONSISTENT_UNAUTHENTICATED"` | Constant, qualitative; `established` / `notEstablished` token lists are a good residual-uncertainty vocabulary. |
32
+ | Verdicts | Yes | `VERIFIED/REJECTED/INCONCLUSIVE/ENGINE_ERROR`, candidate statuses, gate `PASSED/FAILED/NOT_RUN` | All verdicts are about test-evidence quality for one campaign; none says "this change is sufficiently proved". |
33
+ | Infra vs product | Partial | outcome taxonomy; `TIMEOUT`/`INFRA_ERROR` are inconclusive | Observation-level only. Nothing distinguishes a change to the proof infrastructure from a change to the product. |
34
+
35
+ ### Where sufficiency is implicit or global
36
+
37
+ These rules are correct for their own purpose and are **not** weakened by the planner:
38
+
39
+ - `pnpm check` is the single completion gate for every change, and CI runs it on the full matrix
40
+ without path filters.
41
+ - One invalid control makes a whole campaign `INCONCLUSIVE`; one timeout attempt among passes makes a
42
+ candidate `UNSTABLE`; every required target must be killed.
43
+ - Consumer `obligations` are `COVERED` only when all are `EXECUTED`, and `EXECUTED` means "observed at
44
+ least once", not "satisfied".
45
+ - The init lock digests every test, lockfile and CI file; one changed byte makes the runtime doctor
46
+ report `RUNTIME_CONFIGURATION_STALE`, whatever the change touches.
47
+ - The decision digest folds product identity (repository, worlds) and proof-infrastructure identity
48
+ (engine, adapter, budgets, backend) into one value: a budget change re-identifies the evidence.
49
+
50
+ What was missing is upstream of these gates: a planner that decides **whether** a given gate is
51
+ proportionate for a change, and **which** earlier evidence stays admissible after a follow-up commit.
52
+
53
+ ## 2. Model
54
+
55
+ ### Inputs
56
+
57
+ - `ChangeImpact`: the revision (git object id or `sha256:` digest), its merge-base `baseline`, an
58
+ optional `runtimeTreeDigest`, and the surfaces changed since the baseline (cumulative). Each
59
+ surface has a role (`product`, `execution-context`, `evaluation`, `test`, `proof-infrastructure`,
60
+ `documentation`), a reach (`direct`, `transitive`), a runtime (`none`, `build`,
61
+ `offline-analysis`, `live`), boundaries (protocol, persistence, replay, analysis, decision,
62
+ admission, security, public-api, benchmark, holdout), a declared behavior change (`none`,
63
+ `suspected`, `fix`, `feature`), coverage, environment sensitivity, an optional typed content
64
+ digest, and for test and proof-infrastructure surfaces the surfaces they exercise. The analysis
65
+ states its method (`static-graph` or `declared`), completeness, uncertainty, localized unknowns
66
+ and omitted surfaces.
67
+ - `AssuranceClaim[]`: the project's claim registry or the claims a change makes. Pass the registry,
68
+ not a hand-picked subset: untouched claims are filtered by the planner, not by the caller.
69
+ - `ProofSignal[]`: what proof runs revealed, for one revision each. A signal may also name the
70
+ baseline and runtime tree it was observed on: an unresolved signal (a product signal, or a failure
71
+ that may exercise the change) then still counts on a later revision whose product is
72
+ byte-identical, so a proof-only commit cannot erase a regression.
73
+ - `AssurancePolicy`: versioned data, digested into the plan (default `DEFAULT_ASSURANCE_POLICY`).
74
+
75
+ ### Facts, not a level table
76
+
77
+ Code derives a closed vocabulary of **facts** (`product:fix`, `boundary:replay:behavior`,
78
+ `runtime:live:behavior`, `claim:critical:direct`, `impact:unbounded`, `observed:regression`, …).
79
+ The policy is data that maps facts to:
80
+
81
+ - **floors**: the minimal level a fact implies;
82
+ - **raises**: one step up, once per subject, capped at P4 (a raise never creates P5);
83
+ - **triggers**: evidence a fact requires or recommends;
84
+ - **status**: facts that put the plan on `HOLD` or `BLOCKED`.
85
+
86
+ Levels are computed **per subject** (product, evaluation, proof infrastructure, documentation) and
87
+ the plan level is their maximum. A fact may concern several subjects. A level baseline (P1 static
88
+ checks and affected tests, P2 targeted regression and revision identity, P3 targeted integration and
89
+ production-path test, P4 independent review, P5 preregistration and provenance) only applies to the
90
+ subjects that reached that level, and only to the impact surfaces of that subject: a holdout change
91
+ at P5 does not drag product ceremony in, and criticality deepens proof without widening it. Breadth
92
+ (full suite, full corpus, multi-environment) comes only from impact facts.
93
+
94
+ Subjects follow what the evidence is about, not only the role of the surface:
95
+
96
+ - benchmark and holdout boundaries always concern the evaluation subject, whatever role carries them;
97
+ - security, admission and decision boundaries keep their weight on a changed gate or oracle
98
+ (proof-infrastructure surfaces, and test surfaces whose oracle changed), under the
99
+ proof-infrastructure subject;
100
+ - within the impact, a claim is reached through the product only by product, execution-context or
101
+ evaluation surfaces (outside an unbounded impact, its surfaces are treated as reached product).
102
+ A high or critical claim whose test or gate changed (the test lists a claim surface in
103
+ `exercises`, or the claim lists the test) gets `claim:<criticality>:oracle` under the
104
+ proof-infrastructure subject: an `ORACLE_WITNESS` (the changed oracle still fails where the
105
+ violation is present) and, for a critical claim, an independent review, but no product
106
+ requalification;
107
+ - a documentation claim whose documented surfaces change behavior requires the documentation check;
108
+ otherwise it is unaffected.
109
+
110
+ | Level | Typical source facts |
111
+ | --- | --- |
112
+ | P0 | documentation only |
113
+ | P1 | refactor; changed tests that no product evidence runs; proof infrastructure; evaluation touched |
114
+ | P2 | product behavior change, new behavior, execution context, compatibility boundary touched; oracle of a high claim changed |
115
+ | P3 | behavior across protocol, persistence, replay, analysis or public API; live surface touched; security boundary touched; high claim reached directly; oracle of a critical claim changed; unbounded impact; costly reversal |
116
+ | P4 | live behavior; decision, admission or security behavior; critical claim reached directly; system-scope claim; irreversible change; observed corpus divergence or live runtime |
117
+ | P5 | empirical claim; benchmark or holdout behavior |
118
+
119
+ ### Impact bound and exemptions
120
+
121
+ | Bound | When | Effect on heavy evidence |
122
+ | --- | --- | --- |
123
+ | `no-product-runtime` | a complete, static, low-uncertainty analysis finds no product, execution-context or evaluation surface | may be `NOT REQUIRED` |
124
+ | `bounded-confident` | the same analysis, with runtime surfaces | may be `NOT REQUIRED`; untouched claims are unaffected |
125
+ | `bounded-uncertain` | declared analysis, partial with named unknowns, non-low uncertainty or localized unknowns | computed against the **worst case** of the enumerated surfaces; what the worst case needs becomes `RECOMMENDED`, the rest `NOT REQUIRED`; a high or critical claim outside the impact earns a recommended invariant check, never a floor |
126
+ | `unbounded` | unknown completeness, unnamed unknowns, omitted surfaces, execution context changed, unknowns touching live or high-critical surfaces, observed unknown dependency or unplanned impact, whatever the enumerated surfaces are | full test suite `REQUIRED`; claims outside the impact are treated as reached transitively; everything not selected is `UNDETERMINED`, never `NOT REQUIRED` |
127
+
128
+ The bound is checked before the roles: an incomplete analysis of what looks like a proof-only change
129
+ is unbounded, because the omitted part may be product. An unbounded impact is treated as reaching
130
+ the product. The worst case keeps the declared analysis, so it is never more trusting than the plan;
131
+ if the worst case is itself unbounded, nothing can be exempted.
132
+
133
+ Size is not uncertainty: a large, enumerated transitive set bounds the targeted evidence's scope,
134
+ it does not trigger a global suite. A declared impact recommends the full suite, because surfaces
135
+ it omits are not bounded.
136
+
137
+ Every evidence kind of the catalog ends in exactly one status: required, recommended, not required
138
+ or undetermined. Each `NOT REQUIRED` entry carries its activation condition (the triggers and level
139
+ baselines that would require it, with current values) and states whether it was evaluated against
140
+ the actual change or the worst case.
141
+
142
+ ### Product evidence versus proof-infrastructure evidence
143
+
144
+ Product signals (`UNEXPECTED_BEHAVIOR`, `UNPLANNED_IMPACT`, `UNKNOWN_DEPENDENCY`,
145
+ `LIVE_RUNTIME_TOUCHED`, `CORPUS_DIVERGENCE`, `WITNESS_NOT_CAUSAL`, `REGRESSION`) become facts.
146
+ A regression blocks; unexpected behavior and missing causality hold the plan and raise it. A
147
+ regression, an unexpected behavior or a corpus divergence that reproduces on the baseline is a
148
+ pre-existing defect: it does not block or escalate, but the reproduction becomes required evidence
149
+ (`FAILURE_ATTRIBUTION`). An unplanned impact reopens the bound unless the impact already names the
150
+ observed surfaces; a live-runtime observation escalates unless every observed surface is already
151
+ declared live.
152
+
153
+ Infrastructure signals (`TIMEOUT`, `ENVIRONMENT_FAILURE`, `TOOLING_FAILURE`) are attributed in a
154
+ fixed order, after the bound is known:
155
+
156
+ 1. `exercises` is recomputed: if the failing job's surfaces are changed runtime surfaces, or changed
157
+ proof surfaces that exercise one, it is `yes`, whatever was declared. A changed test that
158
+ exercises nothing changed stays proof evidence, so a repaired ceiling can still be attributed
159
+ when it flakes. A declared `no` is only trusted under a confident bound and when the failure
160
+ names its surfaces.
161
+ 2. A job that does not exercise the change is attributed to the proof infrastructure by any
162
+ admissible basis (baseline reproduction, pass on the same revision, outside impact, reported
163
+ infrastructure error, declared environment factor).
164
+ 3. A job that may exercise the change is attributed only by `REPRODUCES_ON_BASELINE`. A retry that
165
+ passes proves nondeterminism, not innocence: a slower code path can time out once and pass once.
166
+ An attribution that rests on the reproduction alone requires `FAILURE_ATTRIBUTION`.
167
+ 4. Otherwise the failure is unattributed: the plan holds and requires `FAILURE_ATTRIBUTION`, but the
168
+ level does **not** rise automatically.
169
+
170
+ An attributed infrastructure failure never changes the level, the status or the evidence required
171
+ on the changed surfaces; it adds `AFFECTED_JOB_RERUN` on the failing job, and `FAILURE_ATTRIBUTION`
172
+ there too when a reproduction on the baseline is its only basis. An unattributed one adds
173
+ `FAILURE_ATTRIBUTION` on the failing job; it widens the product evidence only to a failing job that
174
+ is itself a product surface of the impact, never to a test, a harness or a job outside the impact.
175
+ A proof-infrastructure surface without an `exercises` list is taken not to exercise the change when
176
+ its failure is classified, so an outside-impact attribution declared for it stands under a
177
+ confident bound.
178
+
179
+ ### Evidence bindings and carry-over
180
+
181
+ Each kind declares what it stays valid for:
182
+
183
+ - `exact-revision`: static checks and revision identity (cheap, re-run on every revision);
184
+ - `surface-content`: valid per scoped surface while that surface, the proof surfaces that exercise
185
+ it (a test without an `exercises` list exercises everything; a proof-infrastructure surface
186
+ counts where it names the surface, and a directly changed one whose behavior change is not
187
+ declared `none` and that names nothing is reported as `proof.exercises-unknown`), the baseline and the runtime tree
188
+ keep their digests. Only documentation-only evidence about a surface both plans call
189
+ documentation ignores the runtime tree: a gate delta
190
+ review is redone when the code its budget measures changes, and a documentation claim scoped to a
191
+ runtime surface is checked again when the behavior it describes may have changed. Without a
192
+ scope, it is revision-wide;
193
+ - `runtime-tree`: full suite, full corpus, multi-environment, live shadow, system requalification,
194
+ benchmark and holdout runs: valid while the baseline and runtime tree digest are unchanged.
195
+
196
+ `carryOverEvidence(previous, next)` lists what may be reused and what must be produced again. Reuse
197
+ is keyed on the kind's semantic digest (id, weight, binding, subjects, verification,
198
+ `semanticsVersion`), never on its wording and never on the whole policy digest. Missing digests, a
199
+ changed baseline (rebase) or a changed runtime tree force re-production of every kind that depends
200
+ on them, and evidence never crosses to another revision on the surfaces that a signal unresolved in
201
+ either plan may concern, read in both plans: the surfaces it names and what a named test or proof
202
+ surface lists as exercised, or everything when it names no surface, a surface either impact does not
203
+ know, an execution context, or a proof surface that does not list what it exercises. A signal the
204
+ next plan observes counts too, because it contradicts evidence produced before it. Within one
205
+ revision, only an observation the previous plan did not already have counts, compared by what was
206
+ observed (id, revision, signal, classification, exercises, attribution basis and surfaces), not by
207
+ id: evidence produced beside an observation answers it. Evidence that answers observations, such as
208
+ a failure attribution or a job rerun, is reused only for the observations it was produced for, each
209
+ scoped by the surfaces it names; exact-revision evidence is reused within its revision whatever was
210
+ observed. Otherwise, a fact that names a signal its plan does not list, which only a stored or
211
+ foreign plan can carry, blocks the reuse of the evidence behind it. A block lifts when the previous
212
+ plan is planned again; the next plan does not revise its classification. A failure attributed to the
213
+ proof infrastructure, or unattributed on a job that a confidently bounded impact places outside the
214
+ change, concerns no product evidence. AssertLedger must still verify that reused evidence exists and
215
+ carries the listed digests.
216
+
217
+ ### Escalations
218
+
219
+ Each plan precomputes, by replanning with a canonical hypothetical signal, what every product
220
+ signal and an attributed or unattributed infrastructure failure would do to its level, status and
221
+ required evidence. The tests pin every row of the bounded replay fix with values derived from the
222
+ policy by hand, and check the product rows for consistency with a replan through the public API
223
+ (the same planner, so that check alone would not catch a planner error).
224
+
225
+ ## 3. Examples
226
+
227
+ Produced by the V1 default policy; the tests in `tests/proof-planner.test.ts` assert these
228
+ decisions.
229
+
230
+ | Scenario | Level | Required | Heavy evidence |
231
+ | --- | --- | --- | --- |
232
+ | A documentation only | P0 | documentation check | all 11 broad or ceremonial kinds not required |
233
+ | B local refactor | P1 | static checks, affected tests (+ characterization tests if untested) | all not required |
234
+ | C local bugfix | P2 | causal witness, affected tests, targeted regression, static checks, revision identity | all not required |
235
+ | D bounded replay fix | P3 | C + targeted integration, production-path test, targeted corpus, documentation check | all not required, including full corpus and system requalification |
236
+ | D, declared impact | P3 | D | full test suite, characterization tests and an invariant check for the untouched critical claim recommended; the rest not required against the worst case |
237
+ | D, partial analysis | P3 | D + full test suite, invariant check (the untouched critical claim is treated as reached) | full corpus and independent review recommended; everything else undetermined |
238
+ | E live decoder fix | P4 | C + boundary compatibility, targeted integration, production-path test, live shadow, rollback plan, full corpus, independent review | benchmark, holdout, full suite, multi-environment, system requalification not required |
239
+ | E tactical decision | P4 | acceptance test, affected tests, targeted regression and integration, production-path test, live shadow, rollback plan, full corpus, system requalification, independent review, static checks, revision identity | benchmark, holdout, full suite, multi-environment not required |
240
+ | F holdout evaluator + empirical claim | P5 | preregistration, provenance, benchmark protocol, holdout evaluation, contamination check, independent review, affected tests, static checks, revision identity | live shadow, full suite, full corpus, system requalification not required |
241
+ | G timeout ceiling repair | P1 | gate delta review, affected job rerun, static checks | all not required; no functional requalification |
242
+ | G, the repaired test checks a critical claim | P3 (proof infrastructure) | G + oracle witness, independent review of the test, revision identity | no product requalification: every requirement is scoped to the changed test or revision-wide |
243
+
244
+ Rendered plan for the bounded replay fix (abridged):
245
+
246
+ ```text
247
+ AssurancePlan P3 PROVE - revision 1111… (baseline 0000…)
248
+ policy assertledger.default-assurance@1.0.0; impact bounded-confident; product P3, documentation P0
249
+
250
+ WHY
251
+ - impact: bounded-confident: static analysis, complete, low uncertainty
252
+ - claim: live-decoder-pins-unchanged: none of its runtime surfaces is in an impact bounded with confidence
253
+ - floor P3 boundary:replay:behavior: channel/replay-fold, channel/replay-safe-keys: replay behavior may change
254
+ - floor P2 product:behavior-change: channel/replay-safe-keys: product behavior may change
255
+
256
+ REQUIRED
257
+ - CAUSAL_WITNESS (surface-content) [channel/replay-safe-keys] <- product:fix
258
+ - TARGETED_CORPUS (surface-content) [channel/replay-fold, channel/replay-safe-keys] <- boundary:analysis:behavior, boundary:replay:behavior
259
+ - PRODUCTION_PATH_TEST, TARGETED_INTEGRATION, TARGETED_REGRESSION, AFFECTED_TESTS, REVISION_IDENTITY, STATIC_CHECKS, DOCUMENTATION_CHECK
260
+
261
+ NOT REQUIRED
262
+ - FULL_CORPUS: Not required: the policy asks for FULL_CORPUS when one of [boundary:decision:behavior,
263
+ impact:unbounded, observed:corpus-divergence, observed:live-runtime, runtime:live:behavior] holds;
264
+ none holds for this change, whose impact is bounded with confidence.
265
+ - SYSTEM_REQUALIFICATION, INDEPENDENT_REVIEW, FULL_TEST_SUITE, BENCHMARK_PROTOCOL, HOLDOUT_EVALUATION, …
266
+
267
+ ESCALATE IF
268
+ - CORPUS_DIVERGENCE (product): Evidence about the product: level moves P3 -> P4, status PROVE, adds FAILURE_ATTRIBUTION, FULL_CORPUS, INDEPENDENT_REVIEW.
269
+ - REGRESSION (product): Evidence about the product: level stays P3, status BLOCKED, adds nothing.
270
+ - TIMEOUT (infrastructure-attributed): Evidence about the proof infrastructure: level stays P3, status PROVE, adds AFFECTED_JOB_RERUN, FAILURE_ATTRIBUTION.
271
+ - TIMEOUT (infrastructure-unattributed): Unattributed failure: level stays P3, status HOLD, adds FAILURE_ATTRIBUTION.
272
+ - …
273
+ ```
274
+
275
+ ## 4. Before and after: a localized fix with two peripheral timeouts
276
+
277
+ Abstract reproduction of an observed consumer pull request: a three-line fix in a replay-only
278
+ adapter, legitimate targeted proof, then two wall-clock ceilings in untouched packages that timed
279
+ out one after the other (a property at 30.4 s for 30 s that had passed in 12.2 s on the same tree,
280
+ then a performance guard at 2.78 s for 1.5 s whose literal ignored the declared CI latency factor).
281
+ Observed cost before: 3 candidate commits, 5 CI runs, about 6 fresh audits; each ceiling repair
282
+ created a new commit that invalidated the whole aggregate proof although the product claims never
283
+ changed.
284
+
285
+ The test suite replays the sequence with the default policy and with a modeled revision-bound
286
+ global policy (every kind bound to the exact revision, full suite and independent audit at P3):
287
+
288
+ | Step | Before (revision-bound global model) | After (default policy) |
289
+ | --- | --- | --- |
290
+ | r1, replay fix | P3, 12 required kinds | P3, 10 required kinds (full suite and audit not required) |
291
+ | timeout 1 (outside impact, passed on the same tree) | proof infrastructure | proof infrastructure; level, status and product evidence unchanged |
292
+ | r1 → r2, first ceiling routed | 13 kinds re-produced | 4 kinds: static checks, revision identity, gate delta review on the ceiling, job reruns on the ceiling and on the job of timeout 2 |
293
+ | timeout 2 (outside impact, environment factor) | proof infrastructure | proof infrastructure |
294
+ | r2 → r3, second ceiling routed | 13 kinds re-produced | 4 kinds, on the new ceiling only; the first ceiling's review is reused |
295
+
296
+ The planner stays conservative on the same path: a corpus divergence moves the plan to P4 with the
297
+ full corpus; a timeout on a job that exercises the replay surface holds the plan unless it
298
+ reproduces on the baseline; a rebase, a changed runtime tree, a missing digest, a test without an
299
+ `exercises` list, a harness that names the replay surface, or a product digest changed by a
300
+ so-called infrastructure commit forces the product evidence to be produced again; and a regression
301
+ left unresolved on r1 blocks the reuse of the evidence on its surfaces. Each of these is a test.
302
+
303
+ ## 5. Mapping to neighbours
304
+
305
+ **Impact provider.** Any provider can fill `ChangeImpact`. From semctx, for example:
306
+ `changedFiles`/`changedSymbols` give direct surfaces; `impactedConsumers` and `control_impact`
307
+ paths give transitive ones; `impactedInvariants`/`impactedContracts` and change-contract
308
+ `preserves` give claims (criticality from `criticalInvariantTags`); `unknowns` and the
309
+ `analysis_scope_incomplete` / `index_binding_stale` findings give completeness and unknowns;
310
+ `recommendedTests` only feed test surfaces, never required evidence. semctx reports neither a
311
+ candidate commit nor content digests through MCP: the caller supplies `revision`, `baseline` and
312
+ digests. Such an adapter belongs outside `src/` (a test forbids that name in source files).
313
+
314
+ **AssertLedger.** Today AssertLedger can verify `CAUSAL_WITNESS` and `ACCEPTANCE_TEST` as a
315
+ qualification campaign (TARGET = baseline, REFERENCE = revision, NEUTRAL = justified variation) with
316
+ replay-valid manifests. Other kinds (static checks, CI jobs, reviews, corpora, preregistration) have
317
+ no AssertLedger evidence type yet; kinds marked `attested` can only be recorded, not executed.
318
+
319
+ ## 6. Limits and risks
320
+
321
+ - Inputs are trusted declarations. A caller can mislabel a live surface as offline or omit a surface;
322
+ the planner limits the damage (declared analysis never yields confident exemptions, contradicted
323
+ `exercises` values are overridden, incomplete analyses are unbounded whatever they enumerate,
324
+ untouched critical claims come back as recommendations under uncertainty and as requirements when
325
+ unbounded) but cannot detect a consistent lie. A policy-owned surface classifier is not in V1.
326
+ - Claims come from the caller. Pass the full registry; V1 has no registry of its own. A changed proof
327
+ surface without an `exercises` list only reaches the claims that list it; the plan reports it as
328
+ residual uncertainty (`proof.exercises-unknown`).
329
+ - The default policy is a first calibration, not a measured optimum. A caller-supplied policy must be
330
+ at least as strict as the default: every kind (verification, binding, subjects), baseline, floor,
331
+ trigger, raise and status rule of the default must still hold with at least the same strength, or
332
+ parsing fails with `PROOF_PLANNER_POLICY_BELOW_MINIMUM`. The tests pin the default by digest and
333
+ check its non-negotiable rules against an independent list. A project that wants a looser policy
334
+ must fork the default; that is deliberate in V1.
335
+ - Signals carried across revisions need the caller to report the baseline and runtime tree they were
336
+ observed on. Without them, the carry-over still refuses reuse on the surfaces an unresolved signal
337
+ may concern, but the next plan does not hold or block by itself. An unattributed failure on a job
338
+ that a confidently bounded impact places outside the change holds its own revision only: it is
339
+ never carried and blocks no reuse, so the next revision must observe that job again.
340
+ - Freshness is identity-based (digests, baseline, runtime tree), not time-based: the core forbids a
341
+ clock. Time-bound validity windows remain an AssertLedger-side concern.
342
+ - The satisfaction check (does evidence exist for each requirement, bound to the listed digests) is
343
+ not implemented; the plan is advisory until it is.
344
+ - Attribution bases are declared by whoever reports the signal. A pre-existing product defect, and
345
+ an infrastructure attribution that rests on `REPRODUCES_ON_BASELINE` alone, require
346
+ `FAILURE_ATTRIBUTION`, which AssertLedger should eventually verify; the other bases are trusted
347
+ as declared, within the limits above.
348
+ - The module is compiled into `dist/proof-planner/` but not reachable through the package exports;
349
+ its types and digests are not a public contract yet.
350
+
351
+ ## 7. When should it become a separate tool?
352
+
353
+ Not now. Extract it only when several of these hold, with evidence:
354
+
355
+ 1. **Used without AssertLedger**: at least one consumer plans assurance without producing or
356
+ verifying AssertLedger evidence (for example a CI router or a review bot).
357
+ 2. **Own policy and configuration lifecycle**: projects version their assurance policy independently
358
+ of AssertLedger releases, with their own compatibility promises.
359
+ 3. **Several consumers**: two or more independent harnesses or repositories call it, so its release
360
+ cadence conflicts with AssertLedger's.
361
+ 4. **Autonomous data model**: `ChangeImpact`, `AssurancePlan` and the policy need published JSON
362
+ schemas and conformance fixtures of their own.
363
+ 5. **Significant algorithms**: work beyond V1's fact derivation, such as learned calibration,
364
+ cross-revision planning or cost models, would bloat AssertLedger's evidence core.
365
+ 6. **Stable API**: the input and output shapes survive a few real projects without breaking changes.
366
+
367
+ The module is built for that move: it imports only `zod`, its own files and `sha256Canonical` from
368
+ the public `./core` API, and a test enforces that boundary.
package/docs/reference.md CHANGED
@@ -18,7 +18,11 @@ assertledger audit . --json
18
18
  assertledger analyze . --json
19
19
  assertledger schema verification-request --json
20
20
  assertledger verify assertledger.request.json --allow-unsafe-execution --json
21
+ assertledger verify assertledger.container-request.json --container-runtime '["docker"]' --json
21
22
  assertledger replay assertledger.manifest.json --json
23
+ assertledger provider --json
24
+ assertledger export assertledger.export-request.json --json
25
+ assertledger export-replay assertledger.evidence-export.json --json
22
26
  assertledger profile assertledger.profile-request.json --json
23
27
  assertledger profile-replay assertledger.profile-report.json --json
24
28
  assertledger profile-v2 assertledger.profile-v2-request.json --json
@@ -45,15 +49,35 @@ directory or in a colocated `*.test.*`/`*.spec.*` source file. Comments, string
45
49
  documentation-like config filenames do not count; no evidence produces an empty list. Detection is
46
50
  intentionally incomplete: an unrecognized manifest layout or test-file convention yields no claim.
47
51
 
48
- `verify`, `replay`, `profile`, `profile-replay`, `profile-v2`, `profile-v2-replay`, `benchmark`, and
52
+ `verify`, `replay`, `export`, `export-replay`, `profile`, `profile-replay`, `profile-v2`,
53
+ `profile-v2-replay`, `benchmark`, and
49
54
  `benchmark-replay`, `benchmark-acquire`, and `benchmark-acquire-replay` also accept `-`
50
55
  or an omitted file argument and
51
56
  then read JSON from stdin. The CLI rejects file and stdin JSON inputs larger than 16 MiB. JSON
52
57
  results go to stdout. Diagnostics go to stderr. `assertledger mcp` reserves stdout for JSON-RPC.
53
58
 
54
59
  `--allow-unsafe-execution` is an external authorization signal. The CLI requires it for every
55
- campaign and sets the request's local acknowledgement before validation. The flag does not create a
56
- sandbox.
60
+ trusted-local campaign and sets the request's local acknowledgement before validation. The flag does
61
+ not create a sandbox.
62
+
63
+ A v2 request with `container` isolation runs without that flag: each execution uses a fresh
64
+ container from a digest-pinned local image through the operator's `--container-runtime` JSON argv,
65
+ which defaults to `["docker"]`. `check` selects the same backend with `--container-image`. Combining
66
+ the two modes fails with `ISOLATION_MODE_CONFLICT`. See [container isolation](container-isolation.md).
67
+
68
+ `profile` exits with `0` for `QUALIFIED`, `2` for `NOT_QUALIFIED` or `BUDGET_MISSED`, and `3` for
69
+ `INSUFFICIENT_TIMING_EVIDENCE`. A malformed request or a replay-invalid source manifest writes no
70
+ report, prints its reason code such as `AGENTIC_PROFILE_SOURCE_INVALID` to stderr, and exits with
71
+ `4`. `profile-replay` exits with `0` only when every replay rail is valid and with `4` otherwise.
72
+ `profile-v2` has no `NOT_QUALIFIED` status; it uses the same codes for the statuses it shares with
73
+ v1 and `4` for `OBSERVED_BENCHMARK_FAILURE` and `COMPARISON_SCOPE_MISMATCH`, and
74
+ `profile-v2-replay` follows `profile-replay`. Unexpected engine errors exit with `5`.
75
+
76
+ `provider` prints the evidence provider manifest and exits with `0`. `export` exits with `0` after
77
+ writing an [evidence export](evidence-export.md), whatever its detection result; a malformed request
78
+ or a replay-invalid source manifest writes no export, prints `EVIDENCE_EXPORT_REQUEST_INVALID` or
79
+ `EVIDENCE_EXPORT_SOURCE_INVALID` to stderr, and exits with `4`. `export-replay` exits with `0` only
80
+ when every replay rail is valid and with `4` otherwise.
57
81
 
58
82
  Versioned JSON Schemas are published for the
59
83
  [`verification request`](../schemas/verification-request.v1.json),
@@ -78,8 +102,12 @@ Versioned JSON Schemas are published for the
78
102
  [`corpus trust policy`](../schemas/agentic-corpus-trust-policy.v1.json) and paired
79
103
  [`corpus provenance`](../schemas/agentic-corpus-provenance.v1.json), six corpus allocation and
80
104
  two-party [`commitment/reveal`](../schemas/agentic-corpus-allocation-commitment.v1.json) contracts,
81
- plus six pre-declared [`H3 experiment`](../schemas/agentic-corpus-experiment-plan.v1.json) contracts.
82
- The main CLI `schema` command prints the thirty-one facade schemas by name; the corpus evaluator consumes
105
+ plus six pre-declared [`H3 experiment`](../schemas/agentic-corpus-experiment-plan.v1.json) contracts,
106
+ and the [`evidence provider manifest`](../schemas/evidence-provider-manifest.v1.json),
107
+ [`evidence export request`](../schemas/evidence-export-request.v1.json),
108
+ [`evidence export`](../schemas/evidence-export.v1.json), and
109
+ [`evidence export replay result`](../schemas/evidence-export-replay-result.v1.json).
110
+ The main CLI `schema` command prints the thirty-six facade schemas by name; the corpus evaluator consumes
83
111
  the two trust/provenance schemas directly.
84
112
  A complete runnable verification request
85
113
  is available at
@@ -96,8 +124,8 @@ campaign wall-time proxy with an exact scoped warm-total-wall p95 cost basis. It
96
124
  artifacts and commands remain supported.
97
125
 
98
126
  The checked-in [`conformance v1 bundle`](conformance-v1.md) locks autonomous inputs, complete
99
- expected outputs, negative replay witnesses, all published schema bytes, and selected public
100
- digests. `pnpm check` validates this static oracle without regenerating it.
127
+ expected outputs, negative replay witnesses, the v1 schema bytes, and selected public digests; an
128
+ additive lock covers the schemas published after v1. `pnpm check` validates this static oracle without regenerating it.
101
129
 
102
130
  ## TypeScript SDK
103
131
 
@@ -128,6 +156,14 @@ const profile = assertLedger.profile({
128
156
  });
129
157
  const profileIntegrity = assertLedger.replayProfile(profile);
130
158
 
159
+ const provider = assertLedger.providerManifest();
160
+ const evidenceExport = assertLedger.exportEvidence({
161
+ schemaVersion: "1.0.0",
162
+ manifest,
163
+ consumerRequest: null,
164
+ });
165
+ const exportIntegrity = assertLedger.replayEvidenceExport(evidenceExport);
166
+
131
167
  const benchmarkRequest = JSON.parse(await readFile("assertledger.benchmark-request.json", "utf8"));
132
168
  const benchmark = assertLedger.benchmark(benchmarkRequest);
133
169
  const benchmarkIntegrity = assertLedger.replayBenchmark(benchmark);
@@ -160,6 +196,10 @@ surface, for consumers migrating from the prior name.
160
196
  The SDK accepts plain JSON-compatible values and validates them against the same contracts as the
161
197
  CLI. Unlike the CLI and MCP tool, `AssertLedger.verify()` has no separate authorization parameter: the
162
198
  caller must set `isolation.acknowledgedUnsafeExecution` to `true` after applying its own policy.
199
+ `verify()` and `checkGitRegression()` accept only v1 inputs and return v1 manifests.
200
+ `verifyV2(request, { containerRuntime: { command } })` and `checkGitRegressionV2(options)` run
201
+ [container isolation](container-isolation.md) and return v2 manifests; a container request needs no
202
+ acknowledgement, and the runtime argv never comes from the request.
163
203
 
164
204
  `AssertLedger.replay()` reports schema validity, both digest checks, and deterministic
165
205
  decision-semantic validity. Its aggregate `valid` field is true only when all four checks pass. Replay
@@ -233,7 +273,7 @@ server name reported to clients is `assertledger`. Every tool is registered twic
233
273
  `assertledger_*` name and a legacy `testforge_*` name bound to the same handler and the same tool
234
274
  configuration. The schema lookup pair has no single fixed output schema because its selected JSON
235
275
  Schema document varies; every other pair shares the same output-schema object. The default server
236
- exposes twenty-eight read-only tools:
276
+ exposes thirty-four read-only tools:
237
277
 
238
278
  | Preferred tool | Legacy alias | Purpose |
239
279
  | --- | --- | --- |
@@ -247,6 +287,9 @@ exposes twenty-eight read-only tools:
247
287
  | `assertledger_profile_v2` | `testforge_profile_v2` | Derive a benchmark-backed strength and warm-cost profile |
248
288
  | `assertledger_profile_v2_replay` | `testforge_profile_v2_replay` | Replay a self-contained Profile v2 report |
249
289
  | `assertledger_schema` | `testforge_schema` | Return any of the published JSON Schemas |
290
+ | `assertledger_provider` | `testforge_provider` | Describe the evidence provider, its announced capabilities, cost model, and limits |
291
+ | `assertledger_export` | `testforge_export` | Export replay-valid evidence for an external consumer |
292
+ | `assertledger_export_replay` | `testforge_export_replay` | Replay a self-contained evidence export |
250
293
  | `assertledger_replay` | `testforge_replay` | Validate and replay a manifest's schema, digests, and decision semantics |
251
294
  | `assertledger_benchmark_acquire_replay` | `testforge_benchmark_acquire_replay` | Replay acquisition source, artifact, context, digest, and status bindings |
252
295
  | `assertledger_corpus_allocate` | `testforge_corpus_allocate` | Create a deterministic calibration/holdout allocation |
@@ -280,8 +323,9 @@ evidence manifest. The operator's capability is required for both tools.
280
323
  ## Continuous integration
281
324
 
282
325
  Run `pnpm check` on every change. The included GitHub Actions workflow runs this gate on Node.js 22
283
- and 24 on Ubuntu and Windows. A separate matrix installs and exercises the packed artifact on both
284
- operating systems with Node.js 22.15.0 and 24. A CI job that executes campaigns must
326
+ and 24 on Ubuntu, Windows and macOS. A separate matrix installs and exercises the packed artifact
327
+ on Ubuntu and Windows with Node.js 22.15.0 and 24. On Ubuntu, the gate also runs the real-daemon
328
+ [container isolation](container-isolation.md) suite. A CI job that executes campaigns must
285
329
  also treat `trusted-local` as `UNSANDBOXED`: use an isolated runner without secrets or host
286
330
  credentials, and pass `--allow-unsafe-execution` only from reviewed CI configuration.
287
331
 
@@ -290,7 +334,9 @@ credentials, and pass `--allow-unsafe-execution` only from reviewed CI configura
290
334
  - `VERIFIED`: at least one candidate completed all required evidence and was selected.
291
335
  - `REJECTED`: the campaign completed, but no candidate satisfied the policy.
292
336
  - `INCONCLUSIVE`: controls or candidate evidence were incomplete, unstable, timed out, or affected
293
- by infrastructure failure.
337
+ by infrastructure failure. Once its attempts are complete and agree, a completed candidate run
338
+ that reports no attributed candidate test still makes that candidate invalid, whatever else timed
339
+ out; see the [proof model](proof-model.md#candidate-gates).
294
340
  - `ENGINE_ERROR`: the deterministic core could not normalize the supplied evidence safely.
295
341
 
296
342
  Only an attributed `ASSERTION_FAILURE` can kill a target in protocol v1. Compilation errors,
@@ -328,7 +374,8 @@ deterministic core and protocols are framework-independent; `node:test` is the f
328
374
  framework adapter. Other frameworks integrate through the structured-command protocol described in
329
375
  [docs/adapter-protocol.md](adapter-protocol.md).
330
376
 
331
- Planned work is not shipped behavior. Priorities include a real sandbox backend, additional
377
+ Planned work is not shipped behavior. Priorities include VM isolation, container evidence in
378
+ derived reports, additional
332
379
  framework reporters with runtime attribution, signed provenance, cross-runtime conformance
333
380
  fixtures, and more built-in adapters. See [docs/roadmap.md](roadmap.md).
334
381
 
package/docs/roadmap.md CHANGED
@@ -7,7 +7,9 @@ regression-test qualification workflow, acceptance criteria, and evidence bounda
7
7
  ## Implemented locally in v0.1
8
8
 
9
9
  - deterministic repository analysis;
10
- - thirty-four versioned public JSON Schemas, all frozen by conformance v1;
10
+ - forty versioned public JSON Schemas: thirty-four frozen by conformance v1, and four
11
+ evidence-export schemas plus the v2 verification request and evidence manifest locked by the
12
+ additive schema-extension lock;
11
13
  - reference, target, and neutral overlay worlds;
12
14
  - one campaign snapshot and disposable execution workspaces;
13
15
  - built-in `node:test` runtime reporter and structured-command adapter;
@@ -28,11 +30,16 @@ regression-test qualification workflow, acceptance criteria, and evidence bounda
28
30
  planned-process and planned-timeout equality, content-addressed suites, per-source
29
31
  non-inferiority, and a shared frozen candidate universe;
30
32
  - a static conformance-v1 oracle locking canonicalization, decisions, replay witnesses, Profile v1,
31
- Benchmark v1, and all 34 published schema bytes;
33
+ Benchmark v1, and all 34 v1 schema bytes, plus an additive lock for later schemas;
32
34
  - a pinned 24-case, three-source empirical corpus plan with signed admission and holdout rules,
33
35
  including eight receipt-linked TestExplora cases admitted only for curated calibration;
34
36
  - JSON CLI, TypeScript SDK, MCP v2 stdio server, and integration skill;
35
- - `trusted-local`, explicitly recorded as `UNSANDBOXED`.
37
+ - an interoperable evidence export with a provider manifest, separate result, integrity,
38
+ authenticity, environment, confidence, control, and cost sections, and deterministic replay;
39
+ - `trusted-local`, explicitly recorded as `UNSANDBOXED`;
40
+ - a v2 [container isolation backend](container-isolation.md): a fresh digest-pinned Linux container
41
+ per execution, no network or host mounts, bounded resources, and backend facts bound to the
42
+ decision.
36
43
 
37
44
  ## Release 1.0 priority
38
45
 
@@ -55,7 +62,8 @@ scoped deterministic qualification workflow.
55
62
  tranche is curated calibration evidence and cannot satisfy the holdout requirement.
56
63
  2. Add faithful framework-specific phase adapters. The framework-neutral acquisition path is
57
64
  shipped, but the built-in `node:test` reporter cannot attribute all four phases.
58
- 3. Add an isolation backend backed by an independently administered container or VM boundary.
65
+ 3. Extend container isolation to evidence export, profiles, benchmarks and MCP, and add a VM
66
+ boundary for threats that a shared kernel does not contain.
59
67
  4. Add framework reporters beyond `node:test` that derive discovery and attribution from runtime
60
68
  events.
61
69
  5. Publish cross-runtime conformance fixtures for canonicalization, decisions, and both digests.
@@ -24,5 +24,9 @@ if (manifestPath === undefined) {
24
24
  },
25
25
  });
26
26
  process.stdout.write(`${JSON.stringify(report)}\n`);
27
- process.exitCode = report.status === "QUALIFIED" ? 0 : 2;
27
+ // Same mapping as `assertledger profile`.
28
+ process.exitCode =
29
+ { QUALIFIED: 0, NOT_QUALIFIED: 2, BUDGET_MISSED: 2, INSUFFICIENT_TIMING_EVIDENCE: 3 }[
30
+ report.status
31
+ ] ?? 5;
28
32
  }
@@ -0,0 +1,35 @@
1
+ import { readFile } from "node:fs/promises";
2
+ import { AssertLedger } from "assertledger";
3
+
4
+ // A minimal external consumer. It applies its own obligations to an AssertLedger evidence export
5
+ // and writes nothing back: AssertLedger results stay the same whatever it decides.
6
+ const exportPath = process.argv[2];
7
+ if (exportPath === undefined) {
8
+ console.error("Usage: node consumer.mjs <evidence-export.json>");
9
+ process.exitCode = 64;
10
+ } else {
11
+ const evidence = JSON.parse(await readFile(exportPath, "utf8"));
12
+ const replay = new AssertLedger().replayEvidenceExport(evidence);
13
+ const reasons = [];
14
+ let decision;
15
+ if (!replay.valid) {
16
+ decision = "REJECT";
17
+ reasons.push("EXPORT_REPLAY_INVALID");
18
+ } else if (evidence.result.detection !== "OBSERVED") {
19
+ decision = "IGNORE";
20
+ reasons.push(`DETECTION_${evidence.result.detection}`);
21
+ } else {
22
+ const required = ["CONTROL_WITHOUT_CANDIDATE", "REFERENCE_PASS", "NEUTRAL_PASS"];
23
+ for (const control of required) {
24
+ const executed = evidence.controls.executed.find((item) => item.control === control);
25
+ if (executed?.status !== "EXECUTED") reasons.push(`CONTROL_NOT_EXECUTED_${control}`);
26
+ }
27
+ if (evidence.scope.gitRevisions !== "RECORDED") reasons.push("GIT_REVISIONS_NOT_RECORDED");
28
+ if (evidence.authenticity.status !== "ATTESTED") reasons.push("EVIDENCE_UNAUTHENTICATED");
29
+ // This consumer admits unauthenticated local evidence only as advisory.
30
+ decision = reasons.length === 0 ? "ACCEPT" : "DEGRADE";
31
+ }
32
+ process.stdout.write(
33
+ `${JSON.stringify({ decision, reasons, sourceArtifactDigest: evidence.sourceArtifactDigest ?? null })}\n`,
34
+ );
35
+ }