assertledger 1.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (203) hide show
  1. package/CONTRIBUTING.md +31 -0
  2. package/LICENSE +21 -0
  3. package/README.fr.md +236 -0
  4. package/README.md +224 -0
  5. package/SECURITY.md +51 -0
  6. package/benchmarks/agentic-profile/README.md +15 -0
  7. package/benchmarks/agentic-profile/public/README.md +5 -0
  8. package/benchmarks/self-hosted-core/README.md +113 -0
  9. package/benchmarks/self-hosted-core/adapter.mjs +293 -0
  10. package/benchmarks/self-hosted-core/builder.ts +193 -0
  11. package/benchmarks/self-hosted-core/campaign.ts +233 -0
  12. package/benchmarks/self-hosted-core/existing-tests-builder.ts +217 -0
  13. package/benchmarks/self-hosted-core/existing-tests.ts +146 -0
  14. package/benchmarks/self-hosted-core/liveness.test.mjs +8 -0
  15. package/conformance/v1/bundle.json +104 -0
  16. package/conformance/v1/expected/canonical-order-a.json +4 -0
  17. package/conformance/v1/expected/canonical-order-b.json +4 -0
  18. package/conformance/v1/expected/create-benchmark-v1-measured.json +575 -0
  19. package/conformance/v1/expected/create-profile-v1-qualified.json +280 -0
  20. package/conformance/v1/expected/decide-collection-failure-non-kill.json +192 -0
  21. package/conformance/v1/expected/decide-compile-failure-non-kill.json +192 -0
  22. package/conformance/v1/expected/decide-infra-error-non-kill.json +192 -0
  23. package/conformance/v1/expected/decide-no-test-discovered-non-kill.json +192 -0
  24. package/conformance/v1/expected/decide-process-crash-non-kill.json +192 -0
  25. package/conformance/v1/expected/decide-timeout-non-kill.json +192 -0
  26. package/conformance/v1/expected/decide-verified.json +192 -0
  27. package/conformance/v1/expected/replay-benchmark-v1-resealed-summary-forgery.json +12 -0
  28. package/conformance/v1/expected/replay-evidence-raw-tamper.json +6 -0
  29. package/conformance/v1/expected/replay-evidence-resealed-semantic-forgery.json +6 -0
  30. package/conformance/v1/inputs/canonical-order-a.json +8 -0
  31. package/conformance/v1/inputs/canonical-order-b.json +8 -0
  32. package/conformance/v1/inputs/create-benchmark-v1-measured.json +459 -0
  33. package/conformance/v1/inputs/create-profile-v1-qualified.json +228 -0
  34. package/conformance/v1/inputs/decide-collection-failure-non-kill.json +143 -0
  35. package/conformance/v1/inputs/decide-compile-failure-non-kill.json +143 -0
  36. package/conformance/v1/inputs/decide-infra-error-non-kill.json +143 -0
  37. package/conformance/v1/inputs/decide-no-test-discovered-non-kill.json +143 -0
  38. package/conformance/v1/inputs/decide-process-crash-non-kill.json +143 -0
  39. package/conformance/v1/inputs/decide-timeout-non-kill.json +143 -0
  40. package/conformance/v1/inputs/decide-verified.json +143 -0
  41. package/conformance/v1/inputs/replay-benchmark-v1-resealed-summary-forgery.json +575 -0
  42. package/conformance/v1/inputs/replay-evidence-raw-tamper.json +201 -0
  43. package/conformance/v1/inputs/replay-evidence-resealed-semantic-forgery.json +192 -0
  44. package/conformance/v1/schemas/expected-digests.json +175 -0
  45. package/dist/cli.d.ts +9 -0
  46. package/dist/cli.d.ts.map +1 -0
  47. package/dist/cli.js +951 -0
  48. package/dist/cli.js.map +1 -0
  49. package/dist/contracts/diagnostics.d.ts +18 -0
  50. package/dist/contracts/diagnostics.d.ts.map +1 -0
  51. package/dist/contracts/diagnostics.js +13 -0
  52. package/dist/contracts/diagnostics.js.map +1 -0
  53. package/dist/contracts/index.d.ts +3908 -0
  54. package/dist/contracts/index.d.ts.map +1 -0
  55. package/dist/contracts/index.js +2569 -0
  56. package/dist/contracts/index.js.map +1 -0
  57. package/dist/contracts/runtime-doctor.d.ts +107 -0
  58. package/dist/contracts/runtime-doctor.d.ts.map +1 -0
  59. package/dist/contracts/runtime-doctor.js +91 -0
  60. package/dist/contracts/runtime-doctor.js.map +1 -0
  61. package/dist/core/index.d.ts +200 -0
  62. package/dist/core/index.d.ts.map +1 -0
  63. package/dist/core/index.js +2587 -0
  64. package/dist/core/index.js.map +1 -0
  65. package/dist/diagnostics.d.ts +7 -0
  66. package/dist/diagnostics.d.ts.map +1 -0
  67. package/dist/diagnostics.js +252 -0
  68. package/dist/diagnostics.js.map +1 -0
  69. package/dist/engine/adapters/node-test-profile.d.ts +14 -0
  70. package/dist/engine/adapters/node-test-profile.d.ts.map +1 -0
  71. package/dist/engine/adapters/node-test-profile.js +14 -0
  72. package/dist/engine/adapters/node-test-profile.js.map +1 -0
  73. package/dist/engine/adapters/node-test-runtime.d.ts +39 -0
  74. package/dist/engine/adapters/node-test-runtime.d.ts.map +1 -0
  75. package/dist/engine/adapters/node-test-runtime.js +173 -0
  76. package/dist/engine/adapters/node-test-runtime.js.map +1 -0
  77. package/dist/engine/adapters/runtime-facts.d.ts +26 -0
  78. package/dist/engine/adapters/runtime-facts.d.ts.map +1 -0
  79. package/dist/engine/adapters/runtime-facts.js +73 -0
  80. package/dist/engine/adapters/runtime-facts.js.map +1 -0
  81. package/dist/engine/connection.d.ts +22 -0
  82. package/dist/engine/connection.d.ts.map +1 -0
  83. package/dist/engine/connection.js +343 -0
  84. package/dist/engine/connection.js.map +1 -0
  85. package/dist/engine/git-regression.d.ts +25 -0
  86. package/dist/engine/git-regression.d.ts.map +1 -0
  87. package/dist/engine/git-regression.js +803 -0
  88. package/dist/engine/git-regression.js.map +1 -0
  89. package/dist/engine/index.d.ts +55 -0
  90. package/dist/engine/index.d.ts.map +1 -0
  91. package/dist/engine/index.js +2782 -0
  92. package/dist/engine/index.js.map +1 -0
  93. package/dist/engine/node-test-reporter.d.ts +2 -0
  94. package/dist/engine/node-test-reporter.d.ts.map +1 -0
  95. package/dist/engine/node-test-reporter.js +70 -0
  96. package/dist/engine/node-test-reporter.js.map +1 -0
  97. package/dist/engine/runtime-doctor.d.ts +16 -0
  98. package/dist/engine/runtime-doctor.d.ts.map +1 -0
  99. package/dist/engine/runtime-doctor.js +100 -0
  100. package/dist/engine/runtime-doctor.js.map +1 -0
  101. package/dist/evaluation/agentic-corpus.d.ts +161 -0
  102. package/dist/evaluation/agentic-corpus.d.ts.map +1 -0
  103. package/dist/evaluation/agentic-corpus.js +710 -0
  104. package/dist/evaluation/agentic-corpus.js.map +1 -0
  105. package/dist/index.d.ts +8 -0
  106. package/dist/index.d.ts.map +1 -0
  107. package/dist/index.js +8 -0
  108. package/dist/index.js.map +1 -0
  109. package/dist/mcp/index.d.ts +13 -0
  110. package/dist/mcp/index.d.ts.map +1 -0
  111. package/dist/mcp/index.js +391 -0
  112. package/dist/mcp/index.js.map +1 -0
  113. package/dist/mcp/stdio.d.ts +3 -0
  114. package/dist/mcp/stdio.d.ts.map +1 -0
  115. package/dist/mcp/stdio.js +13 -0
  116. package/dist/mcp/stdio.js.map +1 -0
  117. package/dist/sdk/index.d.ts +52 -0
  118. package/dist/sdk/index.d.ts.map +1 -0
  119. package/dist/sdk/index.js +224 -0
  120. package/dist/sdk/index.js.map +1 -0
  121. package/dist/version.d.ts +2 -0
  122. package/dist/version.d.ts.map +1 -0
  123. package/dist/version.js +10 -0
  124. package/dist/version.js.map +1 -0
  125. package/docs/adapter-protocol.md +196 -0
  126. package/docs/agentic-benchmark.md +118 -0
  127. package/docs/agentic-corpus-experiment-h3.md +89 -0
  128. package/docs/agentic-corpus-plan.md +105 -0
  129. package/docs/agentic-corpus-provenance.md +59 -0
  130. package/docs/agentic-test-profile-pilot.md +57 -0
  131. package/docs/agentic-test-profile-v2.md +116 -0
  132. package/docs/agentic-test-profile.md +274 -0
  133. package/docs/architecture.md +157 -0
  134. package/docs/ci.md +37 -0
  135. package/docs/client-connections.md +61 -0
  136. package/docs/conformance-v1.md +72 -0
  137. package/docs/decisions/0001-typescript-runtime.md +24 -0
  138. package/docs/developer-experience.md +55 -0
  139. package/docs/diagnostics.md +35 -0
  140. package/docs/distribution.md +40 -0
  141. package/docs/git-regression.md +39 -0
  142. package/docs/migration-repository-validation-order.md +35 -0
  143. package/docs/migration-testforge-to-assertledger.md +64 -0
  144. package/docs/project-intent.md +173 -0
  145. package/docs/proof-model.md +116 -0
  146. package/docs/reference.md +336 -0
  147. package/docs/release-1.0.md +63 -0
  148. package/docs/repository-audit.md +52 -0
  149. package/docs/repository-init.md +60 -0
  150. package/docs/research-basis.md +27 -0
  151. package/docs/roadmap.md +74 -0
  152. package/docs/runtime-doctor.md +65 -0
  153. package/docs/testexplora-calibration.md +71 -0
  154. package/examples/agentic-benchmark/benchmark-request.mjs +19 -0
  155. package/examples/agentic-benchmark/structured-phase-adapter-fixture.mjs +35 -0
  156. package/examples/agentic-profile/profile-benchmark.mjs +34 -0
  157. package/examples/agentic-profile/profile-manifest.mjs +28 -0
  158. package/examples/git-history/README.md +44 -0
  159. package/examples/git-history/create-demo.mjs +128 -0
  160. package/examples/git-history/escape-string-regexp/LICENSE +9 -0
  161. package/examples/git-history/escape-string-regexp/before.cjs.txt +11 -0
  162. package/examples/git-history/escape-string-regexp/fixed.cjs.txt +13 -0
  163. package/examples/git-history/escape-string-regexp/provenance.json +28 -0
  164. package/examples/node-test/repository/package.json +5 -0
  165. package/examples/node-test/repository/src/is-even.js +3 -0
  166. package/examples/node-test/repository/tests/base.test.js +6 -0
  167. package/examples/node-test/request.json +93 -0
  168. package/integrations/skill/SKILL.md +51 -0
  169. package/package.json +88 -0
  170. package/schemas/agentic-benchmark-acquisition-replay-result.v1.json +70 -0
  171. package/schemas/agentic-benchmark-acquisition-request.v1.json +564 -0
  172. package/schemas/agentic-benchmark-acquisition-result.v1.json +1409 -0
  173. package/schemas/agentic-benchmark-artifact.v1.json +1251 -0
  174. package/schemas/agentic-benchmark-replay-result.v1.json +84 -0
  175. package/schemas/agentic-benchmark-request.v1.json +1034 -0
  176. package/schemas/agentic-corpus-allocation-commitment-replay-result.v1.json +58 -0
  177. package/schemas/agentic-corpus-allocation-commitment.v1.json +141 -0
  178. package/schemas/agentic-corpus-allocation-replay-result.v1.json +34 -0
  179. package/schemas/agentic-corpus-allocation-request.v1.json +65 -0
  180. package/schemas/agentic-corpus-allocation-reveal.v1.json +66 -0
  181. package/schemas/agentic-corpus-allocation.v1.json +167 -0
  182. package/schemas/agentic-corpus-experiment-artifact.v1.json +329 -0
  183. package/schemas/agentic-corpus-experiment-plan-replay-result.v1.json +50 -0
  184. package/schemas/agentic-corpus-experiment-plan.v1.json +424 -0
  185. package/schemas/agentic-corpus-experiment-replay-request.v1.json +336 -0
  186. package/schemas/agentic-corpus-experiment-replay-result.v1.json +106 -0
  187. package/schemas/agentic-corpus-experiment-request.v1.json +204 -0
  188. package/schemas/agentic-corpus-provenance.v1.json +143 -0
  189. package/schemas/agentic-corpus-trust-policy.v1.json +133 -0
  190. package/schemas/agentic-profile-replay-result.v1.json +56 -0
  191. package/schemas/agentic-profile-replay-result.v2.json +63 -0
  192. package/schemas/agentic-profile-report.v1.json +961 -0
  193. package/schemas/agentic-profile-report.v2.json +1674 -0
  194. package/schemas/agentic-profile-request.v1.json +671 -0
  195. package/schemas/agentic-profile-request.v2.json +1338 -0
  196. package/schemas/evidence-manifest.v1.json +636 -0
  197. package/schemas/replay-result.v1.json +49 -0
  198. package/schemas/repository-analysis.v1.json +119 -0
  199. package/schemas/repository-audit.v1.json +811 -0
  200. package/schemas/repository-init-config.v1.json +183 -0
  201. package/schemas/repository-init-lock.v1.json +162 -0
  202. package/schemas/repository-init-result.v1.json +212 -0
  203. package/schemas/verification-request.v1.json +389 -0
@@ -0,0 +1,89 @@
1
+ # Agentic corpus experiment artifact H3 v1
2
+
3
+ H3 evaluates historical-fault detection only after three independent operator pins are supplied:
4
+ the corpus trust-policy digest, the two-party allocation-commitment digest, and the pre-declared
5
+ experiment-plan digest. None may fall back to a digest embedded in the artifact.
6
+
7
+ ## Allocation commitment and reveal
8
+
9
+ Allocation Commitment/Reveal v1 binds the exact case identity, provenance, source ID and
10
+ source-identity digests plus a calibration count for every source stratum. Each source with at
11
+ least two cases contributes at least one calibration and one holdout case. Distinct Ed25519 AUTHOR
12
+ and REVIEWER principals each commit a
13
+ 32-byte secret share and sign the same domain-separated commitment projection containing both
14
+ share commitments. The reveal must contain both pre-committed shares. AssertLedger sorts them by key
15
+ ID, length-prefixes their decoded bytes, derives the final seed, then recomputes and replays Corpus
16
+ Allocation v1. This prevents either principal from choosing or replacing a share after the joint
17
+ commitment; it does not prove that a principal avoided grinding before committing its own share.
18
+
19
+ ## Pre-declared plan
20
+
21
+ The externally pinned Experiment Plan v1 contains no outcomes, logs, durations or results. It binds
22
+ the allocation and trust anchors, H3 protocol, detailed subject revisions and qualification
23
+ receipts, adapter identities, a shared content-addressed candidate universe, each arm's
24
+ content-addressed selection and derived suite digest, executable identities plus argument arrays,
25
+ repository state, test-suite digests, time/output caps, and the exact schedule. Every
26
+ case/arm/attempt/state tuple appears exactly once. Replay derives both equal
27
+ `PLANNED_PROCESS_EXECUTION` counts and equal total planned timeout ceilings by summing each
28
+ scheduled command's `timeoutMs`. This is comparable pre-declared time allowance, not observed
29
+ wall-time or CPU equality.
30
+
31
+ ## Run evidence and derivation
32
+
33
+ Each scheduled process has one canonical run receipt binding its plan, run and command digests,
34
+ case, arm, attempt, subject state, repository and test-suite digests, the applied timeout limit,
35
+ process exit/timed-out state,
36
+ and SHA-256 digests for stdout, stderr and a structured result. Replay resolves the actual receipt,
37
+ stdout, stderr and structured-result bytes and recomputes every digest. The strict
38
+ `TESTFORGE_H3_STRUCTURED_RESULT_V1` parser derives normalized outcome and attribution; no outcome
39
+ field in the receipt is trusted.
40
+
41
+ Replay resolves every `candidateDigest` in the frozen universe against the supplied evidence
42
+ contents, recomputes its raw SHA-256 digest, and checks the artifact's derived candidate-set digest.
43
+ Missing or mismatched candidate bytes invalidate replay. This proves the identity of the bytes
44
+ provided to AssertLedger, not that the runner actually executed those bytes.
45
+
46
+ Process consistency is exact: `PASS` requires `timedOut=false` and exit code zero;
47
+ `PROCESS_CRASH` requires `timedOut=false` and a null exit code; `TIMEOUT` requires
48
+ `timedOut=true`; every assertion, compile, collection, infrastructure, or no-test failure requires
49
+ `timedOut=false` and a nonzero non-null exit code. A contradiction invalidates the artifact rather
50
+ than becoming insufficient evidence.
51
+
52
+ Detection requires an attributed assertion failure on the buggy revision and PASS on the fixed
53
+ revision and required suite for every stable attempt. Buggy PASS is a valid non-detection. Compile,
54
+ collection, timeout, crash, infrastructure and no-test outcomes are valid operational evidence but
55
+ make H3 `INSUFFICIENT`; contradictory attempts do the same. They never become detections.
56
+ Non-inferiority must hold both in aggregate and independently for every source stratum, so gains
57
+ on one source cannot compensate for regressions on another.
58
+
59
+ The SDK requires the artifact and replay options as separate arguments. The CLI requires separate
60
+ files and flags for all three external pins, policy, commitment/reveal, allocation, plan, subject
61
+ evidence and evidence bytes. H3 tools are intentionally absent from the default agent-facing MCP
62
+ server because an input JSON document cannot establish those operator-owned roots of trust.
63
+
64
+ ## Compatibility and non-claims
65
+
66
+ Existing AgenticCorpusCase v1 `h1`-`h4` members remain wire-compatible as
67
+ `LEGACY_CASE_METRICS`, but any such member makes a subject inadmissible to mature H3 replay. This
68
+ contract establishes deterministic consistency relative to supplied external anchors and bytes. It
69
+ does not authenticate the machine that executed commands, prove private custody, or establish that
70
+ the corpus represents all real defects. Qualification receipt digests are pre-declared identities;
71
+ H3 v1 does not resolve their bytes. A run receipt's applied-timeout field binds the runner's
72
+ attestation to the pre-declared limit; it is not an independent execution authority. H1 latency
73
+ and H4 mutation strength remain separate and cannot
74
+ compensate for H3 failure.
75
+
76
+ ## Migration note
77
+
78
+ The pre-release v1 allocation and H3 projections were tightened before a stable release. Global
79
+ `calibrationCount` plus `caseIds` inputs are rejected; callers must provide source strata. Candidate
80
+ IDs became content-addressed references, plan budgets gained derived timeout ceilings, receipts
81
+ gained `appliedTimeoutMs`, and results gained per-source strata. Existing stored drafts must be
82
+ regenerated and externally repinned. Artifacts also gained a derived candidate-set digest and
83
+ replay gained candidate-byte resolution. Redigesting an old document is insufficient.
84
+
85
+ Recent research reinforces those boundaries: coverage and mutation can be context-dependent
86
+ proxies when the code under test may already be buggy ([arXiv:2607.22880](https://arxiv.org/abs/2607.22880)),
87
+ while a strong recent baseline and evaluation granularity can materially change measured cost
88
+ ([arXiv:2601.09695](https://arxiv.org/abs/2601.09695)). AssertLedger therefore requires a frozen shared
89
+ candidate universe rather than comparison against a conveniently weak historical prompt.
@@ -0,0 +1,105 @@
1
+ # Agentic profile corpus plan
2
+
3
+ This document defines the first empirical calibration corpus for the Agentic Test Profile. It is a
4
+ reconstruction plan, not collected evidence. A case counts only after source-native reproduction,
5
+ dual-signed provenance, and admission to the physical corpus.
6
+
7
+ ## Frozen pilot scope
8
+
9
+ The planned pilot targets 24 independently qualified historical-fault cases: eight from each
10
+ source. Four cases per source are intended for public calibration and four for holdout. Case
11
+ identifiers are frozen only after the buggy/fixed pair reproduces on the pinned environment. At
12
+ present, only the eight TestExplora cases below have qualified, and only for curated calibration.
13
+
14
+ | Source | Pin | Cases | Primary contribution |
15
+ | --- | --- | ---: | --- |
16
+ | Defects4J 3.0.1 | `rjust/defects4j@8c16da8230843cdc918eaf4ddb449637f02b83c6` | 8 | H1, H3, and the pilot's only native H4 mutation baseline |
17
+ | TestExplora | harness `microsoft/TestExplora@11e6952261f58d3ceb9f4d2571e03aa35dad9523`; dataset `91d8edfb851331bc77eddff40c794336b16c79fe` | 8 | Curated H1/H3 calibration candidates from real Python pull requests; not a holdout |
18
+ | SWE-bench Verified | dataset `78f471bf655a3137b2e8a75af1501690ec009ec3`; harness `7a21e05772954cc81471ae19d56f436cecf43c54` | 8 | H1 and H3 for recent Python issue/PR regressions |
19
+
20
+ TestExplora is admitted only as `ADMITTED_CURATED_CALIBRATION`. Eight cases reproduced with stable
21
+ attributable assertion failures on the buggy state and stable passes on the fixed state, but the
22
+ selection was calibrated across multiple pre-declared campaigns. It is therefore neither a hidden
23
+ holdout nor evidence that the corpus is `READY_H3`. The harness also does not publish a frozen
24
+ per-task environment; AssertLedger reconstructed dependency images from pinned inputs and executed
25
+ them without network access. See [TestExplora calibration evidence](testexplora-calibration.md).
26
+ Subject repositories and container images must be recorded by immutable commit or image digest in
27
+ each execution receipt.
28
+
29
+ ## Allocation freeze boundary
30
+
31
+ Calibration and holdout membership must be created once before any profile threshold, portfolio,
32
+ or prompt is tuned. A mature freeze uses Allocation Commitment/Reveal v1: distinct authorized
33
+ AUTHOR and REVIEWER principals pre-commit independent secret shares while signing the exact case
34
+ identity/provenance/source set and the calibration count for every source stratum. Every source
35
+ with at least two admitted cases must retain non-empty calibration and holdout partitions. After
36
+ both shares are revealed, AssertLedger derives a
37
+ domain-separated seed, applies `SHA256_ASCENDING_SPLIT_V1`, and replays the resulting allocation.
38
+ The operator must independently pin the trust-policy and commitment digests.
39
+
40
+ Real allocation seeds, membership, score ordering, and allocation digests remain in private
41
+ holdout custody outside the repository. Public tests use an unrelated synthetic allocation only;
42
+ they establish algorithm conformance without revealing any real holdout membership. The admitted
43
+ cases remain pending until their complete provenance sidecars and required evidence are present.
44
+
45
+ The earlier single-nonce pilot commitment is invalidated: one party could choose the seed and the
46
+ receipt did not bind dual authorization, case provenance, or the allocation protocol. It is not a
47
+ mature freeze and must not be used for an H3 claim. No valid real holdout commitment, seed,
48
+ allocation digest, case identifier, or membership is published in this repository.
49
+
50
+ The existing optional `h1` through `h4` members of an AgenticCorpusCase v1 are retained unchanged
51
+ for wire compatibility and are designated `LEGACY_CASE_METRICS`. They are self-contained summary
52
+ claims, not replayable experiment artifacts, and therefore cannot support a mature profile,
53
+ sweet-spot, or hypothesis claim. Mature H3 evaluation uses the separate versioned
54
+ [experiment artifact](agentic-corpus-experiment-h3.md), which requires an externally pinned
55
+ two-party allocation commitment and pre-declared experiment plan. H1, H2, and H4 remain blocked on
56
+ their own mature experiment artifacts.
57
+
58
+ ## Admission protocol
59
+
60
+ Every case must record the benchmark pin, exact subject buggy and fixed revisions, environment and
61
+ tool versions, an executable-plus-arguments command, input and output digests, and at least three
62
+ repeated source-native runs. The expected fault must fail on the buggy revision and pass on the
63
+ fixed revision together with the source-required regression suite.
64
+
65
+ Timeout, setup, compilation, collection, process, and infrastructure failures are inconclusive.
66
+ They never count as historical-fault detections or mutant kills. Source files are reconstructed from
67
+ their upstream repositories by default; AssertLedger does not assume that a benchmark framework's
68
+ license grants redistribution rights for every subject repository.
69
+
70
+ After reproduction, an authorized author and a different authorized reviewer sign the canonical
71
+ case sidecar described in [Agentic corpus provenance v1](agentic-corpus-provenance.md). The expected
72
+ trust-policy digest must be distributed outside the corpus. Pending or unsigned cases stay outside
73
+ the active `public/` and `private/` splits.
74
+
75
+ ## Hypothesis boundaries
76
+
77
+ - H1 uses warmed repeated execution p95 for the selected portfolio and the declared full eligible
78
+ suite at equal target strength.
79
+ - H2 requires separately constructed, language-specific behavior-preserving neutral worlds and an
80
+ independent preservation review. The three historical-fault datasets do not supply this proof.
81
+ Formatting-only controls may test integrity, but cannot establish substantive neutral robustness.
82
+ - H3 uses source-stratified historical-fault holdouts that remain unseen during policy calibration.
83
+ - H4 is scoped to the Defects4J tranche in the pilot because it alone exposes a native mutation
84
+ interface. No cross-language H4 claim follows from this tranche.
85
+
86
+ The corpus becomes mechanically `READY` only after the existing minimum-case, minimum-source,
87
+ public/holdout, H1-H4 evaluability, and signed-provenance gates pass. `READY` means the corpus may be
88
+ evaluated; it does not mean the hypotheses are supported. A public sweet-spot claim additionally
89
+ requires the predeclared public and holdout evaluation criteria to succeed.
90
+
91
+ ## Primary sources
92
+
93
+ - [Defects4J pinned README](https://raw.githubusercontent.com/rjust/defects4j/8c16da8230843cdc918eaf4ddb449637f02b83c6/README.md)
94
+ and [command reference](https://defects4j.org/html_doc/index.html)
95
+ - [TestExplora pinned harness](https://github.com/microsoft/TestExplora/tree/11e6952261f58d3ceb9f4d2571e03aa35dad9523)
96
+ and [dataset](https://huggingface.co/datasets/microsoft/TestExplora/tree/91d8edfb851331bc77eddff40c794336b16c79fe)
97
+ - [SWE-bench dataset fields](https://www.swebench.com/SWE-bench/guides/datasets/),
98
+ [pinned harness](https://github.com/SWE-bench/SWE-bench/tree/7a21e05772954cc81471ae19d56f436cecf43c54),
99
+ and [benchmark description](https://www.swebench.com/original.html)
100
+
101
+ Defects4J, TestExplora, and the SWE-bench harness use MIT licenses at the listed pins. Each
102
+ TestExplora subject keeps its own license, verified at the selected base commit; the dataset license
103
+ does not replace subject licensing. BugsInPy and Bugs.jar are excluded because their official
104
+ repositories do not provide a usable declared license. GitBug-Java remains a later expansion due
105
+ to its substantially larger documented storage footprint.
@@ -0,0 +1,59 @@
1
+ # Agentic corpus provenance v1
2
+
3
+ An Agentic Test Profile corpus case is not trusted because it names a plausible `sourceId`.
4
+ Every `<name>.case.json` must retain the unchanged case-v1 format and be paired with a canonical
5
+ `<name>.provenance.json`. Readiness additionally requires an operator-supplied trust policy and its
6
+ expected digest.
7
+
8
+ ## Trust root
9
+
10
+ The trust policy registers Ed25519 SPKI public keys, their subjects and AUTHOR or REVIEWER roles,
11
+ then binds each source ID to one immutable source-identity digest and explicit authorized subjects.
12
+ The `policyDigest` is the canonical SHA-256 digest of the policy without that field. Callers must
13
+ pin the same value independently with `--trust-policy-digest`; a policy bundled beside a corpus is
14
+ not its own root of trust.
15
+
16
+ Key IDs are SHA-256 digests of canonical SPKI DER bytes. AssertLedger imports and re-exports every key,
17
+ requires byte-for-byte equality with the supplied DER, and compares canonical fingerprints. This
18
+ rejects alternate encodings such as a valid SPKI followed by ignored suffix bytes. A case requires
19
+ different author and reviewer keys, canonical fingerprints, and subjects. Both principals sign the
20
+ same domain-separated canonical projection, including both key IDs, the case digest, source identity
21
+ and revision, and execution receipt. `provenanceDigest` then binds that projection and both
22
+ signatures.
23
+
24
+ `keyId` values are globally unique within a policy. `subjectId` values intentionally are not: one
25
+ principal may rotate across multiple trusted keys. That rotation does not create reviewer
26
+ independence. For a case replay, the selected author and reviewer must still have different key IDs,
27
+ canonical key fingerprints, and subject IDs.
28
+
29
+ Sidecars use canonical JSON exactly: `canonicalize(parsed) + "\n"`. Formatting variants and JSON
30
+ with duplicate members are rejected before evaluation. A sidecar must be a regular, non-symlink
31
+ file whose resolved parent is the physical split directory. Case identifiers and paired basenames
32
+ also use AssertLedger's NFC and case-folded portable path key, so host-dependent collisions fail closed.
33
+
34
+ ## CLI
35
+
36
+ ```sh
37
+ tsx scripts/evaluate-agentic-profile-corpus.ts status benchmarks/agentic-profile \
38
+ --trust-policy operator-policy.json \
39
+ --trust-policy-digest sha256:...
40
+
41
+ tsx scripts/evaluate-agentic-profile-corpus.ts replay-provenance \
42
+ public/example.case.json public/example.provenance.json \
43
+ --trust-policy operator-policy.json \
44
+ --trust-policy-digest sha256:...
45
+ ```
46
+
47
+ `status`, `evaluate-public`, and `evaluate-holdout` return exit 3 when not ready. Invalid policy or
48
+ provenance returns exit 4. Valid replay or evaluation returns exit 0. AssertLedger intentionally has no
49
+ private-key or signing command.
50
+
51
+ ## Migration and limits
52
+
53
+ Unsigned v1 case files remain parse-compatible, but can never make a corpus READY. Producers must
54
+ retain each case unchanged, create its paired sidecar, and distribute the policy digest through an
55
+ independent trusted channel.
56
+
57
+ The signatures prove that the named authorized principals approved the signed bytes. They do not
58
+ prove that execution actually occurred. Until `evidenceDigests` resolve to independently replayable
59
+ artifacts, the EXECUTED receipt remains an attestation rather than execution proof.
@@ -0,0 +1,57 @@
1
+ # Agentic Test Profile pilot
2
+
3
+ This pilot exercises the implemented profile against two real AssertLedger campaigns from
4
+ `dofus-battlebot`. It is compatibility and local repeatability evidence, not a benchmark of broad
5
+ fault-detection effectiveness.
6
+
7
+ ## Frozen inputs
8
+
9
+ | Campaign | Source artifact digest | Observations | Attempts | Source decision |
10
+ | --- | --- | ---: | ---: | --- |
11
+ | `recover-v1` | `sha256:e696766ae1effc96203d2077cf51e9e7f731f708dad1f0c94b9c6a9876d1cca9` | 90 | 3 | `VERIFIED` |
12
+ | `recover-v2` | `sha256:ed03d03d2892d7b945ef0af0506bb44ccb84eff73ea7f99a990335312137ba9f` | 90 | 3 | `VERIFIED` |
13
+
14
+ Both campaigns contain one reference world, four required target worlds, one required neutral
15
+ world, and four candidate tests. They use the unsandboxed trusted-local backend.
16
+
17
+ ## Policy
18
+
19
+ Both manifests were profiled with three local lanes: `instant` at 2,000 ms, `loop` at 10,000 ms,
20
+ and `gate` at 60,000 ms. The minimum timing sample count was three.
21
+
22
+ ## Results
23
+
24
+ | Campaign | Profile | Qualified candidate | Target weight | Reference p95 | Pareto | Replay |
25
+ | --- | --- | --- | ---: | ---: | --- | --- |
26
+ | `recover-v1` | `QUALIFIED` | `boundary-complete` | 1000/1000 | 159 ms | yes | valid |
27
+ | `recover-v2` | `QUALIFIED` | `boundary-complete` | 1000/1000 | 165 ms | yes | valid |
28
+
29
+ Every lane selected the same one-candidate portfolio. The other candidates remained
30
+ `NOT_QUALIFIED`, including a fast candidate that killed all targets but failed the reference gate.
31
+ This is the intended non-compensation behavior: speed and mutant kills do not rescue an invalid
32
+ oracle.
33
+
34
+ ## Reproduction
35
+
36
+ After `pnpm build`, run the packaged example against either manifest:
37
+
38
+ ```sh
39
+ node examples/agentic-profile/profile-manifest.mjs /path/to/manifest.json campaign/profile-id
40
+ ```
41
+
42
+ Replay the emitted JSON with `assertledger profile-replay`.
43
+
44
+ ## What this pilot establishes
45
+
46
+ - existing v1 evidence manifests remain accepted without migration;
47
+ - the profile derives the same qualitative decision across two independently sealed campaigns;
48
+ - local timing evidence is sufficient for the declared policy;
49
+ - report replay validates source evidence, policy digest, report digest, and recomputed semantics;
50
+ - an invalid fast candidate cannot enter the Pareto frontier or a selected portfolio.
51
+
52
+ ## What remains unproven
53
+
54
+ The campaigns exercise the same repository behavior, candidate family, and target portfolio. They
55
+ do not establish H1 through H4, compare against a coverage-guided baseline, estimate confidence
56
+ intervals, or test a hidden holdout. They also do not separate cold, warm, collection, compilation,
57
+ and execution phases. A public sweet-spot claim remains blocked on a broader versioned corpus.
@@ -0,0 +1,116 @@
1
+ # Agentic Test Profile v2
2
+
3
+ Status: implemented additive public contract for `testforge-agentic-profile/2.0.0`.
4
+
5
+ ## Why v2 exists
6
+
7
+ Profile v1 derives latency from wall times embedded in a verification campaign. Those observations
8
+ are useful local evidence, but they do not distinguish cold startup, compilation or collection,
9
+ warm execution, or the measurement environment. Profile v2 consumes a replay-valid Agentic
10
+ Benchmark Artifact and uses one fixed cost basis:
11
+
12
+ ```text
13
+ regime: WARM
14
+ measure: WALL
15
+ aggregation: TOTAL
16
+ statistic: P95
17
+ unit: MICROSECOND
18
+ portfolio aggregation: SUM_OF_INDIVIDUAL_P95
19
+ ```
20
+
21
+ No CPU time, phase timing, cold timing, candidate size, APFDc, or composite score may compensate
22
+ for this cost basis or enter the Pareto decision.
23
+
24
+ `benchmark-acquire` is the direct path to a replay-valid artifact for this profile. Profile v2 still
25
+ evaluates only the source-selected eligible universe. A successful acquisition establishes neither
26
+ a repository-global optimum nor a universal sweet spot, and timing failures never weaken or revise
27
+ the source `VERIFIED` verdict.
28
+
29
+ ## Preconditions and status precedence
30
+
31
+ The request embeds the complete benchmark artifact and names its required
32
+ `comparisonScopeDigest`. AssertLedger first replays that artifact. Invalid benchmark evidence produces
33
+ no report.
34
+
35
+ For a valid artifact, report status follows this exact precedence:
36
+
37
+ 1. `COMPARISON_SCOPE_MISMATCH` when the policy and artifact scopes differ;
38
+ 2. `OBSERVED_BENCHMARK_FAILURE` when any warm measurement failed;
39
+ 3. `INSUFFICIENT_TIMING_EVIDENCE` when any warm summary lacks enough accepted samples;
40
+ 4. `QUALIFIED` when at least one declared lane admits a non-zero-strength candidate;
41
+ 5. `BUDGET_MISSED` otherwise.
42
+
43
+ The first three states suppress the whole cohort's Pareto frontier and portfolios. A cold failure
44
+ remains visible in the embedded benchmark but does not block the warm-cost profile.
45
+
46
+ ## Candidate universe
47
+
48
+ Profile v2 deliberately evaluates only candidates that the source manifest both classified
49
+ `ELIGIBLE` and selected. The report states this as `SOURCE_SELECTED_ELIGIBLE`, lists every included
50
+ identifier, and separately lists eligible candidates excluded by source selection.
51
+
52
+ This boundary matters: the report does not claim to find the best candidate among every submitted
53
+ or eligible candidate. Broader comparison requires a benchmark artifact whose source selection
54
+ contains that broader cohort.
55
+
56
+ ## Pareto frontier
57
+
58
+ Candidate A dominates B only when A is no worse on all three dimensions and strictly better on at
59
+ least one:
60
+
61
+ - target-world weight killed: greater is better;
62
+ - required target worlds killed: greater is better;
63
+ - warm total wall p95: lower is better.
64
+
65
+ The report does not collapse these dimensions into a single score. Tie-breaking in portfolio
66
+ selection uses exact integer cross-products with `BigInt`, followed by marginal weight, cost,
67
+ candidate size, digest, and identifier. Candidate size is only a deterministic tie-breaker; it is
68
+ not a quality dimension.
69
+
70
+ ## Portfolio semantics
71
+
72
+ For each strictly increasing lane budget, AssertLedger repeatedly selects the remaining candidate with
73
+ the greatest new target weight per unit of warm p95 cost, recomputing marginal gain after every
74
+ step. Zero-cost positive gain ranks first. Each step records its marginal targets, marginal weight,
75
+ individual cost, and cumulative totals.
76
+
77
+ `sumIndividualWarmTotalWallP95Us` is exactly the sum of individual candidate p95 values. It is not
78
+ a measured portfolio p95: shared startup, caching, parallelism, contention, and test-runner
79
+ scheduling could make an executed suite differ materially.
80
+
81
+ ## Replay and compatibility
82
+
83
+ The report embeds its benchmark, policy, explicit cohort, candidate rows, portfolios, limitations,
84
+ and canonical digest. Replay independently verifies:
85
+
86
+ - benchmark replay;
87
+ - benchmark digest and comparison-scope binding;
88
+ - policy digest;
89
+ - report digest; and
90
+ - a complete deterministic semantic recomputation.
91
+
92
+ Profile v2 is additive. Profile v1 schemas, SDK methods, CLI commands, MCP tools, and report digests
93
+ remain unchanged. There is no automatic v1-to-v2 conversion because v1 lacks the fingerprinted,
94
+ cold/warm, phase-aware raw evidence required by v2.
95
+
96
+ ## Usage
97
+
98
+ ```sh
99
+ assertledger benchmark benchmark-request.json --json > benchmark-artifact.json
100
+ assertledger benchmark-replay benchmark-artifact.json --json
101
+ assertledger profile-v2 profile-v2-request.json --json > profile-v2-report.json
102
+ assertledger profile-v2-replay profile-v2-report.json --json
103
+ ```
104
+
105
+ The SDK methods are `profileV2()` and `replayProfileV2()`. MCP exposes the preferred
106
+ `assertledger_profile_v2` and `assertledger_profile_v2_replay` tools alongside the legacy
107
+ `testforge_profile_v2` and `testforge_profile_v2_replay` aliases.
108
+
109
+ ## Non-claims
110
+
111
+ Profile v2 does not prove program correctness, world completeness, permanent stability, provenance
112
+ authenticity beyond the externally pinned corpus policy, cross-machine portability, global subset optimality, or an empirically optimal
113
+ coverage-versus-speed sweet spot. The last claim remains blocked until the public/private corpus
114
+ passes its H1-H4 readiness and holdout gates on enough independently reviewed projects. Corpus
115
+ readiness now requires canonical dual-signed sidecars; see
116
+ [Agentic corpus provenance v1](agentic-corpus-provenance.md).