assertledger 1.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (203) hide show
  1. package/CONTRIBUTING.md +31 -0
  2. package/LICENSE +21 -0
  3. package/README.fr.md +236 -0
  4. package/README.md +224 -0
  5. package/SECURITY.md +51 -0
  6. package/benchmarks/agentic-profile/README.md +15 -0
  7. package/benchmarks/agentic-profile/public/README.md +5 -0
  8. package/benchmarks/self-hosted-core/README.md +113 -0
  9. package/benchmarks/self-hosted-core/adapter.mjs +293 -0
  10. package/benchmarks/self-hosted-core/builder.ts +193 -0
  11. package/benchmarks/self-hosted-core/campaign.ts +233 -0
  12. package/benchmarks/self-hosted-core/existing-tests-builder.ts +217 -0
  13. package/benchmarks/self-hosted-core/existing-tests.ts +146 -0
  14. package/benchmarks/self-hosted-core/liveness.test.mjs +8 -0
  15. package/conformance/v1/bundle.json +104 -0
  16. package/conformance/v1/expected/canonical-order-a.json +4 -0
  17. package/conformance/v1/expected/canonical-order-b.json +4 -0
  18. package/conformance/v1/expected/create-benchmark-v1-measured.json +575 -0
  19. package/conformance/v1/expected/create-profile-v1-qualified.json +280 -0
  20. package/conformance/v1/expected/decide-collection-failure-non-kill.json +192 -0
  21. package/conformance/v1/expected/decide-compile-failure-non-kill.json +192 -0
  22. package/conformance/v1/expected/decide-infra-error-non-kill.json +192 -0
  23. package/conformance/v1/expected/decide-no-test-discovered-non-kill.json +192 -0
  24. package/conformance/v1/expected/decide-process-crash-non-kill.json +192 -0
  25. package/conformance/v1/expected/decide-timeout-non-kill.json +192 -0
  26. package/conformance/v1/expected/decide-verified.json +192 -0
  27. package/conformance/v1/expected/replay-benchmark-v1-resealed-summary-forgery.json +12 -0
  28. package/conformance/v1/expected/replay-evidence-raw-tamper.json +6 -0
  29. package/conformance/v1/expected/replay-evidence-resealed-semantic-forgery.json +6 -0
  30. package/conformance/v1/inputs/canonical-order-a.json +8 -0
  31. package/conformance/v1/inputs/canonical-order-b.json +8 -0
  32. package/conformance/v1/inputs/create-benchmark-v1-measured.json +459 -0
  33. package/conformance/v1/inputs/create-profile-v1-qualified.json +228 -0
  34. package/conformance/v1/inputs/decide-collection-failure-non-kill.json +143 -0
  35. package/conformance/v1/inputs/decide-compile-failure-non-kill.json +143 -0
  36. package/conformance/v1/inputs/decide-infra-error-non-kill.json +143 -0
  37. package/conformance/v1/inputs/decide-no-test-discovered-non-kill.json +143 -0
  38. package/conformance/v1/inputs/decide-process-crash-non-kill.json +143 -0
  39. package/conformance/v1/inputs/decide-timeout-non-kill.json +143 -0
  40. package/conformance/v1/inputs/decide-verified.json +143 -0
  41. package/conformance/v1/inputs/replay-benchmark-v1-resealed-summary-forgery.json +575 -0
  42. package/conformance/v1/inputs/replay-evidence-raw-tamper.json +201 -0
  43. package/conformance/v1/inputs/replay-evidence-resealed-semantic-forgery.json +192 -0
  44. package/conformance/v1/schemas/expected-digests.json +175 -0
  45. package/dist/cli.d.ts +9 -0
  46. package/dist/cli.d.ts.map +1 -0
  47. package/dist/cli.js +951 -0
  48. package/dist/cli.js.map +1 -0
  49. package/dist/contracts/diagnostics.d.ts +18 -0
  50. package/dist/contracts/diagnostics.d.ts.map +1 -0
  51. package/dist/contracts/diagnostics.js +13 -0
  52. package/dist/contracts/diagnostics.js.map +1 -0
  53. package/dist/contracts/index.d.ts +3908 -0
  54. package/dist/contracts/index.d.ts.map +1 -0
  55. package/dist/contracts/index.js +2569 -0
  56. package/dist/contracts/index.js.map +1 -0
  57. package/dist/contracts/runtime-doctor.d.ts +107 -0
  58. package/dist/contracts/runtime-doctor.d.ts.map +1 -0
  59. package/dist/contracts/runtime-doctor.js +91 -0
  60. package/dist/contracts/runtime-doctor.js.map +1 -0
  61. package/dist/core/index.d.ts +200 -0
  62. package/dist/core/index.d.ts.map +1 -0
  63. package/dist/core/index.js +2587 -0
  64. package/dist/core/index.js.map +1 -0
  65. package/dist/diagnostics.d.ts +7 -0
  66. package/dist/diagnostics.d.ts.map +1 -0
  67. package/dist/diagnostics.js +252 -0
  68. package/dist/diagnostics.js.map +1 -0
  69. package/dist/engine/adapters/node-test-profile.d.ts +14 -0
  70. package/dist/engine/adapters/node-test-profile.d.ts.map +1 -0
  71. package/dist/engine/adapters/node-test-profile.js +14 -0
  72. package/dist/engine/adapters/node-test-profile.js.map +1 -0
  73. package/dist/engine/adapters/node-test-runtime.d.ts +39 -0
  74. package/dist/engine/adapters/node-test-runtime.d.ts.map +1 -0
  75. package/dist/engine/adapters/node-test-runtime.js +173 -0
  76. package/dist/engine/adapters/node-test-runtime.js.map +1 -0
  77. package/dist/engine/adapters/runtime-facts.d.ts +26 -0
  78. package/dist/engine/adapters/runtime-facts.d.ts.map +1 -0
  79. package/dist/engine/adapters/runtime-facts.js +73 -0
  80. package/dist/engine/adapters/runtime-facts.js.map +1 -0
  81. package/dist/engine/connection.d.ts +22 -0
  82. package/dist/engine/connection.d.ts.map +1 -0
  83. package/dist/engine/connection.js +343 -0
  84. package/dist/engine/connection.js.map +1 -0
  85. package/dist/engine/git-regression.d.ts +25 -0
  86. package/dist/engine/git-regression.d.ts.map +1 -0
  87. package/dist/engine/git-regression.js +803 -0
  88. package/dist/engine/git-regression.js.map +1 -0
  89. package/dist/engine/index.d.ts +55 -0
  90. package/dist/engine/index.d.ts.map +1 -0
  91. package/dist/engine/index.js +2782 -0
  92. package/dist/engine/index.js.map +1 -0
  93. package/dist/engine/node-test-reporter.d.ts +2 -0
  94. package/dist/engine/node-test-reporter.d.ts.map +1 -0
  95. package/dist/engine/node-test-reporter.js +70 -0
  96. package/dist/engine/node-test-reporter.js.map +1 -0
  97. package/dist/engine/runtime-doctor.d.ts +16 -0
  98. package/dist/engine/runtime-doctor.d.ts.map +1 -0
  99. package/dist/engine/runtime-doctor.js +100 -0
  100. package/dist/engine/runtime-doctor.js.map +1 -0
  101. package/dist/evaluation/agentic-corpus.d.ts +161 -0
  102. package/dist/evaluation/agentic-corpus.d.ts.map +1 -0
  103. package/dist/evaluation/agentic-corpus.js +710 -0
  104. package/dist/evaluation/agentic-corpus.js.map +1 -0
  105. package/dist/index.d.ts +8 -0
  106. package/dist/index.d.ts.map +1 -0
  107. package/dist/index.js +8 -0
  108. package/dist/index.js.map +1 -0
  109. package/dist/mcp/index.d.ts +13 -0
  110. package/dist/mcp/index.d.ts.map +1 -0
  111. package/dist/mcp/index.js +391 -0
  112. package/dist/mcp/index.js.map +1 -0
  113. package/dist/mcp/stdio.d.ts +3 -0
  114. package/dist/mcp/stdio.d.ts.map +1 -0
  115. package/dist/mcp/stdio.js +13 -0
  116. package/dist/mcp/stdio.js.map +1 -0
  117. package/dist/sdk/index.d.ts +52 -0
  118. package/dist/sdk/index.d.ts.map +1 -0
  119. package/dist/sdk/index.js +224 -0
  120. package/dist/sdk/index.js.map +1 -0
  121. package/dist/version.d.ts +2 -0
  122. package/dist/version.d.ts.map +1 -0
  123. package/dist/version.js +10 -0
  124. package/dist/version.js.map +1 -0
  125. package/docs/adapter-protocol.md +196 -0
  126. package/docs/agentic-benchmark.md +118 -0
  127. package/docs/agentic-corpus-experiment-h3.md +89 -0
  128. package/docs/agentic-corpus-plan.md +105 -0
  129. package/docs/agentic-corpus-provenance.md +59 -0
  130. package/docs/agentic-test-profile-pilot.md +57 -0
  131. package/docs/agentic-test-profile-v2.md +116 -0
  132. package/docs/agentic-test-profile.md +274 -0
  133. package/docs/architecture.md +157 -0
  134. package/docs/ci.md +37 -0
  135. package/docs/client-connections.md +61 -0
  136. package/docs/conformance-v1.md +72 -0
  137. package/docs/decisions/0001-typescript-runtime.md +24 -0
  138. package/docs/developer-experience.md +55 -0
  139. package/docs/diagnostics.md +35 -0
  140. package/docs/distribution.md +40 -0
  141. package/docs/git-regression.md +39 -0
  142. package/docs/migration-repository-validation-order.md +35 -0
  143. package/docs/migration-testforge-to-assertledger.md +64 -0
  144. package/docs/project-intent.md +173 -0
  145. package/docs/proof-model.md +116 -0
  146. package/docs/reference.md +336 -0
  147. package/docs/release-1.0.md +63 -0
  148. package/docs/repository-audit.md +52 -0
  149. package/docs/repository-init.md +60 -0
  150. package/docs/research-basis.md +27 -0
  151. package/docs/roadmap.md +74 -0
  152. package/docs/runtime-doctor.md +65 -0
  153. package/docs/testexplora-calibration.md +71 -0
  154. package/examples/agentic-benchmark/benchmark-request.mjs +19 -0
  155. package/examples/agentic-benchmark/structured-phase-adapter-fixture.mjs +35 -0
  156. package/examples/agentic-profile/profile-benchmark.mjs +34 -0
  157. package/examples/agentic-profile/profile-manifest.mjs +28 -0
  158. package/examples/git-history/README.md +44 -0
  159. package/examples/git-history/create-demo.mjs +128 -0
  160. package/examples/git-history/escape-string-regexp/LICENSE +9 -0
  161. package/examples/git-history/escape-string-regexp/before.cjs.txt +11 -0
  162. package/examples/git-history/escape-string-regexp/fixed.cjs.txt +13 -0
  163. package/examples/git-history/escape-string-regexp/provenance.json +28 -0
  164. package/examples/node-test/repository/package.json +5 -0
  165. package/examples/node-test/repository/src/is-even.js +3 -0
  166. package/examples/node-test/repository/tests/base.test.js +6 -0
  167. package/examples/node-test/request.json +93 -0
  168. package/integrations/skill/SKILL.md +51 -0
  169. package/package.json +88 -0
  170. package/schemas/agentic-benchmark-acquisition-replay-result.v1.json +70 -0
  171. package/schemas/agentic-benchmark-acquisition-request.v1.json +564 -0
  172. package/schemas/agentic-benchmark-acquisition-result.v1.json +1409 -0
  173. package/schemas/agentic-benchmark-artifact.v1.json +1251 -0
  174. package/schemas/agentic-benchmark-replay-result.v1.json +84 -0
  175. package/schemas/agentic-benchmark-request.v1.json +1034 -0
  176. package/schemas/agentic-corpus-allocation-commitment-replay-result.v1.json +58 -0
  177. package/schemas/agentic-corpus-allocation-commitment.v1.json +141 -0
  178. package/schemas/agentic-corpus-allocation-replay-result.v1.json +34 -0
  179. package/schemas/agentic-corpus-allocation-request.v1.json +65 -0
  180. package/schemas/agentic-corpus-allocation-reveal.v1.json +66 -0
  181. package/schemas/agentic-corpus-allocation.v1.json +167 -0
  182. package/schemas/agentic-corpus-experiment-artifact.v1.json +329 -0
  183. package/schemas/agentic-corpus-experiment-plan-replay-result.v1.json +50 -0
  184. package/schemas/agentic-corpus-experiment-plan.v1.json +424 -0
  185. package/schemas/agentic-corpus-experiment-replay-request.v1.json +336 -0
  186. package/schemas/agentic-corpus-experiment-replay-result.v1.json +106 -0
  187. package/schemas/agentic-corpus-experiment-request.v1.json +204 -0
  188. package/schemas/agentic-corpus-provenance.v1.json +143 -0
  189. package/schemas/agentic-corpus-trust-policy.v1.json +133 -0
  190. package/schemas/agentic-profile-replay-result.v1.json +56 -0
  191. package/schemas/agentic-profile-replay-result.v2.json +63 -0
  192. package/schemas/agentic-profile-report.v1.json +961 -0
  193. package/schemas/agentic-profile-report.v2.json +1674 -0
  194. package/schemas/agentic-profile-request.v1.json +671 -0
  195. package/schemas/agentic-profile-request.v2.json +1338 -0
  196. package/schemas/evidence-manifest.v1.json +636 -0
  197. package/schemas/replay-result.v1.json +49 -0
  198. package/schemas/repository-analysis.v1.json +119 -0
  199. package/schemas/repository-audit.v1.json +811 -0
  200. package/schemas/repository-init-config.v1.json +183 -0
  201. package/schemas/repository-init-lock.v1.json +162 -0
  202. package/schemas/repository-init-result.v1.json +212 -0
  203. package/schemas/verification-request.v1.json +389 -0
@@ -0,0 +1,274 @@
1
+ # Agentic Test Profile
2
+
3
+ Status: implemented public contract for `testforge-agentic-profile/1.0.0`.
4
+
5
+ ## Purpose
6
+
7
+ The Agentic Test Profile describes how much declared fault-detection evidence a test candidate
8
+ delivers within an observed execution budget. It is a derived, auditable report for developers,
9
+ agents, and CI systems. It does not change the meaning of an evidence manifest or its `VERIFIED`
10
+ decision.
11
+
12
+ The profile answers a bounded question:
13
+
14
+ > Among candidates that already produced valid AssertLedger evidence, which ones delivered the most
15
+ > declared fault-detection value within the repository's stated feedback budget?
16
+
17
+ It does not answer whether the program is correct, whether the supplied worlds are complete, or
18
+ whether a test will remain non-flaky in every future environment.
19
+
20
+ ## Testing mode
21
+
22
+ Version 1 profiles **hardening tests** only. A hardening test:
23
+
24
+ - passes on the operator-declared reference implementation;
25
+ - passes on required behaviour-preserving neutral worlds;
26
+ - produces attributed assertion failures on operator-declared target worlds; and
27
+ - is intended to remain in the repository to detect future regressions.
28
+
29
+ This is different from a **catching test**, which passes on a parent revision and intentionally
30
+ fails on a proposed revision to report a possible bug in that change. Catching tests have a
31
+ different reference direction, false-positive workflow, and lifecycle. They require a separate
32
+ future policy and must not be assigned a hardening profile.
33
+
34
+ ## Two independent layers
35
+
36
+ The evidence decision and efficiency profile remain separate:
37
+
38
+ ```text
39
+ deterministic evidence: VERIFIED | REJECTED | INCONCLUSIVE | ENGINE_ERROR
40
+ empirical profile: QUALIFIED | BUDGET_MISSED | INSUFFICIENT_TIMING_EVIDENCE | NOT_QUALIFIED
41
+ ```
42
+
43
+ An efficiency profile may be positive only when:
44
+
45
+ 1. the source manifest strictly validates;
46
+ 2. replay validates the schema, decision digest, artifact digest, and decision semantics;
47
+ 3. the campaign decision is `VERIFIED`; and
48
+ 4. every profiled candidate is selected and `ELIGIBLE`.
49
+
50
+ Latency never compensates for weak target strength, an invalid reference or neutral observation,
51
+ missing evidence, instability, or an unattributed failure.
52
+
53
+ ## Versioned profile policy
54
+
55
+ Thresholds are repository policy, not universal constants. A policy declares named feedback lanes:
56
+
57
+ ```json
58
+ {
59
+ "profileVersion": "1.0.0",
60
+ "profileId": "example-project/default",
61
+ "mode": "HARDENING",
62
+ "minimumTimingSamples": 5,
63
+ "lanes": [
64
+ { "id": "instant", "maximumReferenceP95Ms": 2000 },
65
+ { "id": "loop", "maximumReferenceP95Ms": 10000 },
66
+ { "id": "gate", "maximumReferenceP95Ms": 60000 }
67
+ ]
68
+ }
69
+ ```
70
+
71
+ Lane identifiers are descriptive labels scoped to the policy. AssertLedger does not claim that two
72
+ repositories using an `instant` lane are directly comparable. Lane thresholds must be positive,
73
+ strictly increasing, and use unique portable identifiers.
74
+
75
+ ## Report binding
76
+
77
+ The derived report binds:
78
+
79
+ - the complete, strictly validated source manifest, preserved in the report so profile replay does
80
+ not depend on a mutable side file;
81
+ - the source manifest's `artifactDigest`, which includes operational durations;
82
+ - the profile policy and its digest;
83
+ - the profile implementation version;
84
+ - candidate metrics and classifications; and
85
+ - a canonical report digest.
86
+
87
+ The evidence manifest v1 remains byte-for-byte and semantically unchanged. Adding the profile to
88
+ the manifest would break its strict schema and digest projections, so the profile is a separate
89
+ public artifact.
90
+
91
+ ## Metrics
92
+
93
+ ### Declared target strength
94
+
95
+ For a candidate `c`:
96
+
97
+ ```text
98
+ targetWeightPermille(c) =
99
+ 1000 * weight of target worlds killed by c / total target-world weight
100
+ ```
101
+
102
+ The profile metrics preserve required targets killed/total and all killed target identifiers. The
103
+ embedded source manifest remains the authoritative location for reference and neutral gate results,
104
+ evidence run identifiers, and candidate reason codes; the derived candidate rows do not duplicate
105
+ them. Code coverage may be reported by external tools as diagnostic context, but it is not a
106
+ qualification gate.
107
+
108
+ ### Observed consistency
109
+
110
+ The report uses `OBSERVED_CONSISTENT` or `OBSERVED_INCONSISTENT` plus the exact attempt count. It
111
+ never calls a test universally stable. Repeating a test three times records three consistent
112
+ observations; it does not establish a useful probabilistic upper bound on future failure. Any
113
+ contradictory attempt is preserved as observed inconsistency even though the candidate is already
114
+ ineligible for a positive profile.
115
+
116
+ ### Latency
117
+
118
+ Version 1 derives timing from candidate observations already recorded in the source manifest.
119
+ Candidate feedback latency is summarized from `REFERENCE` observations because those represent the
120
+ normal passing developer loop. The report includes:
121
+
122
+ - sample count;
123
+ - minimum and maximum;
124
+ - integer-arithmetic p50 and p95 using the nearest-rank method; and
125
+ - total recorded candidate execution time across all worlds and attempts.
126
+
127
+ Nearest rank sorts non-negative finite durations and selects index `ceil(p * n) - 1`. Fractional
128
+ adapter durations are preserved. Empty input has no quantile. The report is
129
+ `INSUFFICIENT_TIMING_EVIDENCE` when reference sample count is below the policy minimum.
130
+
131
+ These durations include the adapter process boundary measured by the current engine. They do not
132
+ distinguish cold start, warm execution, compilation, collection, or test-body time. Therefore v1
133
+ calls them `RECORDED_REFERENCE_WALL_TIME` and makes no cold/warm claim.
134
+
135
+ ### Marginal evidence
136
+
137
+ For selected candidates in deterministic selection order, the report includes the target weight
138
+ first covered by that candidate. A candidate that only repeats already-covered targets has zero
139
+ marginal weight even if its individual target score is high.
140
+
141
+ ### Pareto status
142
+
143
+ Pareto comparison is performed only among selected `ELIGIBLE` candidates with sufficient timing
144
+ evidence, including candidates that miss every declared lane. Candidate A dominates B when A is no
145
+ worse on all of these dimensions and strictly better on at least one:
146
+
147
+ - declared target weight killed: greater is better;
148
+ - required targets killed: greater is better;
149
+ - reference p95 wall time: lower is better;
150
+ - total recorded candidate wall time: lower is better; and
151
+ - candidate size in bytes: lower is better.
152
+
153
+ Candidates without sufficient timing evidence are not placed on the timed frontier. The report
154
+ publishes the frontier; it does not collapse the dimensions into a single opaque score.
155
+
156
+ ## Classification
157
+
158
+ Classification is applied in this order:
159
+
160
+ Invalid schema or replay evidence is rejected before a report is produced. For a valid source,
161
+ classification is applied in this order:
162
+
163
+ 1. `NOT_QUALIFIED` when the campaign is not `VERIFIED`, or a candidate is not both selected and
164
+ `ELIGIBLE`.
165
+ 2. `INSUFFICIENT_TIMING_EVIDENCE` when the evidence gates pass but the declared minimum number of
166
+ reference timing samples is not present.
167
+ 3. `BUDGET_MISSED` when evidence is valid but the candidate satisfies no declared latency lane.
168
+ 4. `QUALIFIED` when evidence is valid, timing evidence is sufficient, and at least one lane is
169
+ satisfied.
170
+
171
+ The report lists every satisfied lane and a `bestLaneId`, defined as the smallest declared budget
172
+ the candidate satisfies. `QUALIFIED` is always scoped to declared worlds, attempts, environment,
173
+ and profile policy.
174
+
175
+ ## Portfolio selection under a budget
176
+
177
+ Each lane also yields a deterministic weighted-coverage portfolio under that lane's wall-time
178
+ budget. Selection operates only after evidence qualification:
179
+
180
+ 1. compute uncovered target weight contributed by each remaining eligible candidate;
181
+ 2. discard candidates whose recorded p95 exceeds the remaining budget;
182
+ 3. select the greatest marginal target weight per unit of p95 cost;
183
+ 4. break ties by greater absolute marginal weight, lower p95, lower candidate size, digest, then id;
184
+ 5. stop when no candidate adds target weight within the remaining budget.
185
+
186
+ The report records selected identifiers, summed reference p95 cost, covered target weight, required
187
+ targets, and coverage permille for every lane. The greedy portfolio is an explainable baseline, not
188
+ a proof of a globally optimal subset. The full Pareto frontier remains available so callers can make
189
+ another policy choice without falsifying the recorded evidence.
190
+
191
+ APFDc may be emitted as a diagnostic for a declared sequential order with declared target severity.
192
+ It is not the primary score because parallel execution and partial portfolios violate the simple
193
+ sequential interpretation.
194
+
195
+ ## Calibration and anti-gaming
196
+
197
+ A public profile is no stronger than its world portfolio. A mature evaluation corpus should combine:
198
+
199
+ - historical real faults with independently verified fixes;
200
+ - operator-reviewed synthetic mutants;
201
+ - behaviour-preserving neutral rewrites;
202
+ - boundary and metamorphic properties; and
203
+ - explicit dispositions for equivalent, redundant, disputed, or out-of-scope mutants.
204
+
205
+ Local development may use fully visible worlds. Competitive or certification-like evaluation needs
206
+ an independently governed holdout. A commit-reveal workflow can publish the corpus digest before a
207
+ campaign and reveal the worlds afterward. Even then, the report claims only performance on that
208
+ corpus.
209
+
210
+ The initial product hypotheses are falsifiable:
211
+
212
+ - H1: at equal declared target strength, Pareto selection reduces reference-loop p95 versus the full
213
+ eligible suite.
214
+ - H2: candidates required to pass neutral worlds survive behaviour-preserving rewrites more often
215
+ than reference-only candidates.
216
+ - H3: under equal time budgets, the qualified portfolio detects no fewer held-out historical faults
217
+ than a coverage-guided baseline.
218
+ - H4: a reviewed reduced mutant portfolio retains non-inferior historical-fault recall at lower
219
+ execution cost than the exhaustive mutant set.
220
+
221
+ If H3 or H4 fails, the project must describe the output as an audited declared-world profile, not as
222
+ an empirically established testing sweet spot.
223
+
224
+ The executable corpus gate lives under `benchmarks/agentic-profile`. It requires at least 20 strict
225
+ cases from three immutable sources, a non-empty private holdout, and evaluable evidence for H1-H4.
226
+ Until those gates pass, the scorer returns `NOT_READY` and no optimization loop is authorized.
227
+ Public evaluation exposes exact reason codes; holdout evaluation exposes aggregate outcomes only.
228
+
229
+ ## Provenance maturity
230
+
231
+ The source manifest does not fingerprint the complete host, effective environment, or dependency
232
+ graph, and structured-command adapters remain part of the trusted computing base. The separate
233
+ Agentic Benchmark Artifact now binds these declarations and a strict comparison scope. It does not
234
+ authenticate them, and the built-in `node:test` reporter cannot yet acquire the four phase markers.
235
+ Exact scoped timing therefore requires a trustworthy producer containing at least:
236
+
237
+ - operating-system and architecture identity;
238
+ - CPU and resource-limit identity;
239
+ - runtime, adapter, dependency, and executable digests;
240
+ - cold/warm preparation protocol;
241
+ - raw per-phase samples; and
242
+ - drift checks against a calibrated workload.
243
+
244
+ Until that artifact exists, v1 profile timing is local observed evidence, not a portable benchmark.
245
+
246
+ ## Research basis
247
+
248
+ - [Just et al., mutants and real faults, FSE 2014](https://homes.cs.washington.edu/~mernst/pubs/mutation-effectiveness-fse2014-abstract.html)
249
+ found mutation detection correlated with real-fault detection independently of code coverage,
250
+ while documenting important limitations.
251
+ - [Elbaum, Rothermel, and Penix, regression testing in CI, FSE 2014](https://research.google/pubs/techniques-for-improving-regression-testing-in-continuous-integration-development-environments/)
252
+ treats fast feedback and test cost as first-class CI objectives.
253
+ - [Meta predictive test selection](https://engineering.fb.com/2018/11/21/developer-tools/predictive-test-selection/)
254
+ demonstrates large-scale cost reduction while explicitly calibrating missed-regression and flaky
255
+ test risk.
256
+ - [Kaufman et al., TCAP mutant prioritization, ICSE 2022](https://doi.org/10.1145/3510003.3510187)
257
+ argues that a mutant is useful insofar as it elicits a test that advances test completeness.
258
+ - [Meta Mutation-Guided LLM-based Test Generation](https://arxiv.org/abs/2501.12862) uses a small,
259
+ concern-specific mutant portfolio to guide production test generation.
260
+ - [Meta Just-in-Time Catching Test Generation](https://arxiv.org/abs/2601.22832) distinguishes
261
+ hardening tests from change-specific catching tests and shows why false-positive drag must be
262
+ measured independently from raw catch generation.
263
+
264
+ ## Explicit non-claims
265
+
266
+ An Agentic Test Profile does not prove:
267
+
268
+ - program correctness or complete fault detection;
269
+ - semantic correctness or completeness of operator-supplied worlds;
270
+ - permanent absence of flakiness;
271
+ - portability of recorded timing to another machine;
272
+ - resistance to a candidate designed with knowledge of visible worlds;
273
+ - provenance authenticity without an external attestation; or
274
+ - that a greedy portfolio is globally optimal.
@@ -0,0 +1,157 @@
1
+ # Architecture
2
+
3
+ AssertLedger separates proposal, integration, orchestration, and decision authority. The separation
4
+ keeps model-specific behavior out of the evidence policy and process I/O out of the deterministic
5
+ core.
6
+
7
+ ## Four layers
8
+
9
+ | Layer | Responsibilities | Must not do |
10
+ | --- | --- | --- |
11
+ | Agent or harness | Analyze context, propose candidate test files, submit requests | Decide that its own tests are valid |
12
+ | Integration facade | Translate Skill, MCP, JSON CLI, or SDK calls into shared use cases | Reimplement gates or selection |
13
+ | AssertLedger orchestrator | Snapshot repositories, enforce budgets, create workspaces, run adapters, normalize observations | Grant target credit outside the core |
14
+ | Deterministic core | Canonicalize evidence, evaluate gates, select candidates, seal manifests | Read files, spawn processes, use the network, clock, randomness, or a model |
15
+
16
+ Dependencies point downward:
17
+
18
+ ```text
19
+ agent or harness
20
+ │
21
+ ▼
22
+ facade: skill / MCP / CLI / SDK
23
+ │
24
+ ▼
25
+ orchestrator: repository / execution / adapters / budgets
26
+ │
27
+ ▼
28
+ core: contracts / gates / selection / digests
29
+ ```
30
+
31
+ The concrete source boundaries are `src/contracts`, `src/core`, `src/engine`, and the thin
32
+ `src/sdk`, `src/cli.ts`, and `src/mcp` facades. `src/contracts` and `src/core` form the deterministic
33
+ core layer: contracts own the versioned wire request, and core owns decision authority.
34
+
35
+ ## Campaign data flow
36
+
37
+ ```text
38
+ versioned request
39
+ ├─ repository root and exclusions
40
+ ├─ operator-owned worlds and provenance
41
+ ├─ candidate test overlays
42
+ ├─ adapter configuration
43
+ ├─ trusted-local authorization and environment allowlist
44
+ └─ policy and budgets
45
+ │
46
+ ▼
47
+ single repository snapshot and repository digest
48
+ │
49
+ ├─ control × world × attempt
50
+ └─ candidate × world × attempt
51
+ │
52
+ ▼
53
+ fresh workspace per execution
54
+ world overlay → candidate overlay → adapter process
55
+ │
56
+ ▼
57
+ normalized observations
58
+ │
59
+ ▼
60
+ pure gate evaluation → deterministic selection → sealed manifest
61
+ ```
62
+
63
+ The engine copies the source repository once into a campaign snapshot. Every execution receives a
64
+ fresh copy of that snapshot. World overlays apply first; candidate overlays apply second. Candidate
65
+ files must remain under a configured `candidateRoots` path.
66
+
67
+ After request parsing and repository-root resolution, the engine validates source file and byte
68
+ budgets before any runtime probe or preflight. It validates the copied snapshot again before
69
+ campaign execution. See [validation error precedence](migration-repository-validation-order.md).
70
+
71
+ The request separates budget scopes:
72
+
73
+ - campaign-wide: `maximumCandidates`, `maximumWorlds`, `maximumExecutions`,
74
+ `maximumRepositoryFiles`, `maximumRepositoryBytes`, `maximumWorldOverlayBytes`, and
75
+ `maximumTotalCandidateBytes`;
76
+ - per candidate: `maximumCandidateBytes`;
77
+ - per execution: `timeoutMsPerExecution` and `maximumOutputBytes`, applied separately to captured
78
+ stdout and stderr.
79
+
80
+ Local timeout enforcement and process-tree termination are best effort host operations, not
81
+ containment.
82
+
83
+ ## Worlds and controls
84
+
85
+ - `REFERENCE`: code the operator expects to be correct; an eligible candidate must pass it.
86
+ - `TARGET`: code containing an operator-declared fault; a strong candidate must produce an attributed
87
+ assertion failure according to policy.
88
+ - `NEUTRAL`: an operator-declared irrelevant or equivalent variation; an eligible candidate must
89
+ continue to pass it.
90
+
91
+ The engine runs every world without a candidate. All controls must complete, remain stable, report
92
+ `PASS`, discover no candidate tests, and claim no candidate attribution. Invalid controls make the
93
+ campaign `INCONCLUSIVE`; a candidate cannot receive credit for a world that was already broken.
94
+
95
+ AssertLedger does not generate mutants in v0.1. Mutant generators may produce operator-reviewed target
96
+ worlds, but they never gain decision authority.
97
+
98
+ ## Determinism boundary
99
+
100
+ Process execution is not deterministic by construction. Scheduling, timing, caches, filesystem
101
+ behavior, dependencies, and the host can change observed runs. AssertLedger guarantees a narrower
102
+ invariant:
103
+
104
+ ```text
105
+ same versioned policy + same normalized evidence = same decision
106
+ ```
107
+
108
+ The core sorts decision inputs before evaluation and excludes declared host-volatile observation
109
+ fields from the decision digest. It does not make nondeterministic executions deterministic. Repeat
110
+ attempts reveal only instability observed within the configured attempt budget.
111
+
112
+ ## Selection
113
+
114
+ The core removes ineligible candidates, then selects candidates by deterministic marginal target
115
+ coverage. Ties use candidate size, content digest, and identifier, in that order. Runtime is not a
116
+ tie-breaker because it depends on the host.
117
+
118
+ ## Provenance and integrity
119
+
120
+ The evidence context binds the engine identity, adapter configuration, achieved isolation level,
121
+ environment allowlist names, budgets, candidate roots, and each world's provenance and overlay
122
+ digest to the decision. It does not record effective environment values. For `node:test`, the engine
123
+ probes the requested executable, resolves its real path, probes the resolved file again, and records
124
+ the matching Node.js version plus the executable's SHA-256 digest. The structured-command adapter
125
+ records its configured command but does not resolve or hash it.
126
+
127
+ The repository digest covers regular files visited by the analyzer. The analyzer omits `.git`,
128
+ `.testforge`, `node_modules`, and operator-excluded path segments. It rejects any encountered
129
+ symbolic link instead of silently excluding it. Candidate contents and world overlays have separate
130
+ digests. Preserve the snapshot and resolved dependency identities for stronger provenance.
131
+
132
+ `detectedTestFrameworks` remains a sorted string array in repository-analysis v1 so this correction
133
+ does not change the schema or existing wire consumers. Detection is deliberately conservative:
134
+ JavaScript frameworks require a dependency declaration or dedicated config file, pytest requires a
135
+ Python dependency/config declaration, and `node:test` requires an import in a file below a recognized
136
+ test directory or in a colocated `*.test.*`/`*.spec.*` file. A lightweight lexer ignores comments,
137
+ quoted examples, and template literals; config filenames must use a supported JavaScript,
138
+ TypeScript, or JSON extension. These rules favor precision over recall, so an unrecognized manifest
139
+ layout or naming convention can yield a false negative. The field does not expose per-framework
140
+ provenance in v1; when evidence is absent, an empty list is more truthful than an inference. A future
141
+ provenance-bearing shape requires an additive schema version.
142
+
143
+ The decision digest covers decision-relevant evidence. The artifact digest covers the emitted
144
+ manifest, including operational fields and disclosure metadata. Both are integrity checks, not
145
+ signatures; neither authenticates the producer. See [proof-model.md](proof-model.md).
146
+
147
+ The emitted manifest contains normalized observations and stdout/stderr digests, not raw process
148
+ output, candidate or world file bodies, or a repository archive. A self-contained audit bundle must
149
+ preserve those inputs and logs beside the manifest.
150
+
151
+ ## Extension points
152
+
153
+ - Built-in adapters translate framework behavior into normalized observations.
154
+ - The structured-command protocol supports external test-framework reporters.
155
+ - A future isolation backend may replace `trusted-local` without changing gate ownership.
156
+ - A future native orchestrator may replace the Node.js engine only after passing shared schema,
157
+ canonicalization, decision, and digest conformance fixtures.
package/docs/ci.md ADDED
@@ -0,0 +1,37 @@
1
+ # CI and trusted-local execution
2
+
3
+ The repository workflow runs the full checks on Windows/Linux with Node 22/24, then tests
4
+ the packed distribution on Windows/Linux with Node 22.15/24. The package job retains
5
+ manifests, summaries, replay results, command logs and the tarball for 14 days. Artifact names
6
+ contain the tested GitHub SHA and matrix values.
7
+
8
+ Both jobs execute repository code. They run on disposable GitHub-hosted runners with
9
+ `contents: read`, no deployment secrets, and checkout credential persistence disabled.
10
+ Never move trusted-local verification onto a privileged or self-hosted runner handling
11
+ untrusted contributions. AssertLedger does not contain hostile code.
12
+
13
+ The workflow permits pushes to `main` and same-repository pull requests from an owner,
14
+ member or collaborator. Fork pull requests are skipped. A skipped job does not establish
15
+ compatibility or release readiness: a maintainer must inspect the exact contribution and
16
+ bring the reviewed changes onto a trusted repository branch before executing them.
17
+
18
+ This is the project's contribution policy, not a sandbox or an immutable security boundary:
19
+ workflow definitions themselves must be reviewed, and GitHub's repository permission and
20
+ workflow-approval settings remain operator-owned controls. Do not use `pull_request_target`
21
+ to execute a contributor's checkout with elevated authority.
22
+
23
+ ## Reuse the result
24
+
25
+ Open the package job's `package-evidence-*` artifact. Start with `historical-strong-summary.md`,
26
+ then inspect the corresponding manifest and `historical-provenance.json`. Run:
27
+
28
+ ```sh
29
+ assertledger replay historical-strong-manifest.json --json
30
+ ```
31
+
32
+ `historical-checkout-preserved.json` records the unchanged source files, index hash and HEAD.
33
+ The weak and generic-crash cases must be rejected, with valid replay. These results prove
34
+ the observed supported case; they do not authenticate an arbitrary evidence producer.
35
+
36
+ Keep release evidence outside the temporary CI retention window when publishing a version.
37
+ Never upload private proof-policy copies, credentials or unrelated workspace artifacts.
@@ -0,0 +1,61 @@
1
+ # Project-local client connections
2
+
3
+ AssertLedger can prepare a read-only MCP connection and install its packaged skill inside one
4
+ repository. Run these commands from the installed package after `dist/cli.js` has been built.
5
+
6
+ ## Preview first
7
+
8
+ `connect` and `disconnect` are previews unless `--write` is present. A connection preview lists
9
+ every managed path and prints the complete planned contents between explicit delimiters. A removal
10
+ preview lists the paths that would be checked. Neither preview changes the repository.
11
+
12
+ ```console
13
+ assertledger connect . --client codex
14
+ assertledger connect . --client claude-code
15
+ assertledger disconnect . --client codex
16
+ ```
17
+
18
+ Use `--write` only after reviewing the paths:
19
+
20
+ ```console
21
+ assertledger connect . --client codex --write
22
+ assertledger disconnect . --client codex --write
23
+ ```
24
+
25
+ The operation is idempotent. Reinstalling identical artifacts reports `UNCHANGED`; removing an
26
+ already absent installation reports `ABSENT`.
27
+
28
+ ## Managed artifacts
29
+
30
+ | Client | MCP configuration | Packaged skill |
31
+ | --- | --- | --- |
32
+ | Codex | `.codex/config.toml` | `.agents/skills/assertledger/SKILL.md` |
33
+ | Claude Code | `.mcp.json` | `.claude/skills/assertledger/SKILL.md` |
34
+
35
+ The skill is copied byte-for-byte from the versioned `integrations/skill/SKILL.md` included in the
36
+ installed AssertLedger package, including its frontmatter.
37
+
38
+ Both configurations start `assertledger mcp --root <canonical-repository-root>`. The default server
39
+ is read-only. Installing a connection does not authorize candidate execution, trust the repository,
40
+ or reload the client; those remain explicit operator actions.
41
+
42
+ AssertLedger creates a managed file only when it is absent. If any target already contains different
43
+ content, the complete operation reports `CONFLICT` before writing anything. Linked paths,
44
+ directories in place of files, and other non-regular targets are refused.
45
+
46
+ `disconnect --write` removes only files that still match the exact bytes shipped or generated by the
47
+ current AssertLedger installation. If a person or another tool changed either file, the complete
48
+ removal reports `CONFLICT` and leaves every artifact in place. Directories are never removed.
49
+
50
+ ## Generic MCP descriptor
51
+
52
+ Other MCP clients can request a transport descriptor:
53
+
54
+ ```console
55
+ assertledger connect . --client mcp
56
+ ```
57
+
58
+ The command prints a JSON object with `transport`, `command`, `args`, and `cwd`. AssertLedger does not
59
+ write this descriptor because MCP clients do not share a universal project configuration path or
60
+ file format. Copy the fields into the client-specific configuration after checking that client's
61
+ documentation.
@@ -0,0 +1,72 @@
1
+ # AssertLedger conformance bundle v1
2
+
3
+ `conformance/v1/` is a checked-in compatibility oracle. It contains autonomous JSON inputs and
4
+ complete expected JSON outputs for canonicalization, evidence decisions and replay, Profile v1,
5
+ and Benchmark v1. The bundle includes positive behavior and negative security witnesses.
6
+
7
+ The published npm package includes this static bundle for inspection and interoperability. The
8
+ maintainer commands below require a source checkout with development dependencies; package scripts
9
+ are repository-maintenance metadata, not an installed-package verification API.
10
+
11
+ `pnpm run check:conformance` never generates fixtures. It first verifies the exact file set, every
12
+ raw file SHA-256, and this root digest construction over ordinally sorted portable paths:
13
+
14
+ ```text
15
+ SHA256(path + NUL + "sha256:" + raw-hex-digest + newline)
16
+ ```
17
+
18
+ It then verifies the raw SHA-256 and `$id` of all 34 published schemas directly from `schemas/`,
19
+ strictly parses `bundle.json`, dispatches an explicit operation allowlist to public AssertLedger APIs,
20
+ and deep-compares every complete expected output. Additional assertions ensure compilation,
21
+ collection, timeout, process-crash, infrastructure-error, and no-test-discovered outcomes never
22
+ become target kills, and that re-sealed semantic forgeries remain invalid even when their public
23
+ digest rails are true.
24
+
25
+ ## Negative witness runner
26
+
27
+ `pnpm run witness:conformance` exercises the checker against seven mutations in an OS-temporary copy,
28
+ never against the checked-in candidate. It first proves the copied oracle green, then verifies that:
29
+
30
+ 1. changing one fixture byte fails the raw-file lock; and
31
+ 2. replacing each of COMPILE_FAILURE, COLLECTION_FAILURE, TIMEOUT, PROCESS_CRASH, INFRA_ERROR, and
32
+ NO_TEST_DISCOVERED, together with its expected output, with VERIFIED semantics still fails the
33
+ manually written non-kill assertion, even after the temporary raw locks and root are resealed.
34
+
35
+ The runner recreates the temporary copy between mutants, verifies a final green run, removes the
36
+ temporary directory, and emits one JSON receipt containing the before/after raw digests, red and
37
+ green exit codes, and output digests. It rejects any mutation target inside the repository. Because
38
+ the runner is deliberately mutating and more expensive, it is an explicit audit command and is not
39
+ part of `pnpm check`.
40
+
41
+ ## Maintainer regeneration
42
+
43
+ The generator requires an explicit output directory and normally writes only to a temporary or
44
+ review directory:
45
+
46
+ ```sh
47
+ pnpm exec tsx scripts/generate-conformance-v1.ts --output-dir ./tmp/conformance-candidate
48
+ ```
49
+
50
+ It refuses `conformance/v1` unless `--maintainer-allow-canonical` is present and never writes
51
+ `scripts/conformance-v1-lock.ts`. Updating the public oracle therefore requires an intentional
52
+ sequence:
53
+
54
+ 1. generate into a temporary directory and review the semantic delta;
55
+ 2. obtain a fresh proof-integrity review of changed operations and witnesses;
56
+ 3. explicitly regenerate the canonical directory with the maintainer flag;
57
+ 4. format the final JSON bytes;
58
+ 5. independently calculate and manually update raw file hashes, the root digest, and locked public
59
+ decision/artifact/profile/benchmark digests;
60
+ 6. run `pnpm check` and review the final aggregate diff.
61
+
62
+ Any fixture or lock change is a public compatibility migration and must be documented. Do not
63
+ accept a regenerated lock merely because the new checker is green: the checked-in oracle is an
64
+ initial assumption, not an independent source of truth.
65
+
66
+ ## Scope and non-claims
67
+
68
+ The bundle establishes deterministic compatibility with these checked-in examples and preserves
69
+ named negative witnesses. It does not prove that the initial expected outputs are correct, that all
70
+ semantic branches are covered, that execution evidence is truthful, or that a self-modifying
71
+ change deserves acceptance. A fresh human or independent proof audit remains required whenever
72
+ the fixtures, lock, checker, schemas, digest projections, or decision semantics change.
@@ -0,0 +1,24 @@
1
+ # ADR 0001: TypeScript runtime for the first vertical release
2
+
3
+ Status: accepted for v0.1
4
+
5
+ ## Decision
6
+
7
+ Implement the first complete AssertLedger release in strict TypeScript on maintained Node.js releases.
8
+
9
+ ## Rationale
10
+
11
+ The first release must deliver a pure decision core, JSON schemas, process orchestration, SDK, CLI,
12
+ and current MCP integration together. TypeScript makes those integration surfaces directly usable by
13
+ agent harnesses while preserving a strict protocol boundary.
14
+
15
+ Rust would improve native distribution and OS process control. The schema and CLI protocols are
16
+ therefore versioned so a native orchestrator can replace the Node implementation without moving gate
17
+ authority or changing consumers.
18
+
19
+ ## Consequences
20
+
21
+ - Node.js 22.15 or newer is a runtime dependency.
22
+ - `trusted-local` process-tree termination is best effort, especially on Windows.
23
+ - The core remains I/O-free and portable at the algorithmic level.
24
+ - A future Rust backend must prove byte-compatible canonical decisions with shared fixtures.