@tangle-network/agent-eval 0.180.0 → 0.181.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (221) hide show
  1. package/CHANGELOG.md +60 -0
  2. package/README.md +119 -159
  3. package/dist/adapters/http.d.ts +2 -2
  4. package/dist/{agent-profile-B7yErX0q.d.ts → agent-profile-CivaSsSy.d.ts} +4 -4
  5. package/dist/{agent-profile-B7yErX0q.d.ts.map → agent-profile-CivaSsSy.d.ts.map} +1 -1
  6. package/dist/{agent-profile-cell-0gSi5ffD.js → agent-profile-cell-Cv6UA-W_.js} +20 -57
  7. package/dist/agent-profile-cell-Cv6UA-W_.js.map +1 -0
  8. package/dist/{agent-profile-cell-CTOZJUuE.d.ts → agent-profile-cell-s__adRnK.d.ts} +3 -3
  9. package/dist/agent-profile-cell-s__adRnK.d.ts.map +1 -0
  10. package/dist/analyst/index.d.ts +10 -10
  11. package/dist/analyst/index.js +3 -3
  12. package/dist/ast-CP9ae9B0.js +557 -0
  13. package/dist/ast-CP9ae9B0.js.map +1 -0
  14. package/dist/ast-hI-vjW6J.d.ts +457 -0
  15. package/dist/ast-hI-vjW6J.d.ts.map +1 -0
  16. package/dist/{benchmark-command-D4vpnAdO.js → benchmark-command-B57n9vjz.js} +7 -6
  17. package/dist/{benchmark-command-D4vpnAdO.js.map → benchmark-command-B57n9vjz.js.map} +1 -1
  18. package/dist/benchmarks/index.d.ts +4 -4
  19. package/dist/benchmarks/index.js +3 -3
  20. package/dist/campaign/index.d.ts +6 -6
  21. package/dist/campaign/index.js +8 -8
  22. package/dist/{campaign-BGEurASO.js → campaign-4_ppJW5X.js} +12 -12
  23. package/dist/{campaign-BGEurASO.js.map → campaign-4_ppJW5X.js.map} +1 -1
  24. package/dist/{campaign-evidence-D8DBLqLI.js → campaign-evidence-B8oF9xQ6.js} +515 -471
  25. package/dist/campaign-evidence-B8oF9xQ6.js.map +1 -0
  26. package/dist/cli.js +4 -7
  27. package/dist/cli.js.map +1 -1
  28. package/dist/{client-BlLY6o2w.js → client-CXE-U1SA.js} +3 -1
  29. package/dist/client-CXE-U1SA.js.map +1 -0
  30. package/dist/{client-CuQgX33c.d.ts → client-kh2jOjTK.d.ts} +4 -4
  31. package/dist/{client-CuQgX33c.d.ts.map → client-kh2jOjTK.d.ts.map} +1 -1
  32. package/dist/contract/index.d.ts +13 -13
  33. package/dist/contract/index.js +10 -9
  34. package/dist/contract/index.js.map +1 -1
  35. package/dist/{default-registry-IGDE9XIC.d.ts → default-registry-BwDSWVzg.d.ts} +6 -6
  36. package/dist/{default-registry-IGDE9XIC.d.ts.map → default-registry-BwDSWVzg.d.ts.map} +1 -1
  37. package/dist/{define-agent-eval-Cx4Ls9ta.d.ts → define-agent-eval-CwOWWQt_.d.ts} +33 -12
  38. package/dist/define-agent-eval-CwOWWQt_.d.ts.map +1 -0
  39. package/dist/{define-agent-eval-Dzidv34q.js → define-agent-eval-Ddu33JH9.js} +134 -67
  40. package/dist/define-agent-eval-Ddu33JH9.js.map +1 -0
  41. package/dist/{dspy-rlm-engine-xKiWmj_G.js → dspy-rlm-engine-S53V0HhE.js} +2 -2
  42. package/dist/{dspy-rlm-engine-xKiWmj_G.js.map → dspy-rlm-engine-S53V0HhE.js.map} +1 -1
  43. package/dist/{engine-CX8ReXkn.d.ts → engine-DS1cysJy.d.ts} +10 -7
  44. package/dist/engine-DS1cysJy.d.ts.map +1 -0
  45. package/dist/{eval-campaign-Cs-7MiCs.js → eval-campaign-aYdtjtJR.js} +4 -4
  46. package/dist/{eval-campaign-Cs-7MiCs.js.map → eval-campaign-aYdtjtJR.js.map} +1 -1
  47. package/dist/{exact-types-B7LC1EyX.d.ts → exact-types-BZDe0W2D.d.ts} +2 -2
  48. package/dist/{exact-types-B7LC1EyX.d.ts.map → exact-types-BZDe0W2D.d.ts.map} +1 -1
  49. package/dist/experiment/index.d.ts +27 -477
  50. package/dist/experiment/index.d.ts.map +1 -1
  51. package/dist/experiment/index.js +95 -559
  52. package/dist/experiment/index.js.map +1 -1
  53. package/dist/{experiment-tracker-B3TiF5-u.d.ts → experiment-tracker-C7PfnF4b.d.ts} +2 -2
  54. package/dist/{experiment-tracker-B3TiF5-u.d.ts.map → experiment-tracker-C7PfnF4b.d.ts.map} +1 -1
  55. package/dist/{external-optimizer-process-Dlz8YxrT.js → external-optimizer-process-QDRURJAM.js} +3 -3
  56. package/dist/{external-optimizer-process-Dlz8YxrT.js.map → external-optimizer-process-QDRURJAM.js.map} +1 -1
  57. package/dist/{external-optimizer-subprocess-q3VzlGAO.js → external-optimizer-subprocess-D4dzUBZI.js} +3 -2
  58. package/dist/{external-optimizer-subprocess-q3VzlGAO.js.map → external-optimizer-subprocess-D4dzUBZI.js.map} +1 -1
  59. package/dist/{feedback-trajectory-eHWNv5Aj.d.ts → feedback-trajectory-CXmtITBo.d.ts} +3 -3
  60. package/dist/{feedback-trajectory-eHWNv5Aj.d.ts.map → feedback-trajectory-CXmtITBo.d.ts.map} +1 -1
  61. package/dist/hosted/index.d.ts +2 -2
  62. package/dist/hosted/index.d.ts.map +1 -1
  63. package/dist/hosted/index.js +1 -1
  64. package/dist/{index-BxWvILU8.d.ts → index-Bp_6sj3x.d.ts} +109 -56
  65. package/dist/index-Bp_6sj3x.d.ts.map +1 -0
  66. package/dist/{index-e7LXeRVa.d.ts → index-CJ3LhKIX.d.ts} +2 -2
  67. package/dist/{index-e7LXeRVa.d.ts.map → index-CJ3LhKIX.d.ts.map} +1 -1
  68. package/dist/{index-CiUjjEIa.d.ts → index-DNntP4ch.d.ts} +7 -7
  69. package/dist/{index-CiUjjEIa.d.ts.map → index-DNntP4ch.d.ts.map} +1 -1
  70. package/dist/{index-DxNYmx4a.d.ts → index-DoykkxW0.d.ts} +11 -11
  71. package/dist/{index-DxNYmx4a.d.ts.map → index-DoykkxW0.d.ts.map} +1 -1
  72. package/dist/index.d.ts +28 -28
  73. package/dist/index.js +24 -15
  74. package/dist/index.js.map +1 -1
  75. package/dist/{insight-report-DETqPc_A.d.ts → insight-report-D1qa0HWs.d.ts} +9 -5
  76. package/dist/{insight-report-DETqPc_A.d.ts.map → insight-report-D1qa0HWs.d.ts.map} +1 -1
  77. package/dist/{integrity-BKTcA-HP.d.ts → integrity-rGOfSUle.d.ts} +2 -2
  78. package/dist/{integrity-BKTcA-HP.d.ts.map → integrity-rGOfSUle.d.ts.map} +1 -1
  79. package/dist/{ledger-core-Cs9f7385.js → journal-Cs9f7385.js} +1 -1
  80. package/dist/journal-Cs9f7385.js.map +1 -0
  81. package/dist/{judge-calibration-C5CbMYce.d.ts → judge-calibration-DFtEMlde.d.ts} +31 -2
  82. package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
  83. package/dist/{judge-calibration-BnpVKtnb.js → judge-calibration-DYmaBtJr.js} +48 -2
  84. package/dist/{judge-calibration-BnpVKtnb.js.map → judge-calibration-DYmaBtJr.js.map} +1 -1
  85. package/dist/ledger-core/index.d.ts +1 -1
  86. package/dist/ledger-core/index.js +1 -1
  87. package/dist/{llm-judge-v80Kmu9g.js → llm-judge-DEFZeSiu.js} +645 -456
  88. package/dist/llm-judge-DEFZeSiu.js.map +1 -0
  89. package/dist/{matrix-DeMmnWrP.d.ts → matrix-CyhW-vgJ.d.ts} +2 -2
  90. package/dist/{matrix-DeMmnWrP.d.ts.map → matrix-CyhW-vgJ.d.ts.map} +1 -1
  91. package/dist/meta-eval/index.d.ts +138 -7
  92. package/dist/meta-eval/index.d.ts.map +1 -1
  93. package/dist/meta-eval/index.js +245 -97
  94. package/dist/meta-eval/index.js.map +1 -1
  95. package/dist/{mint-Cc1_zwRQ.js → mint-ySIIkKlV.js} +2 -2
  96. package/dist/{mint-Cc1_zwRQ.js.map → mint-ySIIkKlV.js.map} +1 -1
  97. package/dist/multishot/golden/index.d.ts +1 -1
  98. package/dist/multishot/index.d.ts +2 -2
  99. package/dist/openapi.json +1 -1
  100. package/dist/outcome-store-BXlkwMPR.js +131 -0
  101. package/dist/outcome-store-BXlkwMPR.js.map +1 -0
  102. package/dist/{outcome-store-BYHIuO0e.d.ts → outcome-store-CNt4iZ67.d.ts} +18 -25
  103. package/dist/outcome-store-CNt4iZ67.d.ts.map +1 -0
  104. package/dist/{paired-promotion-decision-CGzg0cI_.d.ts → paired-promotion-decision-DPsMQm-0.d.ts} +13 -7
  105. package/dist/{paired-promotion-decision-CGzg0cI_.d.ts.map → paired-promotion-decision-DPsMQm-0.d.ts.map} +1 -1
  106. package/dist/pipelines/index.js +1 -1
  107. package/dist/{produced-state-Cv0kJJuP.js → produced-state-BHboMaab.js} +3 -3
  108. package/dist/{produced-state-Cv0kJJuP.js.map → produced-state-BHboMaab.js.map} +1 -1
  109. package/dist/profile-cell.d.ts +1 -1
  110. package/dist/profile-cell.js +1 -1
  111. package/dist/{promotion-policy-DWOm70gx.js → promotion-policy-CDMMxzb6.js} +28 -40
  112. package/dist/promotion-policy-CDMMxzb6.js.map +1 -0
  113. package/dist/{registry-ByVld1-5.d.ts → registry-BRbB6Y0v.d.ts} +4 -4
  114. package/dist/{registry-ByVld1-5.d.ts.map → registry-BRbB6Y0v.d.ts.map} +1 -1
  115. package/dist/{release-confidence-BAcNYOf1.d.ts → release-confidence-BcqeQTHW.d.ts} +3 -3
  116. package/dist/{release-confidence-BAcNYOf1.d.ts.map → release-confidence-BcqeQTHW.d.ts.map} +1 -1
  117. package/dist/{release-confidence-BcGCclTB.js → release-confidence-DMg8n18l.js} +2 -2
  118. package/dist/{release-confidence-BcGCclTB.js.map → release-confidence-DMg8n18l.js.map} +1 -1
  119. package/dist/reporting.d.ts +4 -4
  120. package/dist/reporting.js +3 -3
  121. package/dist/{researcher-jsW1X94L.d.ts → researcher-64T49THL.d.ts} +6 -6
  122. package/dist/{researcher-jsW1X94L.d.ts.map → researcher-64T49THL.d.ts.map} +1 -1
  123. package/dist/{reward-hacking-ZXEi9VCq.d.ts → reward-hacking-uzO_ihep.d.ts} +2 -2
  124. package/dist/{reward-hacking-ZXEi9VCq.d.ts.map → reward-hacking-uzO_ihep.d.ts.map} +1 -1
  125. package/dist/rl.d.ts +53 -99
  126. package/dist/rl.d.ts.map +1 -1
  127. package/dist/rl.js +182 -169
  128. package/dist/rl.js.map +1 -1
  129. package/dist/rollout/index.d.ts +1 -1
  130. package/dist/rollout/index.js +2 -2
  131. package/dist/{rollout-DmoJVqrF.js → rollout-B-UF5R6w.js} +2 -2
  132. package/dist/{rollout-DmoJVqrF.js.map → rollout-B-UF5R6w.js.map} +1 -1
  133. package/dist/rubric-predictive-validity-Bmj2_cll.d.ts +79 -0
  134. package/dist/rubric-predictive-validity-Bmj2_cll.d.ts.map +1 -0
  135. package/dist/rubric-predictive-validity-CCK-1B7w.js +178 -0
  136. package/dist/rubric-predictive-validity-CCK-1B7w.js.map +1 -0
  137. package/dist/{run-record-DTv1MdjK.d.ts → run-record-BiTWauyO.d.ts} +2 -2
  138. package/dist/{run-record-DTv1MdjK.d.ts.map → run-record-BiTWauyO.d.ts.map} +1 -1
  139. package/dist/run-record-Br-Yzt_k.js +464 -0
  140. package/dist/run-record-Br-Yzt_k.js.map +1 -0
  141. package/dist/{run-record-DQpSf7t-.js → run-record-DualPTn2.js} +2 -2
  142. package/dist/{run-record-DQpSf7t-.js.map → run-record-DualPTn2.js.map} +1 -1
  143. package/dist/{semantic-concept-judge-Bi6_iGqg.js → semantic-concept-judge-Bm5JDEKO.js} +3 -3
  144. package/dist/{semantic-concept-judge-Bi6_iGqg.js.map → semantic-concept-judge-Bm5JDEKO.js.map} +1 -1
  145. package/dist/{sequential-B5gXgcyp.js → sequential-DAsyV2T9.js} +42 -25
  146. package/dist/sequential-DAsyV2T9.js.map +1 -0
  147. package/dist/{series-convergence-DeG33RpC.d.ts → series-convergence-BnMs_uAr.d.ts} +3 -3
  148. package/dist/{series-convergence-DeG33RpC.d.ts.map → series-convergence-BnMs_uAr.d.ts.map} +1 -1
  149. package/dist/{skillopt-optimization-method-C3oYul8v.js → skillopt-optimization-method-CL_0aArC.js} +5 -5
  150. package/dist/{skillopt-optimization-method-C3oYul8v.js.map → skillopt-optimization-method-CL_0aArC.js.map} +1 -1
  151. package/dist/{statistical-heldout-0La5ZTlv.d.ts → statistical-heldout-CpVd6FmY.d.ts} +207 -144
  152. package/dist/statistical-heldout-CpVd6FmY.d.ts.map +1 -0
  153. package/dist/{store-tool-spans-4J1EDElP.d.ts → store-tool-spans-Dt-YdAuE.d.ts} +6 -6
  154. package/dist/{store-tool-spans-4J1EDElP.d.ts.map → store-tool-spans-Dt-YdAuE.d.ts.map} +1 -1
  155. package/dist/{summary-report-gMrbYawB.d.ts → summary-report-D1h4dlrK.d.ts} +3 -3
  156. package/dist/{summary-report-gMrbYawB.d.ts.map → summary-report-D1h4dlrK.d.ts.map} +1 -1
  157. package/dist/{summary-report-B16xy9Kd.js → summary-report-e-MaOAHV.js} +2 -2
  158. package/dist/{summary-report-B16xy9Kd.js.map → summary-report-e-MaOAHV.js.map} +1 -1
  159. package/dist/{tool-groups-2QA0S7dK.d.ts → tool-groups-B2bSNaJB.d.ts} +3 -3
  160. package/dist/tool-groups-B2bSNaJB.d.ts.map +1 -0
  161. package/dist/{tool-waste-B9tdWV6g.js → tool-waste-C7MU9u1e.js} +2 -2
  162. package/dist/{tool-waste-B9tdWV6g.js.map → tool-waste-C7MU9u1e.js.map} +1 -1
  163. package/dist/trace-repair/index.d.ts +2 -2
  164. package/dist/traces.d.ts +6 -6
  165. package/dist/traces.js +1 -1
  166. package/dist/{types-BmlkCrg0.d.ts → types-BvZoPTGa.d.ts} +3 -3
  167. package/dist/{types-BmlkCrg0.d.ts.map → types-BvZoPTGa.d.ts.map} +1 -1
  168. package/dist/{types-gvRsyJLh.d.ts → types-CBbLtr2J.d.ts} +38 -3
  169. package/dist/{types-gvRsyJLh.d.ts.map → types-CBbLtr2J.d.ts.map} +1 -1
  170. package/dist/{types-C34V4Vto.d.ts → types-CS0qk_Yp.d.ts} +4 -4
  171. package/dist/{types-C34V4Vto.d.ts.map → types-CS0qk_Yp.d.ts.map} +1 -1
  172. package/dist/{types-DzuaM493.d.ts → types-D7gEdPoQ.d.ts} +3 -3
  173. package/dist/{types-DzuaM493.d.ts.map → types-D7gEdPoQ.d.ts.map} +1 -1
  174. package/dist/wire/index.d.ts +2 -2
  175. package/docs/adapters-observability.md +14 -0
  176. package/docs/campaign-proposers.md +86 -128
  177. package/docs/charter.md +108 -112
  178. package/docs/concepts.md +157 -69
  179. package/docs/design/mlbenchmarks-book-review.md +440 -0
  180. package/docs/design/mlbenchmarks-review/observations.json +713 -0
  181. package/docs/design/mlbenchmarks-review/probes.mts +476 -0
  182. package/docs/design/mlbenchmarks-review/sources.json +200 -0
  183. package/docs/design/self-improvement-evidence-audit.md +263 -0
  184. package/docs/design.md +2 -1
  185. package/docs/eval-surface-map.md +95 -42
  186. package/docs/evaluation-integrity.md +220 -0
  187. package/docs/experiment.md +111 -55
  188. package/docs/feature-guide.md +5 -6
  189. package/docs/hosted-ingest-spec.md +4 -11
  190. package/docs/insight-report.md +187 -455
  191. package/docs/outcome-validity.md +182 -0
  192. package/docs/product-eval-adoption.md +1 -2
  193. package/docs/research-report-methodology.md +7 -7
  194. package/docs/search-history-receipts.md +8 -0
  195. package/docs/statistical-evidence.md +129 -0
  196. package/docs/verdicts.md +76 -49
  197. package/package.json +1 -1
  198. package/dist/agent-profile-cell-0gSi5ffD.js.map +0 -1
  199. package/dist/agent-profile-cell-CTOZJUuE.d.ts.map +0 -1
  200. package/dist/campaign-evidence-D8DBLqLI.js.map +0 -1
  201. package/dist/client-BlLY6o2w.js.map +0 -1
  202. package/dist/define-agent-eval-Cx4Ls9ta.d.ts.map +0 -1
  203. package/dist/define-agent-eval-Dzidv34q.js.map +0 -1
  204. package/dist/engine-CX8ReXkn.d.ts.map +0 -1
  205. package/dist/index-BxWvILU8.d.ts.map +0 -1
  206. package/dist/judge-calibration-C5CbMYce.d.ts.map +0 -1
  207. package/dist/ledger-core-Cs9f7385.js.map +0 -1
  208. package/dist/llm-judge-v80Kmu9g.js.map +0 -1
  209. package/dist/outcome-store-BYHIuO0e.d.ts.map +0 -1
  210. package/dist/outcome-store-ChBKlTd_.js +0 -75
  211. package/dist/outcome-store-ChBKlTd_.js.map +0 -1
  212. package/dist/promotion-policy-DWOm70gx.js.map +0 -1
  213. package/dist/rubric-predictive-validity-2D5Gw9z9.js +0 -131
  214. package/dist/rubric-predictive-validity-2D5Gw9z9.js.map +0 -1
  215. package/dist/rubric-predictive-validity-Dl1dvKCv.d.ts +0 -75
  216. package/dist/rubric-predictive-validity-Dl1dvKCv.d.ts.map +0 -1
  217. package/dist/run-record-CR63CpHK.js +0 -216
  218. package/dist/run-record-CR63CpHK.js.map +0 -1
  219. package/dist/sequential-B5gXgcyp.js.map +0 -1
  220. package/dist/statistical-heldout-0La5ZTlv.d.ts.map +0 -1
  221. package/dist/tool-groups-2QA0S7dK.d.ts.map +0 -1
@@ -1,11 +1,264 @@
1
- import { t as AgentEvalError } from "./errors-Dngq5h35.js";
1
+ import { s as ValidationError, t as AgentEvalError } from "./errors-Dngq5h35.js";
2
2
  import { a as hashCanonical, i as compareCodeUnits, r as canonicalString } from "./canonical-DPyQ_rpt.js";
3
- import { g as pairedRiskDifferenceScore, h as pairedRiskDifferenceExact, p as pairedBinaryScale } from "./paired-arms-D4aeIHUy.js";
4
- import { a as pairedSignTest, i as pairedDeltaTieFraction, r as pairedBootstrap } from "./paired-tests-C8iCsioC.js";
3
+ import { r as pairedBootstrap } from "./paired-tests-C8iCsioC.js";
4
+ import { n as campaignCellExecutionEvidence, o as projectCampaignCellQuality, s as decidePairedPromotion } from "./run-record-Br-Yzt_k.js";
5
5
  import { n as contentHash } from "./verdict-cache-B3eCVQtY.js";
6
- import { n as campaignCellExecutionEvidence, o as projectCampaignCellQuality } from "./run-record-CR63CpHK.js";
6
+ import { t as FileLedgerJournal } from "./journal-Cs9f7385.js";
7
+ import { g as readField } from "./ast-CP9ae9B0.js";
7
8
  import { createHash } from "node:crypto";
9
+ import { z } from "zod";
8
10
  import { join } from "node:path";
11
+ //#region src/campaign/gates/statistical-heldout.ts
12
+ /**
13
+ * Held-out inference pairs execution cells and judge identities before taking
14
+ * means within registered independent units. Repetitions can improve a unit's
15
+ * precision without increasing n. Ungrouped inference concerns independently
16
+ * sampled execution cells conditional on a fixed scenario roster.
17
+ *
18
+ * The shared paired decision rule selects the estimator and statistical test.
19
+ * Dimension reports retain missing coverage so required safety checks cannot
20
+ * pass through absent evidence. Thresholds use the judge's native score scale.
21
+ */
22
+ /** Tie fraction at/above which a gate annotates its verdict with the tie share.
23
+ * Tie-domination of the median bites structurally at >= 0.5 (the median is then
24
+ * 0 by construction); 0.4 is a softer warn threshold that flags a run APPROACHING
25
+ * that regime, so an operator sees it before the median goes fully blind. */
26
+ const TIE_WARN_FRACTION = .4;
27
+ /** Campaign cell IDs append a numeric repetition after the scenario's full ID. */
28
+ function scenarioIdFromCellId(cellId) {
29
+ const separator = cellId.lastIndexOf(":");
30
+ const repetition = cellId.slice(separator + 1);
31
+ if (separator < 1 || cellId.trim() !== cellId || !/^(0|[1-9]\d*)$/.test(repetition) || !Number.isSafeInteger(Number(repetition))) throw new Error(`pairHoldout: malformed cellId '${cellId}'; expected scenarioId:rep`);
32
+ return cellId.slice(0, separator);
33
+ }
34
+ /** Preserve cell pairing before taking equal-weight independent-unit means. */
35
+ function aggregatePairedHoldout(paired, independentUnitByScenarioId) {
36
+ if (paired.before.length !== paired.after.length || paired.before.length !== paired.cellIds.length) throw new Error("aggregatePairedHoldout: scores and cellIds must have the same length");
37
+ if (new Set(paired.cellIds).size !== paired.cellIds.length) throw new Error("aggregatePairedHoldout: duplicate cellIds cannot count as new observations");
38
+ if (paired.before.some((value) => !Number.isFinite(value)) || paired.after.some((value) => !Number.isFinite(value))) throw new Error("aggregatePairedHoldout: paired scores must be finite");
39
+ if (independentUnitByScenarioId === void 0) return {
40
+ before: [...paired.before],
41
+ after: [...paired.after],
42
+ unitIds: [...paired.cellIds]
43
+ };
44
+ const scenarioIds = paired.cellIds.map(scenarioIdFromCellId);
45
+ const groups = /* @__PURE__ */ new Map();
46
+ for (let i = 0; i < paired.cellIds.length; i++) {
47
+ const scenarioId = scenarioIds[i];
48
+ const unitId = independentUnitByScenarioId.get(scenarioId);
49
+ if (typeof unitId !== "string" || unitId.length === 0 || unitId.trim() !== unitId) throw new Error(`aggregatePairedHoldout: missing independent unit for scenario '${scenarioId}'`);
50
+ const group = groups.get(unitId) ?? {
51
+ before: 0,
52
+ after: 0,
53
+ n: 0
54
+ };
55
+ group.before += paired.before[i];
56
+ group.after += paired.after[i];
57
+ group.n += 1;
58
+ groups.set(unitId, group);
59
+ }
60
+ const unitIds = [...groups.keys()].sort();
61
+ return {
62
+ before: unitIds.map((id) => groups.get(id).before / groups.get(id).n),
63
+ after: unitIds.map((id) => groups.get(id).after / groups.get(id).n),
64
+ unitIds
65
+ };
66
+ }
67
+ /**
68
+ * Pair candidate vs baseline holdout observations by FULL cellId. `select`
69
+ * pulls the scalar from a cell's judge reports (composite, or a named
70
+ * dimension); a cell contributes the mean of `select` across its judges. Cells
71
+ * whose scenario is not in `scenarioIds`, or where `select` is undefined for
72
+ * every judge on both sides, are skipped. The selected judge IDs must agree
73
+ * within each pair. Throws when the two maps disagree on holdout cell IDs — a
74
+ * load-bearing invariant: the baseline + winner holdout campaigns run the same
75
+ * scenarios with the same seed base, so their cellIds MUST align; a mismatch
76
+ * means a silent pairing bug, not a soft fallback.
77
+ */
78
+ function pairHoldout(candidate, baseline, scenarioIds, select) {
79
+ const cellValues = (byCell, cellId) => {
80
+ const scores = byCell.get(cellId);
81
+ const values = /* @__PURE__ */ new Map();
82
+ if (!scores) return values;
83
+ for (const [judgeId, s] of Object.entries(scores)) {
84
+ if (s.failed === true) throw new Error(`pairHoldout: cell '${cellId}' contains a failed judge score`);
85
+ const v = select(s);
86
+ if (typeof v === "number" && !Number.isFinite(v)) throw new Error(`pairHoldout: cell '${cellId}' contains a non-finite selected score`);
87
+ if (typeof v === "number") values.set(judgeId, v);
88
+ }
89
+ return values;
90
+ };
91
+ const inScope = (cellId) => scenarioIds.has(scenarioIdFromCellId(cellId));
92
+ const candCells = [...candidate.keys()].filter(inScope).sort();
93
+ const baseCells = [...baseline.keys()].filter(inScope).sort();
94
+ if (candCells.length !== baseCells.length || candCells.some((c, i) => c !== baseCells[i])) throw new Error(`pairHoldout: candidate/baseline holdout cells do not align — candidate=[${candCells.join(",")}] baseline=[${baseCells.join(",")}]. Both holdout campaigns must run the same scenarios with the same seed base.`);
95
+ const before = [];
96
+ const after = [];
97
+ const cellIds = [];
98
+ for (const cellId of candCells) {
99
+ const b = cellValues(baseline, cellId);
100
+ const a = cellValues(candidate, cellId);
101
+ if (b.size === 0 && a.size === 0) continue;
102
+ if (b.size === 0 || a.size === 0) throw new Error(`pairHoldout: cell '${cellId}' has a selected score on only one arm`);
103
+ if (b.size !== a.size || [...b.keys()].some((id) => !a.has(id))) throw new Error(`pairHoldout: cell '${cellId}' selected judge IDs do not align`);
104
+ const judgeIds = [...b.keys()].sort();
105
+ before.push(meanSelectedScores(judgeIds.map((id) => b.get(id))));
106
+ after.push(meanSelectedScores(judgeIds.map((id) => a.get(id))));
107
+ cellIds.push(cellId);
108
+ }
109
+ return {
110
+ before,
111
+ after,
112
+ cellIds
113
+ };
114
+ }
115
+ function meanSelectedScores(values) {
116
+ const first = values[0];
117
+ return values.every((value) => value === first) ? first : values.reduce((sum, value) => sum + value, 0) / values.length;
118
+ }
119
+ /**
120
+ * Significance of the held-out composite lift: ship only when the lower bound
121
+ * of the interval the outcome's shape ADMITS exceeds `deltaThreshold` (default
122
+ * 0 ⇒ "confidently positive"). Interpret `deltaThreshold` in the judge's native
123
+ * scale.
124
+ *
125
+ * The decision is delegated whole to {@link decidePairedPromotion}, the one
126
+ * copy of the rule (`src/paired-promotion-decision.ts`), which `HeldOutGate`
127
+ * also calls. That module's header carries the measurements; the short version
128
+ * is three guards a bare `bootstrap.low > threshold` does not have:
129
+ *
130
+ * - a two-point (pass/fail) outcome decides on Tango's SCORE interval, the
131
+ * only paired-binary construction that stays valid at a nonzero margin;
132
+ * - McNemar's exact test VETOES at any non-negative threshold;
133
+ * - a ZERO-WIDTH interval is refused rather than promoted, in either
134
+ * direction — [0,0] clears every negative threshold and [g,g] clears every
135
+ * threshold below g, and both are an absence of evidence, not a result.
136
+ *
137
+ * Measured on this function before those guards landed, at a nominal 5 %:
138
+ * 14.60 % false promotion at n = 40 on a paired-binary noninferiority boundary,
139
+ * and 88.50 % at n = 6 under a bounded asymmetric null whose true mean paired
140
+ * delta is exactly 0.
141
+ *
142
+ * Continuous mean targets require bootstrap eligibility. Explicit median
143
+ * targets can use the exact sign test at its confidence-dependent minimum.
144
+ */
145
+ function heldoutSignificance(paired, opts = {}) {
146
+ const deltaThreshold = opts.deltaThreshold ?? 0;
147
+ const confidence = opts.confidence ?? .95;
148
+ const resamples = opts.resamples ?? 2e3;
149
+ const seed = opts.seed ?? 1337;
150
+ const statistic = opts.statistic ?? "mean";
151
+ const observations = aggregatePairedHoldout(paired, opts.independentUnitByScenarioId);
152
+ const decision = decidePairedPromotion(observations.before, observations.after, {
153
+ confidence,
154
+ resamples,
155
+ statistic,
156
+ seed,
157
+ threshold: deltaThreshold,
158
+ minPairs: opts.minProductiveRuns
159
+ });
160
+ const bootstrap = decision.bootstrap ?? pairedBootstrap(observations.before, observations.after, {
161
+ confidence,
162
+ resamples,
163
+ statistic,
164
+ seed
165
+ });
166
+ const medianBootstrap = statistic === "median" ? bootstrap : pairedBootstrap(observations.before, observations.after, {
167
+ confidence,
168
+ resamples,
169
+ statistic: "median",
170
+ seed
171
+ });
172
+ const n = observations.before.length;
173
+ let ties = 0;
174
+ for (let i = 0; i < n; i += 1) {
175
+ const after = observations.after[i];
176
+ const before = observations.before[i];
177
+ if (Math.abs(after - before) < 1e-9) ties += 1;
178
+ }
179
+ const tieFraction = n === 0 ? 0 : ties / n;
180
+ return {
181
+ paired,
182
+ bootstrap,
183
+ medianBootstrap,
184
+ decision,
185
+ decisionStatistic: decision.statistic,
186
+ mcnemar: decision.mcnemar,
187
+ tieFraction,
188
+ n,
189
+ pairedCellN: paired.cellIds.length,
190
+ observationUnit: opts.independentUnitByScenarioId === void 0 ? "cell" : "registered",
191
+ unitIds: observations.unitIds,
192
+ minimumRequired: decision.minimumPairs,
193
+ decisionMethod: decision.method,
194
+ pValue: decision.pValue,
195
+ significant: decision.promote,
196
+ fewRuns: !decision.sufficient
197
+ };
198
+ }
199
+ /** Detect the native scale of a set of scores: 0-100 when any magnitude clears
200
+ * 1.5, else [0,1]. Used to auto-scale the regression tolerance so a default
201
+ * expressed for [0,1] is not silently a no-op on a 0-100 dimension. */
202
+ function detectScale(values) {
203
+ return values.some((v) => Math.abs(v) > 1.5) ? 100 : 1;
204
+ }
205
+ /**
206
+ * Report required-dimension evidence after full pairing and optional unit means.
207
+ * A bootstrap floor breach or a shared paired test supporting a drop marks
208
+ * regression. These two criteria are distinct; `ci` records the shared
209
+ * estimator and `bootstrap` records the floor interval. Missing observations
210
+ * and insufficient n remain explicit for the caller's evidence policy.
211
+ * The default tolerance is 0.05 on [0,1] and 5 on a detected 0-100 scale.
212
+ */
213
+ function dimensionRegressions(candidate, baseline, scenarioIds, criticalDimensions, opts = {}) {
214
+ const out = [];
215
+ const expectedCellIds = [...baseline.keys()].filter((cellId) => scenarioIds.has(scenarioIdFromCellId(cellId))).sort();
216
+ for (const dim of criticalDimensions) {
217
+ const paired = pairHoldout(candidate, baseline, scenarioIds, (s) => s.dimensions[dim]);
218
+ if (paired.before.length === 0) continue;
219
+ const observations = aggregatePairedHoldout(paired, opts.independentUnitByScenarioId);
220
+ const measuredCells = new Set(paired.cellIds);
221
+ const measuredScenarios = new Set(paired.cellIds.map(scenarioIdFromCellId));
222
+ const tolerance = opts.tolerance ?? .05 * detectScale([...paired.before, ...paired.after]);
223
+ const bootstrapStatistic = opts.statistic ?? "mean";
224
+ const shared = {
225
+ confidence: opts.confidence ?? .95,
226
+ resamples: opts.resamples ?? 2e3,
227
+ statistic: bootstrapStatistic,
228
+ seed: opts.seed ?? 1337,
229
+ minPairs: opts.minProductiveRuns
230
+ };
231
+ const guard = decidePairedPromotion(observations.before, observations.after, shared);
232
+ const regression = decidePairedPromotion(observations.after, observations.before, {
233
+ ...shared,
234
+ threshold: tolerance
235
+ });
236
+ const bootstrap = guard.bootstrap ?? pairedBootstrap(observations.before, observations.after, shared);
237
+ out.push({
238
+ dimension: dim,
239
+ bootstrap,
240
+ bootstrapStatistic,
241
+ ci: {
242
+ low: guard.low,
243
+ high: guard.high
244
+ },
245
+ decisionStatistic: guard.statistic,
246
+ mcnemar: guard.mcnemar,
247
+ indeterminate: guard.indeterminate,
248
+ regressed: bootstrap.low < -tolerance || regression.promote,
249
+ tolerance,
250
+ n: observations.before.length,
251
+ pairedCellN: paired.cellIds.length,
252
+ observationUnit: opts.independentUnitByScenarioId === void 0 ? "cell" : "registered",
253
+ minimumRequired: guard.minimumPairs,
254
+ fewRuns: !guard.sufficient,
255
+ missingCellIds: expectedCellIds.filter((cellId) => !measuredCells.has(cellId)),
256
+ missingScenarioIds: [...scenarioIds].filter((id) => !measuredScenarios.has(id)).sort()
257
+ });
258
+ }
259
+ return out;
260
+ }
261
+ //#endregion
9
262
  //#region src/campaign/coverage.ts
10
263
  /** Reject campaign designs whose denominator cannot be identified exactly. */
11
264
  function assertCampaignDesign(scenarios, reps) {
@@ -401,457 +654,227 @@ function surfaceDispatchRef(surface, executionRef = "anonymous") {
401
654
  return `surface:${executionRef}:${surfaceContentHash(surface)}`;
402
655
  }
403
656
  //#endregion
404
- //#region src/paired-delta-test.ts
405
- /** Smallest all-positive sample that can clear a one-sided exact sign test. */
406
- function minimumPairsForPairedDeltaTest(confidence = .95) {
407
- if (!Number.isFinite(confidence) || confidence <= 0 || confidence >= 1) throw new Error(`minimumPairsForPairedDeltaTest: confidence must be in (0,1), got ${confidence}`);
408
- const oneSidedAlpha = (1 - confidence) / 2;
409
- return Math.ceil(Math.log2(1 / oneSidedAlpha));
657
+ //#region src/experiment/claim.ts
658
+ const nonEmpty = z.string().min(1).refine((value) => value.trim() === value);
659
+ const claimSchema = z.object({
660
+ use: z.enum([
661
+ "development",
662
+ "comparison",
663
+ "certification"
664
+ ]),
665
+ population: z.object({
666
+ id: nonEmpty,
667
+ description: nonEmpty
668
+ }).strict(),
669
+ samplingFrame: nonEmpty,
670
+ independentUnit: nonEmpty,
671
+ generalization: z.enum(["fixed-roster", "new-units"]),
672
+ minimumEffect: z.number().finite().positive().optional()
673
+ }).strict();
674
+ /** Validate a claim before binding it to a sealed experiment or final evidence. */
675
+ function defineEvaluationClaim(input) {
676
+ const parsed = claimSchema.safeParse(input);
677
+ if (!parsed.success) throw new ValidationError(`invalid evaluation claim: ${parsed.error.message}`);
678
+ return Object.freeze({
679
+ ...parsed.data,
680
+ population: Object.freeze(parsed.data.population)
681
+ });
410
682
  }
411
- /**
412
- * Tests whether a paired candidate-minus-baseline delta clears a threshold.
413
- *
414
- * At 20 or more pairs, the percentile bootstrap lower bound carries the
415
- * decision. Below that point the interval is descriptive only, so the function
416
- * switches to a pre-registered one-sided exact sign test. The exact path is
417
- * deliberately conservative: it requires both a point estimate above the
418
- * threshold and enough consistently positive paired differences.
419
- *
420
- * ## A zero-width interval is never significant
421
- *
422
- * When every paired delta is identical the resample distribution is a point
423
- * mass and the interval collapses: `[0, 0]` when all pairs tie, `[g, g]` on n
424
- * identical deltas of g. Neither says the effect is certain — both say the
425
- * sample carries no information about how far the estimate could be wrong, and
426
- * `low > threshold` then answers on the point estimate alone. It fails in both
427
- * directions: `[0, 0]` clears every NEGATIVE threshold, which is how a
428
- * tie-dominated pass/fail comparison laundered a regression into a
429
- * noninferiority pass, and `[g, g]` clears every threshold below g with no
430
- * spread behind it. Under a bounded asymmetric null whose true mean paired
431
- * delta is exactly 0 — 2 % of pairs dropping by 1.0, the rest gaining 0.0204 —
432
- * every sample that misses the drop is exactly that shape, and deciding on
433
- * `low > 0` promoted 65.65 % of samples at n = 20 against a nominal 5 %.
434
- *
435
- * So `indeterminate` is reported and `significant` is false whenever the
436
- * interval has zero width, on BOTH paths: at small n the exact sign test is a
437
- * test of the MEDIAN and a zero-spread sample is precisely where it stops
438
- * saying anything about the mean the caller is thresholding.
439
- *
440
- * `threshold` may be negative — that is a noninferiority margin, and it is the
441
- * regime the zero-width hole is worst in. For a two-point (pass/fail) outcome
442
- * the percentile bootstrap is not a valid interval at a nonzero margin at all;
443
- * use {@link decidePairedPromotion}, which routes those to Tango's score
444
- * interval, rather than thresholding this function's bootstrap directly.
445
- */
446
- function pairedDeltaTest(before, after, options = {}) {
447
- const threshold = options.threshold ?? 0;
448
- if (!Number.isFinite(threshold)) throw new Error(`pairedDeltaTest: threshold must be finite, got ${threshold}`);
449
- const exactMinimum = minimumPairsForPairedDeltaTest(options.confidence ?? .95);
450
- const requestedMinimum = options.minPairs ?? exactMinimum;
451
- if (!Number.isInteger(requestedMinimum) || requestedMinimum < 1) throw new Error(`pairedDeltaTest: minPairs must be a positive integer, got ${requestedMinimum}`);
452
- const minimumPairs = Math.max(requestedMinimum, exactMinimum);
453
- const bootstrap = pairedBootstrap(before, after, options);
454
- const sufficient = bootstrap.n >= minimumPairs;
455
- const indeterminate = bootstrap.n > 0 && (!Number.isFinite(bootstrap.low) || !Number.isFinite(bootstrap.high) || bootstrap.low === bootstrap.high);
456
- if (bootstrap.gateEligible) return {
457
- bootstrap,
458
- method: "bootstrap-ci",
459
- pValue: null,
460
- minimumPairs,
461
- sufficient,
462
- indeterminate,
463
- significant: sufficient && !indeterminate && bootstrap.low > threshold
464
- };
465
- const exact = pairedSignTest(before.map((value, index) => after[index] - value - threshold), "greater");
466
- const estimate = options.statistic === "mean" ? bootstrap.mean : bootstrap.median;
683
+ /** Repetitions retain their denominator without becoming additional independent units. */
684
+ function summarizeEvaluationUnits(claim, rows) {
685
+ const validated = defineEvaluationClaim(claim);
686
+ const counts = /* @__PURE__ */ new Map();
687
+ for (const [index, row] of rows.entries()) {
688
+ if (row === null || typeof row !== "object" || Array.isArray(row)) throw new ValidationError(`evaluation claim: row ${index} must be an object`);
689
+ const value = readField(row, validated.independentUnit);
690
+ if (typeof value !== "string" || !value.trim() || value.trim() !== value) throw new ValidationError(`evaluation claim: row ${index} needs a nonempty string at '${validated.independentUnit}'`);
691
+ counts.set(value, (counts.get(value) ?? 0) + 1);
692
+ }
467
693
  return {
468
- bootstrap,
469
- method: "exact-sign",
470
- pValue: exact.pValue,
471
- minimumPairs,
472
- sufficient,
473
- indeterminate,
474
- significant: sufficient && !indeterminate && estimate > threshold && exact.pValue <= (1 - bootstrap.confidence) / 2
694
+ observations: rows.length,
695
+ independentUnits: counts.size,
696
+ units: [...counts].sort(([left], [right]) => compareCodeUnits(left, right)).map(([id, observations]) => ({
697
+ id,
698
+ observations
699
+ }))
475
700
  };
476
701
  }
477
702
  //#endregion
478
- //#region src/paired-promotion-decision.ts
479
- /**
480
- * @module
481
- * ONE rule for "does this paired interval clear a promotion threshold".
482
- *
483
- * The rule below was derived on `HeldOutGate` (#479) after the same estimator
484
- * bug shipped twice. It then turned out that a SECOND gate — the composable
485
- * `heldOutGate`, plus everything else routed through `heldoutSignificance` —
486
- * still carried the original defect, because the rule had been written into one
487
- * gate's method body rather than into a shared function. Two copies of a
488
- * statistical rule is how a defect survives in one of them, so there is now
489
- * exactly one copy and both gates call it.
490
- *
491
- * Three things the rule does that a bare `pairedBootstrap(...).low > threshold`
492
- * does not:
493
- *
494
- * 1. **Two-point (pass/fail) outcomes decide on Tango's SCORE interval.** On a
495
- * pass/fail eval the paired delta vector is dominated by ties, so the
496
- * bootstrap of the mean is a resample of a lattice with three atoms and its
497
- * percentile interval is not valid at a nonzero margin. The score interval
498
- * (`pairedRiskDifferenceScore`) re-estimates the nuisance loss rate under
499
- * each hypothesised margin instead of fixing it at the observed value, which
500
- * is the only construction that stays a confidence interval as the margin
501
- * moves off zero — the regime every noninferiority threshold lives in.
502
- * Measured on the composable gate before this change, at a true risk
503
- * difference sitting exactly on the production caller's -0.05 margin and a
504
- * nominal 5 %: 14.60 % false promotion at n = 40 and 10.10 % at n = 76.
505
- * 2. **McNemar's exact test holds a VETO at every non-negative threshold.**
506
- * Redundant with the interval by construction and kept anyway, so that
507
- * swapping the estimator for one without that duality cannot silently
508
- * reintroduce "promotes what the exact test refuses". Witness: n = 6, b = 5,
509
- * c = 0 — no exact argument reaches alpha = 0.05 with 5 discordant pairs
510
- * (two-sided floor 2/2^5 = 0.0625), whatever an interval says. A NEGATIVE
511
- * threshold is a noninferiority question, which McNemar's test of "no
512
- * difference" is not the right test for, so the veto does not apply there.
513
- * 3. **A ZERO-WIDTH interval is refused, wherever it sits.** At [0, 0] it
514
- * cannot tell a gain from a regression and clears every negative threshold.
515
- * Away from zero it fails the opposite way: n identical positive deltas give
516
- * [g, g], which clears threshold 0 on no spread at all. Both are an absence
517
- * of evidence. Measured on the composable gate before this change, under a
518
- * bounded asymmetric null whose true mean paired delta is exactly 0: 88.50 %
519
- * false promotion at n = 6 and 65.65 % at n = 20 against a nominal 5 %.
520
- *
521
- * Orthogonal to the small-sample switch inside {@link pairedDeltaTest}: that
522
- * picks the TEST from the sample size (bootstrap CI at n >= 20, pre-registered
523
- * exact sign test below it), this picks the ESTIMATOR from the outcome's shape.
524
- * Both are needed — an exact sign test applied to a tie-pinned median is still
525
- * blind, and a mean bootstrap CI at n = 6 is still not a valid test.
526
- */
527
- /**
528
- * Which estimator {@link decidePairedPromotion} would use on this data, and the
529
- * shape facts behind it — for callers that must report the shape on a path
530
- * where no interval is computed at all (an early rejection, or zero pairs).
531
- * Cheap: no bootstrap, no interval.
532
- */
533
- function pairedDecisionShape(before, after, statistic = "mean") {
534
- const tieFraction = before.length === 0 ? null : pairedDeltaTieFraction(before, after);
535
- if (statistic === "median") return {
536
- statistic: "median_bootstrap",
537
- binaryScale: null,
538
- tieFraction
539
- };
540
- const binaryScale = pairedBinaryScale(before, after);
541
- if (binaryScale !== null) return {
542
- statistic: "paired_risk_difference",
543
- binaryScale,
544
- tieFraction
545
- };
546
- return {
547
- statistic: "mean_bootstrap",
548
- binaryScale: null,
549
- tieFraction
550
- };
551
- }
552
- /**
553
- * Decide whether a paired candidate-minus-baseline delta clears a promotion
554
- * threshold. `before` is the baseline arm, `after` the candidate arm, paired by
555
- * position. Throws on unequal lengths.
556
- */
557
- function decidePairedPromotion(before, after, options = {}) {
558
- if (before.length !== after.length) throw new Error(`decidePairedPromotion: unequal sample sizes (${before.length} vs ${after.length})`);
559
- const threshold = options.threshold ?? 0;
560
- if (!Number.isFinite(threshold)) throw new Error(`decidePairedPromotion: threshold must be finite, got ${threshold}`);
561
- const confidence = options.confidence ?? .95;
562
- const exactMinimum = minimumPairsForPairedDeltaTest(confidence);
563
- const requestedMinimum = options.minPairs ?? exactMinimum;
564
- if (!Number.isInteger(requestedMinimum) || requestedMinimum < 1) throw new Error(`decidePairedPromotion: minPairs must be a positive integer, got ${requestedMinimum}`);
565
- const minimumPairs = Math.max(requestedMinimum, exactMinimum);
566
- const n = before.length;
567
- const sufficient = n >= minimumPairs;
568
- const { binaryScale, tieFraction } = pairedDecisionShape(before, after, options.statistic);
569
- let core;
570
- if (binaryScale !== null) {
571
- const unitControl = before.map((v) => v / binaryScale);
572
- const unitTreatment = after.map((v) => v / binaryScale);
573
- const exact = pairedRiskDifferenceExact(unitControl, unitTreatment, confidence);
574
- const score = pairedRiskDifferenceScore(unitControl, unitTreatment, confidence);
575
- const low = score.lower * binaryScale;
576
- core = {
577
- statistic: "paired_risk_difference",
578
- method: "score-interval",
579
- delta: score.riskDifference * binaryScale,
580
- low,
581
- high: score.upper * binaryScale,
582
- bootstrap: null,
583
- mcnemar: {
584
- b: exact.b,
585
- c: exact.c,
586
- nDiscordant: exact.nDiscordant,
587
- pValue: exact.pValue
588
- },
589
- pValue: null,
590
- clearsThreshold: low > threshold,
591
- label: "success-rate",
592
- methodDetail: ""
593
- };
594
- } else {
595
- const bootstrapStatistic = options.statistic === "median" ? "median" : "mean";
596
- const test = pairedDeltaTest(before, after, {
597
- confidence,
598
- resamples: options.resamples,
599
- statistic: bootstrapStatistic,
600
- seed: options.seed,
601
- threshold,
602
- minPairs: options.minPairs
603
- });
604
- const ci = test.bootstrap;
605
- core = {
606
- statistic: bootstrapStatistic === "mean" ? "mean_bootstrap" : "median_bootstrap",
607
- method: test.method,
608
- delta: bootstrapStatistic === "mean" ? ci.mean : ci.median,
609
- low: ci.low,
610
- high: ci.high,
611
- bootstrap: ci,
612
- mcnemar: null,
613
- pValue: test.pValue,
614
- clearsThreshold: test.significant,
615
- label: bootstrapStatistic,
616
- methodDetail: test.method === "exact-sign" ? ` Below ${test.minimumPairs} pairs the interval is descriptive only; the decision is the exact one-sided sign test, p=${fmt(test.pValue ?? 1)}.` : ""
617
- };
703
+ //#region src/experiment/final-evidence.ts
704
+ const identity = z.string().min(1).refine((value) => value.trim() === value);
705
+ const digest = z.string().regex(/^sha256:[a-f0-9]{64}$/);
706
+ const reservationSchema = z.object({
707
+ requestId: identity,
708
+ claimDigest: digest,
709
+ populationId: identity,
710
+ inputDigest: digest,
711
+ unitIds: z.array(identity).min(1)
712
+ }).strict();
713
+ const measurementSchema = z.object({
714
+ evaluatorDigest: digest,
715
+ candidateDigests: z.array(digest).min(1)
716
+ }).strict();
717
+ const eventSchema = z.discriminatedUnion("kind", [z.object({
718
+ kind: z.literal("reserved"),
719
+ eventId: identity,
720
+ reservation: reservationSchema
721
+ }).strict(), z.object({
722
+ kind: z.literal("exposed"),
723
+ eventId: identity,
724
+ requestId: identity,
725
+ measurement: measurementSchema
726
+ }).strict()]);
727
+ const schema = "agent-eval.final-evidence.v1";
728
+ const entrySchema = z.object({
729
+ schema: z.literal(schema),
730
+ sequence: z.number().int().nonnegative(),
731
+ previousHash: digest.nullable(),
732
+ event: eventSchema,
733
+ entryHash: digest
734
+ }).strict();
735
+ /** Retains the distinction between invalid input, consumed evidence, and unavailable storage. */
736
+ var FinalEvidenceError = class extends Error {
737
+ kind;
738
+ constructor(kind, message) {
739
+ super(message);
740
+ this.kind = kind;
741
+ this.name = "FinalEvidenceError";
618
742
  }
619
- const indeterminate = !Number.isFinite(core.low) || !Number.isFinite(core.high) || core.low === core.high;
620
- const indeterminateCause = !indeterminate ? "" : tieFraction === 1 ? "every paired delta is an exact tie" : core.mcnemar !== null && core.mcnemar.nDiscordant === 0 ? "every pair is concordant (0 discordant pairs)" : `the ${core.label} CI collapsed to a point at ${fmt(core.low)}`;
621
- const exactTestVetoes = core.mcnemar !== null && threshold >= 0 && !(core.mcnemar.pValue < 1 - confidence);
743
+ };
744
+ /** A consumed or conflicting dataset cannot authorize another adaptive decision. */
745
+ var FinalEvidenceConflictError = class extends FinalEvidenceError {
746
+ constructor(message) {
747
+ super("conflict", message);
748
+ this.name = "FinalEvidenceConflictError";
749
+ }
750
+ };
751
+ function invalid(message) {
752
+ return /* @__PURE__ */ new TypeError(`final evidence: ${message}`);
753
+ }
754
+ function normalizeReservation(input) {
755
+ const parsed = reservationSchema.parse(input);
756
+ if (new Set(parsed.unitIds).size !== parsed.unitIds.length) throw invalid("unitIds must be unique independent source identities");
622
757
  return {
623
- n,
624
- threshold,
625
- confidence,
626
- binaryScale,
627
- tieFraction,
628
- minimumPairs,
629
- sufficient,
630
- indeterminate,
631
- indeterminateCause,
632
- exactTestVetoes,
633
- promote: sufficient && !indeterminate && core.clearsThreshold && !exactTestVetoes,
634
- ...core
758
+ ...parsed,
759
+ unitIds: [...parsed.unitIds].sort(compareCodeUnits)
635
760
  };
636
761
  }
637
- function fmt(x) {
638
- return x.toFixed(4);
639
- }
640
- //#endregion
641
- //#region src/campaign/gates/statistical-heldout.ts
642
- /**
643
- * Statistical held-out promotion machinery — the trustworthy core the
644
- * point-estimate `heldout-delta` gate lacked.
645
- *
646
- * The shipped false positive it prevents: a winner re-scored against the
647
- * baseline on the holdout read run-to-run model NOISE (e.g. 91 vs 95) as a
648
- * "+4 lift" and shipped, because the gate compared point estimates with no
649
- * confidence interval. Here we pair candidate vs baseline holdout observations
650
- * and bootstrap a CI on the paired delta — a candidate ships only when the CI
651
- * lower bound clears the effect-size threshold (the gain is real at the
652
- * confidence level, not noise), and is blocked when a critical dimension
653
- * (e.g. `hallucination_free` for a legal agent) significantly regresses even if
654
- * the net composite rose (anti-Goodhart).
655
- *
656
- * Two traps this module is built around (both produce a NEW false positive if
657
- * gotten wrong):
658
- * 1. PAIRING GRANULARITY — pairs by FULL `cellId` (`scenario:rep`), never by
659
- * `scenarioId` (which averages reps away and destroys the within-pair
660
- * variance reduction that makes a paired bootstrap tighter than unpaired).
661
- * One paired observation per cell ⇒ reps multiply n.
662
- * 2. SCALE — a judge may emit composites/dimensions on [0,1] or 0-100. The
663
- * threshold + tolerance are interpreted in the judge's NATIVE scale; the
664
- * per-dimension tolerance auto-scales off the observed baseline magnitudes
665
- * so `-0.10` on [0,1] doesn't silently become a no-op on a 0-100 dimension.
666
- */
667
- /** Tie fraction at/above which a gate annotates its verdict with the tie share.
668
- * Tie-domination of the median bites structurally at >= 0.5 (the median is then
669
- * 0 by construction); 0.4 is a softer warn threshold that flags a run APPROACHING
670
- * that regime, so an operator sees it before the median goes fully blind. */
671
- const TIE_WARN_FRACTION = .4;
672
- /**
673
- * Pair candidate vs baseline holdout observations by FULL cellId. `select`
674
- * pulls the scalar from a cell's judge reports (composite, or a named
675
- * dimension); a cell contributes the mean of `select` across its judges. Cells
676
- * whose scenario is not in `scenarioIds`, or where `select` is undefined for
677
- * every judge on either side, are skipped on BOTH sides so the arrays stay
678
- * paired. Throws when the two maps disagree on which holdout cells exist — a
679
- * load-bearing invariant: the baseline + winner holdout campaigns run the same
680
- * scenarios with the same seed base, so their cellIds MUST align; a mismatch
681
- * means a silent pairing bug, not a soft fallback.
682
- */
683
- function pairHoldout(candidate, baseline, scenarioIds, select) {
684
- const cellValue = (byCell, cellId) => {
685
- const scores = byCell.get(cellId);
686
- if (!scores) return void 0;
687
- const vals = [];
688
- for (const s of Object.values(scores)) {
689
- if (s.failed === true) throw new Error(`pairHoldout: cell '${cellId}' contains a failed judge score`);
690
- const v = select(s);
691
- if (typeof v === "number" && !Number.isFinite(v)) throw new Error(`pairHoldout: cell '${cellId}' contains a non-finite selected score`);
692
- if (typeof v === "number") vals.push(v);
693
- }
694
- if (vals.length === 0) return void 0;
695
- return vals.reduce((a, b) => a + b, 0) / vals.length;
696
- };
697
- const inScope = (cellId) => scenarioIds.has(cellId.split(":")[0] ?? "");
698
- const candCells = [...candidate.keys()].filter(inScope).sort();
699
- const baseCells = [...baseline.keys()].filter(inScope).sort();
700
- if (candCells.length !== baseCells.length || candCells.some((c, i) => c !== baseCells[i])) throw new Error(`pairHoldout: candidate/baseline holdout cells do not align — candidate=[${candCells.join(",")}] baseline=[${baseCells.join(",")}]. Both holdout campaigns must run the same scenarios with the same seed base.`);
701
- const before = [];
702
- const after = [];
703
- const cellIds = [];
704
- for (const cellId of candCells) {
705
- const b = cellValue(baseline, cellId);
706
- const a = cellValue(candidate, cellId);
707
- if (b === void 0 && a === void 0) continue;
708
- if (b === void 0 || a === void 0) throw new Error(`pairHoldout: cell '${cellId}' has a selected score on only one arm`);
709
- before.push(b);
710
- after.push(a);
711
- cellIds.push(cellId);
712
- }
762
+ function normalizeMeasurement(input) {
763
+ const parsed = measurementSchema.parse(input);
713
764
  return {
714
- before,
715
- after,
716
- cellIds
765
+ ...parsed,
766
+ candidateDigests: [...new Set(parsed.candidateDigests)].sort(compareCodeUnits)
717
767
  };
718
768
  }
719
- /**
720
- * Significance of the held-out composite lift: ship only when the lower bound
721
- * of the interval the outcome's shape ADMITS exceeds `deltaThreshold` (default
722
- * 0 ⇒ "confidently positive"). Interpret `deltaThreshold` in the judge's native
723
- * scale.
724
- *
725
- * The decision is delegated whole to {@link decidePairedPromotion}, the one
726
- * copy of the rule (`src/paired-promotion-decision.ts`), which `HeldOutGate`
727
- * also calls. That module's header carries the measurements; the short version
728
- * is three guards a bare `bootstrap.low > threshold` does not have:
729
- *
730
- * - a two-point (pass/fail) outcome decides on Tango's SCORE interval, the
731
- * only paired-binary construction that stays valid at a nonzero margin;
732
- * - McNemar's exact test VETOES at any non-negative threshold;
733
- * - a ZERO-WIDTH interval is refused rather than promoted, in either
734
- * direction — [0,0] clears every negative threshold and [g,g] clears every
735
- * threshold below g, and both are an absence of evidence, not a result.
736
- *
737
- * Measured on this function before those guards landed, at a nominal 5 %:
738
- * 14.60 % false promotion at n = 40 on a paired-binary noninferiority boundary,
739
- * and 88.50 % at n = 6 under a bounded asymmetric null whose true mean paired
740
- * delta is exactly 0.
741
- *
742
- * At small n, where the percentile bootstrap is descriptive only, a
743
- * pre-registered exact sign test still carries the bootstrap path.
744
- */
745
- function heldoutSignificance(paired, opts = {}) {
746
- const deltaThreshold = opts.deltaThreshold ?? 0;
747
- const confidence = opts.confidence ?? .95;
748
- const resamples = opts.resamples ?? 2e3;
749
- const seed = opts.seed ?? 1337;
750
- const statistic = opts.statistic ?? "mean";
751
- const decision = decidePairedPromotion(paired.before, paired.after, {
752
- confidence,
753
- resamples,
754
- statistic,
755
- seed,
756
- threshold: deltaThreshold,
757
- minPairs: opts.minProductiveRuns
758
- });
759
- const bootstrap = decision.bootstrap ?? pairedBootstrap(paired.before, paired.after, {
760
- confidence,
761
- resamples,
762
- statistic,
763
- seed
764
- });
765
- const medianBootstrap = statistic === "median" ? bootstrap : pairedBootstrap(paired.before, paired.after, {
766
- confidence,
767
- resamples,
768
- statistic: "median",
769
- seed
770
- });
771
- const n = paired.before.length;
772
- let ties = 0;
773
- for (let i = 0; i < n; i += 1) {
774
- const after = paired.after[i] ?? 0;
775
- const before = paired.before[i] ?? 0;
776
- if (Math.abs(after - before) < 1e-9) ties += 1;
777
- }
778
- const tieFraction = n === 0 ? 0 : ties / n;
769
+ function codec() {
779
770
  return {
780
- paired,
781
- bootstrap,
782
- medianBootstrap,
783
- decision,
784
- decisionStatistic: decision.statistic,
785
- mcnemar: decision.mcnemar,
786
- tieFraction,
787
- n,
788
- minimumRequired: decision.minimumPairs,
789
- decisionMethod: decision.method,
790
- pValue: decision.pValue,
791
- significant: decision.promote,
792
- fewRuns: !decision.sufficient
771
+ subject: "final evidence ledger",
772
+ header: { schema },
773
+ integrityError: (message, options) => new Error(message, options),
774
+ conflictError: (message) => new FinalEvidenceConflictError(message),
775
+ parseEntry: (raw, context) => {
776
+ const decoded = entrySchema.safeParse(raw);
777
+ if (!decoded.success) throw new Error(`final evidence ledger ${context.path}:${context.line} is invalid: ${decoded.error.message}`);
778
+ const parsed = decoded.data;
779
+ return {
780
+ ...parsed,
781
+ previousHash: parsed.previousHash,
782
+ entryHash: parsed.entryHash
783
+ };
784
+ },
785
+ checkEntryHeader: (entry) => {
786
+ if (entry.schema !== schema) throw invalid("unsupported ledger schema");
787
+ },
788
+ createProjector: () => {
789
+ const records = /* @__PURE__ */ new Map();
790
+ const owners = /* @__PURE__ */ new Map();
791
+ const inputOwners = /* @__PURE__ */ new Map();
792
+ return {
793
+ apply: (entry) => {
794
+ const event = entry.event;
795
+ if (event.kind === "reserved") {
796
+ const reservation = normalizeReservation(event.reservation);
797
+ if (event.eventId !== `reserve:${reservation.requestId}` || records.has(reservation.requestId)) throw invalid("invalid or duplicate reservation identity");
798
+ const inputOwner = inputOwners.get(reservation.inputDigest);
799
+ if (inputOwner !== void 0) throw new FinalEvidenceConflictError(`final input is already reserved by '${inputOwner}'`);
800
+ for (const unitId of reservation.unitIds) {
801
+ const owner = owners.get(unitId);
802
+ if (owner !== void 0) throw new FinalEvidenceConflictError(`final unit '${unitId}' is already reserved by '${owner}'`);
803
+ owners.set(unitId, reservation.requestId);
804
+ }
805
+ inputOwners.set(reservation.inputDigest, reservation.requestId);
806
+ records.set(reservation.requestId, {
807
+ reservation,
808
+ reservationHash: entry.entryHash,
809
+ exposure: null
810
+ });
811
+ } else {
812
+ const record = records.get(event.requestId);
813
+ if (event.eventId !== `expose:${event.requestId}` || !record || record.exposure !== null) throw invalid("exposure requires one unexposed reservation");
814
+ record.exposure = {
815
+ measurement: normalizeMeasurement(event.measurement),
816
+ entryHash: entry.entryHash
817
+ };
818
+ }
819
+ },
820
+ finish: () => [...records.values()]
821
+ };
822
+ }
793
823
  };
794
824
  }
795
- /** Detect the native scale of a set of scores: 0-100 when any magnitude clears
796
- * 1.5, else [0,1]. Used to auto-scale the regression tolerance so a default
797
- * expressed for [0,1] is not silently a no-op on a 0-100 dimension. */
798
- function detectScale(values) {
799
- return values.some((v) => Math.abs(v) > 1.5) ? 100 : 1;
800
- }
801
- /** Per-critical-dimension regression guard. For each dimension, pair the
802
- * candidate vs baseline values by full cellId and bootstrap the paired delta;
803
- * a dimension is "regressed" when the CI lower bound < −tolerance (conservative
804
- * — blocks if the credible worst case exceeds tolerance, which is the right
805
- * posture for safety dimensions like `hallucination_free`). When `tolerance`
806
- * is omitted it auto-scales: 0.05 on [0,1], 5 on 0-100.
807
- *
808
- * The interval comes from {@link decidePairedPromotion}, so a pass/fail
809
- * dimension is judged on Tango's score interval rather than a percentile
810
- * bootstrap of the mean — `tolerance` is a NONZERO margin, and the bootstrap
811
- * is not a valid interval at one. That matters most here because this guard
812
- * fails OPEN by construction: `tolerance` is positive, so an interval pinned at
813
- * [0,0] never satisfies `low < −tolerance` and a real regression on a safety
814
- * dimension would be reported as `regressed: false`. On the median it fails the
815
- * same way for the same reason — when most pairs tie, which is automatic for a
816
- * pass/fail dimension on {0,1} and on the 0-100 encoding `detectScale` exists
817
- * to support, the median CI collapses to [0,0]. Pass `statistic: 'median'` to
818
- * restore the pre-0.134 behaviour. */
819
- function dimensionRegressions(candidate, baseline, scenarioIds, criticalDimensions, opts = {}) {
820
- const out = [];
821
- for (const dim of criticalDimensions) {
822
- const paired = pairHoldout(candidate, baseline, scenarioIds, (s) => s.dimensions[dim]);
823
- if (paired.before.length === 0) continue;
824
- const tolerance = opts.tolerance ?? .05 * detectScale([...paired.before, ...paired.after]);
825
- const bootstrapStatistic = opts.statistic ?? "mean";
826
- const shared = {
827
- confidence: opts.confidence ?? .95,
828
- resamples: opts.resamples ?? 2e3,
829
- statistic: bootstrapStatistic,
830
- seed: opts.seed ?? 1337
825
+ async function outcome(operation) {
826
+ try {
827
+ return {
828
+ succeeded: true,
829
+ value: await operation()
830
+ };
831
+ } catch (error) {
832
+ return {
833
+ succeeded: false,
834
+ error: {
835
+ kind: error instanceof FinalEvidenceConflictError ? "conflict" : error instanceof TypeError || error instanceof z.ZodError ? "invalid" : "unavailable",
836
+ message: error instanceof Error ? error.message : String(error)
837
+ }
831
838
  };
832
- const guard = decidePairedPromotion(paired.before, paired.after, shared);
833
- const regression = decidePairedPromotion(paired.after, paired.before, {
834
- ...shared,
835
- threshold: tolerance
836
- });
837
- const bootstrap = guard.bootstrap ?? pairedBootstrap(paired.before, paired.after, shared);
838
- out.push({
839
- dimension: dim,
840
- bootstrap,
841
- bootstrapStatistic,
842
- ci: {
843
- low: guard.low,
844
- high: guard.high
845
- },
846
- decisionStatistic: guard.statistic,
847
- mcnemar: guard.mcnemar,
848
- indeterminate: guard.indeterminate,
849
- regressed: bootstrap.low < -tolerance || regression.promote,
850
- tolerance,
851
- n: paired.before.length
852
- });
853
839
  }
854
- return out;
840
+ }
841
+ /** Uses the shared locked journal and requires its trusted head on every reopen. */
842
+ function openFinalEvidenceLedger(options) {
843
+ if (!options.path.trim()) throw invalid("path is empty");
844
+ const journal = new FileLedgerJournal(options.path, codec(), { requireTrustedHead: true });
845
+ return {
846
+ reserve: (input) => outcome(async () => {
847
+ const reservation = normalizeReservation(input);
848
+ const result = await journal.append({
849
+ kind: "reserved",
850
+ eventId: `reserve:${reservation.requestId}`,
851
+ reservation
852
+ }, { pinHead: true });
853
+ const record = result.projection.find((item) => item.reservation.requestId === reservation.requestId);
854
+ if (!record) throw invalid("reserved record is missing after append");
855
+ return {
856
+ record,
857
+ replayed: !result.appended
858
+ };
859
+ }),
860
+ expose: (requestId, input) => outcome(async () => {
861
+ identity.parse(requestId);
862
+ const measurement = normalizeMeasurement(input);
863
+ const result = await journal.append({
864
+ kind: "exposed",
865
+ eventId: `expose:${requestId}`,
866
+ requestId,
867
+ measurement
868
+ }, { pinHead: true });
869
+ const record = result.projection.find((item) => item.reservation.requestId === requestId);
870
+ if (!record) throw invalid("exposed record is missing after append");
871
+ return {
872
+ record,
873
+ replayed: !result.appended
874
+ };
875
+ }),
876
+ read: () => outcome(async () => (await journal.replay()).projection)
877
+ };
855
878
  }
856
879
  //#endregion
857
880
  //#region src/campaign/gates/power-preflight.ts
@@ -862,10 +885,7 @@ function zFor(confidence) {
862
885
  if (confidence >= .9) return 1.645;
863
886
  return 1.282;
864
887
  }
865
- /** Estimate the minimum detectable lift a paired-holdout improvement run can
866
- * ship at a given budget, from the baseline holdout composites — call it BEFORE
867
- * spending a search to learn whether the effect you are hunting is even
868
- * observable at this holdout size and worker variance. */
888
+ /** Estimate detectable lift from baseline independent observations before budgeting a comparison. */
869
889
  function powerPreflight(opts) {
870
890
  const composites = opts.baselineComposites.filter((v) => Number.isFinite(v));
871
891
  if (composites.length < 3) throw new Error(`powerPreflight: need >= 3 finite baseline composites to estimate variance, got ${composites.length}`);
@@ -881,8 +901,8 @@ function powerPreflight(opts) {
881
901
  const scaleAssumed = composites.every((v) => v >= -.001 && v <= 1.5);
882
902
  const headroom = Math.max(0, 1 - mean);
883
903
  const underpowered = scaleAssumed && mde > headroom;
884
- const sharedChannelCaveat = opts.sharedScorerChannel ? "Holdout and gate share one scoring channel: raising n/reps reduces only idiosyncratic noise — systematic judge bias remains and this MDE is a lower bound. Full debiasing needs an independent second scoring channel (different judge/benchmark family)." : void 0;
885
- const recommendation = underpowered ? `UNDERPOWERED: minimum detectable lift ${mde.toFixed(3)} exceeds the ${headroom.toFixed(3)} headroom above the baseline (${mean.toFixed(3)}) — no achievable effect can ship at this budget. Raise paired n (scenarios x reps) to ~${Math.ceil((z * Math.SQRT2 * sd / Math.max(headroom - deltaThreshold, .01)) ** 2)} or reduce worker variance before searching.` : `Minimum detectable lift at n=${n}: ${mde.toFixed(3)} (baseline sd ${sd.toFixed(3)}). Effects smaller than this cannot clear the gate; budget the search for effects you believe exceed it.`;
904
+ const sharedChannelCaveat = opts.sharedScorerChannel ? "Holdout and gate share one scoring channel: systematic judge bias remains outside this estimate. An independent second scoring channel can help test that bias." : void 0;
905
+ const recommendation = underpowered ? `UNDERPOWERED under this approximation: detectable lift ${mde.toFixed(3)} exceeds the ${headroom.toFixed(3)} headroom above the baseline (${mean.toFixed(3)}). Raise paired n using independent observations to ~${Math.ceil((z * Math.SQRT2 * sd / Math.max(headroom - deltaThreshold, .01)) ** 2)} or reduce observation variance. Recheck with measured paired deltas.` : `Approximate detectable lift at n=${n}: ${mde.toFixed(3)} (baseline sd ${sd.toFixed(3)}). Compare this estimate with the effect you expect, then recheck using measured paired deltas.`;
886
906
  return {
887
907
  n,
888
908
  sd,
@@ -1202,6 +1222,35 @@ function parseReflectionResponse(raw, maxProposals) {
1202
1222
  * "composite of a campaign" and "per-scenario / per-dimension breakdown" so
1203
1223
  * the optimizers cannot drift on how a surface's score is computed.
1204
1224
  */
1225
+ /** Reduce complete paired cells using the same observation units as held-out inference. */
1226
+ function pairedCampaignComposites(baseline, candidate, independentUnitByScenarioId) {
1227
+ const scoresByCell = (campaign) => {
1228
+ const scores = /* @__PURE__ */ new Map();
1229
+ for (const cell of campaign.cells) {
1230
+ if (scores.has(cell.cellId)) throw new Error(`pairedCampaignComposites: duplicate cell '${cell.cellId}'`);
1231
+ const quality = projectCampaignCellQuality(cell);
1232
+ scores.set(cell.cellId, quality.score === void 0 ? {} : quality.successfulJudgeScores);
1233
+ }
1234
+ return scores;
1235
+ };
1236
+ const scenarioIds = new Set([...baseline.cells, ...candidate.cells].map((cell) => cell.scenarioId));
1237
+ const paired = pairHoldout(scoresByCell(candidate), scoresByCell(baseline), scenarioIds, (score) => score.composite);
1238
+ const observations = aggregatePairedHoldout(paired, independentUnitByScenarioId);
1239
+ const scoredCellIds = new Set(paired.cellIds);
1240
+ if (observations.before.length === 0) throw new Error("pairedCampaignComposites: campaigns have no paired quality scores");
1241
+ return {
1242
+ before: observations.before,
1243
+ after: observations.after,
1244
+ beforeMean: observations.before.reduce((sum, score) => sum + score, 0) / observations.before.length,
1245
+ afterMean: observations.after.reduce((sum, score) => sum + score, 0) / observations.after.length,
1246
+ observations: {
1247
+ pairedCellN: paired.cellIds.length,
1248
+ unitIds: observations.unitIds,
1249
+ unscoredCellIds: baseline.cells.filter((cell) => !scoredCellIds.has(cell.cellId)).map((cell) => cell.cellId).sort(),
1250
+ ...independentUnitByScenarioId ? { independentUnitByScenarioId: Object.fromEntries([...scenarioIds].sort().map((id) => [id, independentUnitByScenarioId.get(id)])) } : {}
1251
+ }
1252
+ };
1253
+ }
1205
1254
  /** Mean composite across cells with complete task-quality evidence.
1206
1255
  * Partial judge results remain on their cells but never enter this value.
1207
1256
  * A campaign with no complete score has no numeric mean and fails loudly. */
@@ -1344,6 +1393,8 @@ function loopProvenanceArgsFromResult(input) {
1344
1393
  ...result.holdout === "deferred" ? { holdout: "deferred" } : {},
1345
1394
  baselineOnHoldout: result.baselineOnHoldout,
1346
1395
  winnerOnHoldout: result.winnerOnHoldout,
1396
+ ...result.claim ? { claim: result.claim } : {},
1397
+ ...input.independentUnitByScenarioId ? { independentUnitByScenarioId: input.independentUnitByScenarioId } : {},
1347
1398
  ...result.neutralizedSurface && result.neutralizedOnHoldout ? {
1348
1399
  neutralizedSurface: result.neutralizedSurface,
1349
1400
  neutralizedOnHoldout: result.neutralizedOnHoldout
@@ -1353,9 +1404,6 @@ function loopProvenanceArgsFromResult(input) {
1353
1404
  totalDurationMs: input.totalDurationMs
1354
1405
  };
1355
1406
  }
1356
- function meanHoldoutComposite(campaign) {
1357
- return campaignMeanComposite(campaign);
1358
- }
1359
1407
  /** Build the durable provenance record from a completed loop result. */
1360
1408
  function buildLoopProvenanceRecord(args) {
1361
1409
  if (!args.runId.trim() || !args.runDir.trim()) throw new Error("buildLoopProvenanceRecord: runId and runDir must be non-empty");
@@ -1422,15 +1470,18 @@ function buildLoopProvenanceRecord(args) {
1422
1470
  }
1423
1471
  if (surfaceHash(args.winnerSurface) !== incumbentSurfaceHash) throw new Error("buildLoopProvenanceRecord: winner surface does not match the final promoted incumbent");
1424
1472
  const holdoutDeferred = args.holdout === "deferred";
1473
+ const claim = args.claim ? defineEvaluationClaim(args.claim) : void 0;
1474
+ if (claim && !holdoutDeferred && args.independentUnitByScenarioId === void 0) throw new Error("buildLoopProvenanceRecord: a measured claim requires its independent-unit map");
1425
1475
  if (args.baselineOnHoldout.splitDigest !== args.winnerOnHoldout.splitDigest) throw new Error("buildLoopProvenanceRecord: baseline and winner use different holdout splits");
1426
1476
  if (args.neutralizedSurface === void 0 !== (args.neutralizedOnHoldout === void 0)) throw new Error("buildLoopProvenanceRecord: neutralized surface and campaign must be supplied together");
1427
1477
  if (args.neutralizedOnHoldout && args.neutralizedOnHoldout.splitDigest !== args.baselineOnHoldout.splitDigest) throw new Error("buildLoopProvenanceRecord: neutralized campaign uses a different holdout split");
1428
1478
  if (holdoutDeferred && args.neutralizedOnHoldout) throw new Error("buildLoopProvenanceRecord: a deferred holdout cannot include a neutralized measurement");
1429
- const holdoutMeasurement = holdoutDeferred ? { kind: "deferred" } : {
1479
+ const holdoutScores = holdoutDeferred ? void 0 : pairedCampaignComposites(args.baselineOnHoldout, args.winnerOnHoldout, args.independentUnitByScenarioId);
1480
+ const holdoutMeasurement = holdoutScores === void 0 ? { kind: "deferred" } : {
1430
1481
  kind: "measured",
1431
- baseline: meanHoldoutComposite(args.baselineOnHoldout),
1432
- winner: meanHoldoutComposite(args.winnerOnHoldout),
1433
- ...args.neutralizedOnHoldout ? { neutralized: meanHoldoutComposite(args.neutralizedOnHoldout) } : {}
1482
+ baseline: holdoutScores.beforeMean,
1483
+ winner: holdoutScores.afterMean,
1484
+ ...args.neutralizedOnHoldout ? { neutralized: pairedCampaignComposites(args.baselineOnHoldout, args.neutralizedOnHoldout, args.independentUnitByScenarioId).afterMean } : {}
1434
1485
  };
1435
1486
  const diff = surfaceContentHash(args.baselineSurface) === surfaceContentHash(args.winnerSurface) ? "" : renderSurfaceDiff(args.winnerSurface, args.baselineSurface);
1436
1487
  const recordWithoutDigest = {
@@ -1451,6 +1502,7 @@ function buildLoopProvenanceRecord(args) {
1451
1502
  splitDigest: args.baselineOnHoldout.splitDigest,
1452
1503
  baselineCampaignDigest: campaignMeasurementDigest(args.baselineOnHoldout),
1453
1504
  winnerCampaignDigest: campaignMeasurementDigest(args.winnerOnHoldout),
1505
+ ...holdoutScores ? { observations: holdoutScores.observations } : {},
1454
1506
  ...args.neutralizedSurface && args.neutralizedOnHoldout && holdoutMeasurement.kind === "measured" && holdoutMeasurement.neutralized !== void 0 ? { neutralized: {
1455
1507
  contentHash: surfaceContentHash(args.neutralizedSurface),
1456
1508
  campaignDigest: campaignMeasurementDigest(args.neutralizedOnHoldout),
@@ -1461,6 +1513,7 @@ function buildLoopProvenanceRecord(args) {
1461
1513
  costReceiptsDigest: canonicalDigest([...args.costReceipts].sort((left, right) => compareCodeUnits(left.callId, right.callId)))
1462
1514
  },
1463
1515
  baselineSearchComposite,
1516
+ ...claim ? { claim } : {},
1464
1517
  gate: {
1465
1518
  decision: args.gate.decision,
1466
1519
  reasons: args.gate.reasons,
@@ -1890,15 +1943,18 @@ function attest(report, provenance) {
1890
1943
  * canonicalizes) is a verification failure with the cause in `reason`, not a
1891
1944
  * crash — verifiers run in pipelines that must record WHY, not die.
1892
1945
  *
1893
- * Legacy attestations without `envelopeHash` remain readable, but verification
1894
- * explicitly marks their provenance as unbound so a promotion path can refuse
1895
- * them instead of accidentally treating old metadata as cryptographic proof.
1946
+ * Both the report hash and the provenance envelope must verify.
1947
+ * Missing envelope hashes cannot establish provenance and fail verification.
1896
1948
  */
1897
1949
  function verifyAttestation(report, attested) {
1898
1950
  if (attested.algorithm !== "sha256/canonical-json") return {
1899
1951
  valid: false,
1900
1952
  reason: `unknown algorithm '${attested.algorithm}' — this verifier only checks '${ATTESTATION_ALGORITHM}'`
1901
1953
  };
1954
+ if (typeof attested.envelopeHash !== "string" || !/^[0-9a-f]{64}$/.test(attested.envelopeHash)) return {
1955
+ valid: false,
1956
+ reason: "attestation envelope hash is missing or invalid"
1957
+ };
1902
1958
  let recomputed;
1903
1959
  try {
1904
1960
  recomputed = contentHash(report);
@@ -1912,10 +1968,6 @@ function verifyAttestation(report, attested) {
1912
1968
  valid: false,
1913
1969
  reason: `report hash mismatch: attested ${attested.reportHash}, recomputed ${recomputed}`
1914
1970
  };
1915
- if (attested.envelopeHash === void 0) return {
1916
- valid: true,
1917
- legacyUnboundProvenance: true
1918
- };
1919
1971
  let envelopeHash;
1920
1972
  try {
1921
1973
  envelopeHash = contentHash(envelopeMaterial(attested.reportHash, attested.provenance, attested.algorithm));
@@ -1989,10 +2041,8 @@ function createEvidenceReceipt(input, provenance) {
1989
2041
  });
1990
2042
  }
1991
2043
  /**
1992
- * Verify promotion-grade evidence. Generic report attestation keeps a legacy read path,
1993
- * but an EvidenceReceipt never accepts unbound provenance: changing the evaluator code,
1994
- * model versions, input commitment provenance, or creation record must invalidate the
1995
- * evidence rather than merely annotating it as legacy.
2044
+ * Verify evidence identity and its provenance envelope. Changing the evaluator,
2045
+ * model versions, input commitment, or creation record invalidates the evidence.
1996
2046
  */
1997
2047
  function verifyEvidenceReceipt(receipt) {
1998
2048
  if (receipt.binding.schemaVersion !== "1.0.0") return {
@@ -2018,13 +2068,7 @@ function verifyEvidenceReceipt(receipt) {
2018
2068
  reason: error instanceof Error ? error.message : String(error)
2019
2069
  };
2020
2070
  }
2021
- const verification = verifyAttestation(receipt.binding, receipt.attestation);
2022
- if (!verification.valid) return verification;
2023
- if (verification.legacyUnboundProvenance === true || receipt.attestation.envelopeHash === void 0) return {
2024
- valid: false,
2025
- reason: "evidence receipt provenance is not bound by an attestation envelope"
2026
- };
2027
- return { valid: true };
2071
+ return verifyAttestation(receipt.binding, receipt.attestation);
2028
2072
  }
2029
2073
  /**
2030
2074
  * Promotion may choose a stricter policy, but this primitive makes the basic separation
@@ -2078,6 +2122,6 @@ function createCampaignEvidenceReceipt(input) {
2078
2122
  });
2079
2123
  }
2080
2124
  //#endregion
2081
- export { campaignScenarioIdentity as $, detectScale as A, componentSurfaceIdentityMaterial as B, campaignMeanCompositeOrNull as C, recoverTruncatedJson as D, parseReflectionResponse as E, pairedDecisionShape as F, surfaceHashMatches as G, surfaceContentHash as H, minimumPairsForPairedDeltaTest as I, summarizeBackendIntegrity as J, BackendIntegrityError as K, pairedDeltaTest as L, heldoutSignificance as M, pairHoldout as N, powerPreflight as O, decidePairedPromotion as P, campaignCoverage as Q, assertCodeSurfaceIdentity as R, campaignMeanComposite as S, buildReflectionPrompt as T, surfaceDispatchRef as U, renderSurfaceDiff as V, surfaceHash as W, assertCampaignSplitIdentity as X, assertCampaignDesign as Y, assertCompleteCampaign as Z, provenanceRecordPath as _, createEvidenceReceipt as a, assertFiniteRankKey as b, ATTESTATION_ALGORITHM as c, buildLoopProvenanceRecord as d, campaignSplitDigest as et, campaignMeasurementDigest as f, loopProvenanceSpans as g, loopProvenanceArgsFromResult as h, INDEPENDENT_EVIDENCE_AUTHORITY_KINDS as i, dimensionRegressions as j, TIE_WARN_FRACTION as k, attest as l, emitLoopProvenance as m, EVIDENCE_AUTHORITY_KINDS as n, formatCoverageFailures as nt, isIndependentEvidence as o, canonicalDigest as p, assertRealBackend as q, EVIDENCE_RECEIPT_VERSION as r, verifyEvidenceReceipt as s, createCampaignEvidenceReceipt as t, campaignSplitDigestFromIdentities as tt, verifyAttestation as u, provenanceSpansPath as v, compareRankKeys as w, campaignBreakdown as x, verifyLoopProvenanceRecord as y, codeSurfaceIdentityMaterial as z };
2125
+ export { formatCoverageFailures as $, FinalEvidenceConflictError as A, surfaceDispatchRef as B, campaignMeanCompositeOrNull as C, parseReflectionResponse as D, buildReflectionPrompt as E, assertCodeSurfaceIdentity as F, summarizeBackendIntegrity as G, surfaceHashMatches as H, codeSurfaceIdentityMaterial as I, assertCompleteCampaign as J, assertCampaignDesign as K, componentSurfaceIdentityMaterial as L, openFinalEvidenceLedger as M, defineEvaluationClaim as N, recoverTruncatedJson as O, summarizeEvaluationUnits as P, campaignSplitDigestFromIdentities as Q, renderSurfaceDiff as R, campaignMeanComposite as S, pairedCampaignComposites as T, BackendIntegrityError as U, surfaceHash as V, assertRealBackend as W, campaignScenarioIdentity as X, campaignCoverage as Y, campaignSplitDigest as Z, provenanceRecordPath as _, createEvidenceReceipt as a, pairHoldout as at, assertFiniteRankKey as b, ATTESTATION_ALGORITHM as c, buildLoopProvenanceRecord as d, TIE_WARN_FRACTION as et, campaignMeasurementDigest as f, loopProvenanceSpans as g, loopProvenanceArgsFromResult as h, INDEPENDENT_EVIDENCE_AUTHORITY_KINDS as i, heldoutSignificance as it, FinalEvidenceError as j, powerPreflight as k, attest as l, emitLoopProvenance as m, EVIDENCE_AUTHORITY_KINDS as n, detectScale as nt, isIndependentEvidence as o, canonicalDigest as p, assertCampaignSplitIdentity as q, EVIDENCE_RECEIPT_VERSION as r, dimensionRegressions as rt, verifyEvidenceReceipt as s, createCampaignEvidenceReceipt as t, aggregatePairedHoldout as tt, verifyAttestation as u, provenanceSpansPath as v, compareRankKeys as w, campaignBreakdown as x, verifyLoopProvenanceRecord as y, surfaceContentHash as z };
2082
2126
 
2083
- //# sourceMappingURL=campaign-evidence-D8DBLqLI.js.map
2127
+ //# sourceMappingURL=campaign-evidence-B8oF9xQ6.js.map