@tangle-network/agent-eval 0.180.0 → 0.181.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (221) hide show
  1. package/CHANGELOG.md +60 -0
  2. package/README.md +119 -159
  3. package/dist/adapters/http.d.ts +2 -2
  4. package/dist/{agent-profile-B7yErX0q.d.ts → agent-profile-CivaSsSy.d.ts} +4 -4
  5. package/dist/{agent-profile-B7yErX0q.d.ts.map → agent-profile-CivaSsSy.d.ts.map} +1 -1
  6. package/dist/{agent-profile-cell-0gSi5ffD.js → agent-profile-cell-Cv6UA-W_.js} +20 -57
  7. package/dist/agent-profile-cell-Cv6UA-W_.js.map +1 -0
  8. package/dist/{agent-profile-cell-CTOZJUuE.d.ts → agent-profile-cell-s__adRnK.d.ts} +3 -3
  9. package/dist/agent-profile-cell-s__adRnK.d.ts.map +1 -0
  10. package/dist/analyst/index.d.ts +10 -10
  11. package/dist/analyst/index.js +3 -3
  12. package/dist/ast-CP9ae9B0.js +557 -0
  13. package/dist/ast-CP9ae9B0.js.map +1 -0
  14. package/dist/ast-hI-vjW6J.d.ts +457 -0
  15. package/dist/ast-hI-vjW6J.d.ts.map +1 -0
  16. package/dist/{benchmark-command-D4vpnAdO.js → benchmark-command-B57n9vjz.js} +7 -6
  17. package/dist/{benchmark-command-D4vpnAdO.js.map → benchmark-command-B57n9vjz.js.map} +1 -1
  18. package/dist/benchmarks/index.d.ts +4 -4
  19. package/dist/benchmarks/index.js +3 -3
  20. package/dist/campaign/index.d.ts +6 -6
  21. package/dist/campaign/index.js +8 -8
  22. package/dist/{campaign-BGEurASO.js → campaign-4_ppJW5X.js} +12 -12
  23. package/dist/{campaign-BGEurASO.js.map → campaign-4_ppJW5X.js.map} +1 -1
  24. package/dist/{campaign-evidence-D8DBLqLI.js → campaign-evidence-B8oF9xQ6.js} +515 -471
  25. package/dist/campaign-evidence-B8oF9xQ6.js.map +1 -0
  26. package/dist/cli.js +4 -7
  27. package/dist/cli.js.map +1 -1
  28. package/dist/{client-BlLY6o2w.js → client-CXE-U1SA.js} +3 -1
  29. package/dist/client-CXE-U1SA.js.map +1 -0
  30. package/dist/{client-CuQgX33c.d.ts → client-kh2jOjTK.d.ts} +4 -4
  31. package/dist/{client-CuQgX33c.d.ts.map → client-kh2jOjTK.d.ts.map} +1 -1
  32. package/dist/contract/index.d.ts +13 -13
  33. package/dist/contract/index.js +10 -9
  34. package/dist/contract/index.js.map +1 -1
  35. package/dist/{default-registry-IGDE9XIC.d.ts → default-registry-BwDSWVzg.d.ts} +6 -6
  36. package/dist/{default-registry-IGDE9XIC.d.ts.map → default-registry-BwDSWVzg.d.ts.map} +1 -1
  37. package/dist/{define-agent-eval-Cx4Ls9ta.d.ts → define-agent-eval-CwOWWQt_.d.ts} +33 -12
  38. package/dist/define-agent-eval-CwOWWQt_.d.ts.map +1 -0
  39. package/dist/{define-agent-eval-Dzidv34q.js → define-agent-eval-Ddu33JH9.js} +134 -67
  40. package/dist/define-agent-eval-Ddu33JH9.js.map +1 -0
  41. package/dist/{dspy-rlm-engine-xKiWmj_G.js → dspy-rlm-engine-S53V0HhE.js} +2 -2
  42. package/dist/{dspy-rlm-engine-xKiWmj_G.js.map → dspy-rlm-engine-S53V0HhE.js.map} +1 -1
  43. package/dist/{engine-CX8ReXkn.d.ts → engine-DS1cysJy.d.ts} +10 -7
  44. package/dist/engine-DS1cysJy.d.ts.map +1 -0
  45. package/dist/{eval-campaign-Cs-7MiCs.js → eval-campaign-aYdtjtJR.js} +4 -4
  46. package/dist/{eval-campaign-Cs-7MiCs.js.map → eval-campaign-aYdtjtJR.js.map} +1 -1
  47. package/dist/{exact-types-B7LC1EyX.d.ts → exact-types-BZDe0W2D.d.ts} +2 -2
  48. package/dist/{exact-types-B7LC1EyX.d.ts.map → exact-types-BZDe0W2D.d.ts.map} +1 -1
  49. package/dist/experiment/index.d.ts +27 -477
  50. package/dist/experiment/index.d.ts.map +1 -1
  51. package/dist/experiment/index.js +95 -559
  52. package/dist/experiment/index.js.map +1 -1
  53. package/dist/{experiment-tracker-B3TiF5-u.d.ts → experiment-tracker-C7PfnF4b.d.ts} +2 -2
  54. package/dist/{experiment-tracker-B3TiF5-u.d.ts.map → experiment-tracker-C7PfnF4b.d.ts.map} +1 -1
  55. package/dist/{external-optimizer-process-Dlz8YxrT.js → external-optimizer-process-QDRURJAM.js} +3 -3
  56. package/dist/{external-optimizer-process-Dlz8YxrT.js.map → external-optimizer-process-QDRURJAM.js.map} +1 -1
  57. package/dist/{external-optimizer-subprocess-q3VzlGAO.js → external-optimizer-subprocess-D4dzUBZI.js} +3 -2
  58. package/dist/{external-optimizer-subprocess-q3VzlGAO.js.map → external-optimizer-subprocess-D4dzUBZI.js.map} +1 -1
  59. package/dist/{feedback-trajectory-eHWNv5Aj.d.ts → feedback-trajectory-CXmtITBo.d.ts} +3 -3
  60. package/dist/{feedback-trajectory-eHWNv5Aj.d.ts.map → feedback-trajectory-CXmtITBo.d.ts.map} +1 -1
  61. package/dist/hosted/index.d.ts +2 -2
  62. package/dist/hosted/index.d.ts.map +1 -1
  63. package/dist/hosted/index.js +1 -1
  64. package/dist/{index-BxWvILU8.d.ts → index-Bp_6sj3x.d.ts} +109 -56
  65. package/dist/index-Bp_6sj3x.d.ts.map +1 -0
  66. package/dist/{index-e7LXeRVa.d.ts → index-CJ3LhKIX.d.ts} +2 -2
  67. package/dist/{index-e7LXeRVa.d.ts.map → index-CJ3LhKIX.d.ts.map} +1 -1
  68. package/dist/{index-CiUjjEIa.d.ts → index-DNntP4ch.d.ts} +7 -7
  69. package/dist/{index-CiUjjEIa.d.ts.map → index-DNntP4ch.d.ts.map} +1 -1
  70. package/dist/{index-DxNYmx4a.d.ts → index-DoykkxW0.d.ts} +11 -11
  71. package/dist/{index-DxNYmx4a.d.ts.map → index-DoykkxW0.d.ts.map} +1 -1
  72. package/dist/index.d.ts +28 -28
  73. package/dist/index.js +24 -15
  74. package/dist/index.js.map +1 -1
  75. package/dist/{insight-report-DETqPc_A.d.ts → insight-report-D1qa0HWs.d.ts} +9 -5
  76. package/dist/{insight-report-DETqPc_A.d.ts.map → insight-report-D1qa0HWs.d.ts.map} +1 -1
  77. package/dist/{integrity-BKTcA-HP.d.ts → integrity-rGOfSUle.d.ts} +2 -2
  78. package/dist/{integrity-BKTcA-HP.d.ts.map → integrity-rGOfSUle.d.ts.map} +1 -1
  79. package/dist/{ledger-core-Cs9f7385.js → journal-Cs9f7385.js} +1 -1
  80. package/dist/journal-Cs9f7385.js.map +1 -0
  81. package/dist/{judge-calibration-C5CbMYce.d.ts → judge-calibration-DFtEMlde.d.ts} +31 -2
  82. package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
  83. package/dist/{judge-calibration-BnpVKtnb.js → judge-calibration-DYmaBtJr.js} +48 -2
  84. package/dist/{judge-calibration-BnpVKtnb.js.map → judge-calibration-DYmaBtJr.js.map} +1 -1
  85. package/dist/ledger-core/index.d.ts +1 -1
  86. package/dist/ledger-core/index.js +1 -1
  87. package/dist/{llm-judge-v80Kmu9g.js → llm-judge-DEFZeSiu.js} +645 -456
  88. package/dist/llm-judge-DEFZeSiu.js.map +1 -0
  89. package/dist/{matrix-DeMmnWrP.d.ts → matrix-CyhW-vgJ.d.ts} +2 -2
  90. package/dist/{matrix-DeMmnWrP.d.ts.map → matrix-CyhW-vgJ.d.ts.map} +1 -1
  91. package/dist/meta-eval/index.d.ts +138 -7
  92. package/dist/meta-eval/index.d.ts.map +1 -1
  93. package/dist/meta-eval/index.js +245 -97
  94. package/dist/meta-eval/index.js.map +1 -1
  95. package/dist/{mint-Cc1_zwRQ.js → mint-ySIIkKlV.js} +2 -2
  96. package/dist/{mint-Cc1_zwRQ.js.map → mint-ySIIkKlV.js.map} +1 -1
  97. package/dist/multishot/golden/index.d.ts +1 -1
  98. package/dist/multishot/index.d.ts +2 -2
  99. package/dist/openapi.json +1 -1
  100. package/dist/outcome-store-BXlkwMPR.js +131 -0
  101. package/dist/outcome-store-BXlkwMPR.js.map +1 -0
  102. package/dist/{outcome-store-BYHIuO0e.d.ts → outcome-store-CNt4iZ67.d.ts} +18 -25
  103. package/dist/outcome-store-CNt4iZ67.d.ts.map +1 -0
  104. package/dist/{paired-promotion-decision-CGzg0cI_.d.ts → paired-promotion-decision-DPsMQm-0.d.ts} +13 -7
  105. package/dist/{paired-promotion-decision-CGzg0cI_.d.ts.map → paired-promotion-decision-DPsMQm-0.d.ts.map} +1 -1
  106. package/dist/pipelines/index.js +1 -1
  107. package/dist/{produced-state-Cv0kJJuP.js → produced-state-BHboMaab.js} +3 -3
  108. package/dist/{produced-state-Cv0kJJuP.js.map → produced-state-BHboMaab.js.map} +1 -1
  109. package/dist/profile-cell.d.ts +1 -1
  110. package/dist/profile-cell.js +1 -1
  111. package/dist/{promotion-policy-DWOm70gx.js → promotion-policy-CDMMxzb6.js} +28 -40
  112. package/dist/promotion-policy-CDMMxzb6.js.map +1 -0
  113. package/dist/{registry-ByVld1-5.d.ts → registry-BRbB6Y0v.d.ts} +4 -4
  114. package/dist/{registry-ByVld1-5.d.ts.map → registry-BRbB6Y0v.d.ts.map} +1 -1
  115. package/dist/{release-confidence-BAcNYOf1.d.ts → release-confidence-BcqeQTHW.d.ts} +3 -3
  116. package/dist/{release-confidence-BAcNYOf1.d.ts.map → release-confidence-BcqeQTHW.d.ts.map} +1 -1
  117. package/dist/{release-confidence-BcGCclTB.js → release-confidence-DMg8n18l.js} +2 -2
  118. package/dist/{release-confidence-BcGCclTB.js.map → release-confidence-DMg8n18l.js.map} +1 -1
  119. package/dist/reporting.d.ts +4 -4
  120. package/dist/reporting.js +3 -3
  121. package/dist/{researcher-jsW1X94L.d.ts → researcher-64T49THL.d.ts} +6 -6
  122. package/dist/{researcher-jsW1X94L.d.ts.map → researcher-64T49THL.d.ts.map} +1 -1
  123. package/dist/{reward-hacking-ZXEi9VCq.d.ts → reward-hacking-uzO_ihep.d.ts} +2 -2
  124. package/dist/{reward-hacking-ZXEi9VCq.d.ts.map → reward-hacking-uzO_ihep.d.ts.map} +1 -1
  125. package/dist/rl.d.ts +53 -99
  126. package/dist/rl.d.ts.map +1 -1
  127. package/dist/rl.js +182 -169
  128. package/dist/rl.js.map +1 -1
  129. package/dist/rollout/index.d.ts +1 -1
  130. package/dist/rollout/index.js +2 -2
  131. package/dist/{rollout-DmoJVqrF.js → rollout-B-UF5R6w.js} +2 -2
  132. package/dist/{rollout-DmoJVqrF.js.map → rollout-B-UF5R6w.js.map} +1 -1
  133. package/dist/rubric-predictive-validity-Bmj2_cll.d.ts +79 -0
  134. package/dist/rubric-predictive-validity-Bmj2_cll.d.ts.map +1 -0
  135. package/dist/rubric-predictive-validity-CCK-1B7w.js +178 -0
  136. package/dist/rubric-predictive-validity-CCK-1B7w.js.map +1 -0
  137. package/dist/{run-record-DTv1MdjK.d.ts → run-record-BiTWauyO.d.ts} +2 -2
  138. package/dist/{run-record-DTv1MdjK.d.ts.map → run-record-BiTWauyO.d.ts.map} +1 -1
  139. package/dist/run-record-Br-Yzt_k.js +464 -0
  140. package/dist/run-record-Br-Yzt_k.js.map +1 -0
  141. package/dist/{run-record-DQpSf7t-.js → run-record-DualPTn2.js} +2 -2
  142. package/dist/{run-record-DQpSf7t-.js.map → run-record-DualPTn2.js.map} +1 -1
  143. package/dist/{semantic-concept-judge-Bi6_iGqg.js → semantic-concept-judge-Bm5JDEKO.js} +3 -3
  144. package/dist/{semantic-concept-judge-Bi6_iGqg.js.map → semantic-concept-judge-Bm5JDEKO.js.map} +1 -1
  145. package/dist/{sequential-B5gXgcyp.js → sequential-DAsyV2T9.js} +42 -25
  146. package/dist/sequential-DAsyV2T9.js.map +1 -0
  147. package/dist/{series-convergence-DeG33RpC.d.ts → series-convergence-BnMs_uAr.d.ts} +3 -3
  148. package/dist/{series-convergence-DeG33RpC.d.ts.map → series-convergence-BnMs_uAr.d.ts.map} +1 -1
  149. package/dist/{skillopt-optimization-method-C3oYul8v.js → skillopt-optimization-method-CL_0aArC.js} +5 -5
  150. package/dist/{skillopt-optimization-method-C3oYul8v.js.map → skillopt-optimization-method-CL_0aArC.js.map} +1 -1
  151. package/dist/{statistical-heldout-0La5ZTlv.d.ts → statistical-heldout-CpVd6FmY.d.ts} +207 -144
  152. package/dist/statistical-heldout-CpVd6FmY.d.ts.map +1 -0
  153. package/dist/{store-tool-spans-4J1EDElP.d.ts → store-tool-spans-Dt-YdAuE.d.ts} +6 -6
  154. package/dist/{store-tool-spans-4J1EDElP.d.ts.map → store-tool-spans-Dt-YdAuE.d.ts.map} +1 -1
  155. package/dist/{summary-report-gMrbYawB.d.ts → summary-report-D1h4dlrK.d.ts} +3 -3
  156. package/dist/{summary-report-gMrbYawB.d.ts.map → summary-report-D1h4dlrK.d.ts.map} +1 -1
  157. package/dist/{summary-report-B16xy9Kd.js → summary-report-e-MaOAHV.js} +2 -2
  158. package/dist/{summary-report-B16xy9Kd.js.map → summary-report-e-MaOAHV.js.map} +1 -1
  159. package/dist/{tool-groups-2QA0S7dK.d.ts → tool-groups-B2bSNaJB.d.ts} +3 -3
  160. package/dist/tool-groups-B2bSNaJB.d.ts.map +1 -0
  161. package/dist/{tool-waste-B9tdWV6g.js → tool-waste-C7MU9u1e.js} +2 -2
  162. package/dist/{tool-waste-B9tdWV6g.js.map → tool-waste-C7MU9u1e.js.map} +1 -1
  163. package/dist/trace-repair/index.d.ts +2 -2
  164. package/dist/traces.d.ts +6 -6
  165. package/dist/traces.js +1 -1
  166. package/dist/{types-BmlkCrg0.d.ts → types-BvZoPTGa.d.ts} +3 -3
  167. package/dist/{types-BmlkCrg0.d.ts.map → types-BvZoPTGa.d.ts.map} +1 -1
  168. package/dist/{types-gvRsyJLh.d.ts → types-CBbLtr2J.d.ts} +38 -3
  169. package/dist/{types-gvRsyJLh.d.ts.map → types-CBbLtr2J.d.ts.map} +1 -1
  170. package/dist/{types-C34V4Vto.d.ts → types-CS0qk_Yp.d.ts} +4 -4
  171. package/dist/{types-C34V4Vto.d.ts.map → types-CS0qk_Yp.d.ts.map} +1 -1
  172. package/dist/{types-DzuaM493.d.ts → types-D7gEdPoQ.d.ts} +3 -3
  173. package/dist/{types-DzuaM493.d.ts.map → types-D7gEdPoQ.d.ts.map} +1 -1
  174. package/dist/wire/index.d.ts +2 -2
  175. package/docs/adapters-observability.md +14 -0
  176. package/docs/campaign-proposers.md +86 -128
  177. package/docs/charter.md +108 -112
  178. package/docs/concepts.md +157 -69
  179. package/docs/design/mlbenchmarks-book-review.md +440 -0
  180. package/docs/design/mlbenchmarks-review/observations.json +713 -0
  181. package/docs/design/mlbenchmarks-review/probes.mts +476 -0
  182. package/docs/design/mlbenchmarks-review/sources.json +200 -0
  183. package/docs/design/self-improvement-evidence-audit.md +263 -0
  184. package/docs/design.md +2 -1
  185. package/docs/eval-surface-map.md +95 -42
  186. package/docs/evaluation-integrity.md +220 -0
  187. package/docs/experiment.md +111 -55
  188. package/docs/feature-guide.md +5 -6
  189. package/docs/hosted-ingest-spec.md +4 -11
  190. package/docs/insight-report.md +187 -455
  191. package/docs/outcome-validity.md +182 -0
  192. package/docs/product-eval-adoption.md +1 -2
  193. package/docs/research-report-methodology.md +7 -7
  194. package/docs/search-history-receipts.md +8 -0
  195. package/docs/statistical-evidence.md +129 -0
  196. package/docs/verdicts.md +76 -49
  197. package/package.json +1 -1
  198. package/dist/agent-profile-cell-0gSi5ffD.js.map +0 -1
  199. package/dist/agent-profile-cell-CTOZJUuE.d.ts.map +0 -1
  200. package/dist/campaign-evidence-D8DBLqLI.js.map +0 -1
  201. package/dist/client-BlLY6o2w.js.map +0 -1
  202. package/dist/define-agent-eval-Cx4Ls9ta.d.ts.map +0 -1
  203. package/dist/define-agent-eval-Dzidv34q.js.map +0 -1
  204. package/dist/engine-CX8ReXkn.d.ts.map +0 -1
  205. package/dist/index-BxWvILU8.d.ts.map +0 -1
  206. package/dist/judge-calibration-C5CbMYce.d.ts.map +0 -1
  207. package/dist/ledger-core-Cs9f7385.js.map +0 -1
  208. package/dist/llm-judge-v80Kmu9g.js.map +0 -1
  209. package/dist/outcome-store-BYHIuO0e.d.ts.map +0 -1
  210. package/dist/outcome-store-ChBKlTd_.js +0 -75
  211. package/dist/outcome-store-ChBKlTd_.js.map +0 -1
  212. package/dist/promotion-policy-DWOm70gx.js.map +0 -1
  213. package/dist/rubric-predictive-validity-2D5Gw9z9.js +0 -131
  214. package/dist/rubric-predictive-validity-2D5Gw9z9.js.map +0 -1
  215. package/dist/rubric-predictive-validity-Dl1dvKCv.d.ts +0 -75
  216. package/dist/rubric-predictive-validity-Dl1dvKCv.d.ts.map +0 -1
  217. package/dist/run-record-CR63CpHK.js +0 -216
  218. package/dist/run-record-CR63CpHK.js.map +0 -1
  219. package/dist/sequential-B5gXgcyp.js.map +0 -1
  220. package/dist/statistical-heldout-0La5ZTlv.d.ts.map +0 -1
  221. package/dist/tool-groups-2QA0S7dK.d.ts.map +0 -1
@@ -1,523 +1,18 @@
1
1
  import { n as CaptureIntegrityError, s as ValidationError } from "../errors-Dngq5h35.js";
2
- import { r as canonicalString } from "../canonical-DPyQ_rpt.js";
3
- import { _ as verifyManifest, g as signManifest, h as manifestContentDigest, m as hashJson, p as evaluateHypothesis } from "../agent-profile-cell-0gSi5ffD.js";
2
+ import { a as hashCanonical } from "../canonical-DPyQ_rpt.js";
3
+ import { _ as verifyManifest, g as signManifest, h as manifestContentDigest, m as hashJson, p as evaluateHypothesis } from "../agent-profile-cell-Cv6UA-W_.js";
4
4
  import { t as mulberry32 } from "../random-Dn5fPWkt.js";
5
5
  import { a as requiredSampleSize, c as holm, i as requiredPairedSampleSize, n as mcnemarRequiredN, o as benjaminiHochberg, r as pairedMde, s as bonferroni, t as mcnemarPower } from "../power-and-mde-B8F2RdcD.js";
6
6
  import { f as mcnemar, g as pairedRiskDifferenceScore, h as pairedRiskDifferenceExact, m as pairedRiskDifference, n as pairArms, r as pairRunRecords, t as comparePairedArms, v as wilson } from "../paired-arms-D4aeIHUy.js";
7
7
  import { r as pairedBootstrap, t as BOOTSTRAP_GATE_MIN_N } from "../paired-tests-C8iCsioC.js";
8
8
  import { t as eProcess } from "../sequential-eprocess-D1jKoihe.js";
9
- import { M as heldoutSignificance, N as pairHoldout, O as powerPreflight, a as createEvidenceReceipt, i as INDEPENDENT_EVIDENCE_AUTHORITY_KINDS, n as EVIDENCE_AUTHORITY_KINDS, o as isIndependentEvidence, r as EVIDENCE_RECEIPT_VERSION, s as verifyEvidenceReceipt, t as createCampaignEvidenceReceipt } from "../campaign-evidence-D8DBLqLI.js";
9
+ import { A as FinalEvidenceConflictError, M as openFinalEvidenceLedger, N as defineEvaluationClaim, P as summarizeEvaluationUnits, a as createEvidenceReceipt, at as pairHoldout, i as INDEPENDENT_EVIDENCE_AUTHORITY_KINDS, it as heldoutSignificance, j as FinalEvidenceError, k as powerPreflight, n as EVIDENCE_AUTHORITY_KINDS, o as isIndependentEvidence, r as EVIDENCE_RECEIPT_VERSION, s as verifyEvidenceReceipt, t as createCampaignEvidenceReceipt } from "../campaign-evidence-B8oF9xQ6.js";
10
+ import { _ as runSelectionRule, a as evaluateHaltRule, c as evaluatePopulationReproducibilityGate, d as evaluateProvenanceGate, f as executeDecisionRule, g as readField, h as projectNLadderBudget, i as evaluateCondition, l as evaluatePowerFloorGate, m as powerFloorProblems, n as computeEstimand, o as evaluateIdentityGate, p as intervalSpecProblems, r as computeInterval, s as evaluateOracleDeterminismGate, t as classifyReissue, u as evaluatePredicate, v as runUniformPassBudget } from "../ast-CP9ae9B0.js";
10
11
  import { a as inMemoryExperimentStore, r as fileExperimentStore, t as ExperimentTracker } from "../experiment-tracker-BKEumQug.js";
11
12
  import { n as pairedEvalueSequence, r as sequentialCrossingHorizon } from "../sequential-CzK5DarL.js";
12
- import { n as paretoPolicy, r as paretoSignificanceGate, t as buildEvidenceVector } from "../promotion-policy-DWOm70gx.js";
13
- import { n as sequentialPairedGate, t as sequentialDecide } from "../sequential-B5gXgcyp.js";
14
- import { createHash } from "node:crypto";
13
+ import { n as paretoPolicy, r as paretoSignificanceGate, t as buildEvidenceVector } from "../promotion-policy-CDMMxzb6.js";
14
+ import { n as sequentialPairedGate, t as sequentialDecide } from "../sequential-DAsyV2T9.js";
15
15
  import { z } from "zod";
16
- //#region src/experiment/ast.ts
17
- /**
18
- * The registered-rule AST: every rule an experiment registers is DATA.
19
- *
20
- * A closure cannot be canonicalized or hashed; a node tree can. `sealExperiment`
21
- * hashes the whole tree, and every interpreter in this file takes only a node
22
- * plus evidence records — no parameter for alpha, threshold, metric, or
23
- * stopping rule exists on any executable surface. The registered object and
24
- * the executed object are therefore the same object, and registered-vs-ran
25
- * drift is unrepresentable rather than checked.
26
- *
27
- * Node families:
28
- * Predicate closed-key comparisons — the only leaf
29
- * AdmissionRule monotone funnel stages with registered waivers
30
- * SelectionRule deterministic subsets over a closed field set
31
- * Estimand what the experiment measures
32
- * IntervalSpec how uncertainty is computed, seed included
33
- * Condition decision guards over named derived quantities
34
- * DecisionRule ordered verdict table, or a registered absence of one
35
- * Obligation a control that must exist before a verdict class is read
36
- * ValidityGate pre-spend design checks
37
- * HaltRule gates as prerequisites — failure refuses the spend
38
- * BudgetRule spend schedules with a named ledger
39
- * MatchedBudgetRule arm budget matching as a refusal
40
- * ReissuePolicy carrier faults are reissued; model outcomes stand
41
- */
42
- /** A decision rule's branches did not cover the evidence. */
43
- var DecisionTableNotTotalError = class extends ValidationError {};
44
- /** Read a dot-separated field path. Missing segments yield `undefined`. */
45
- function readField(record, path) {
46
- let current = record;
47
- for (const key of path.split(".")) {
48
- if (current === null || typeof current !== "object") return void 0;
49
- current = current[key];
50
- }
51
- return current;
52
- }
53
- /** Evaluate a predicate against one evidence record. */
54
- function evaluatePredicate(predicate, record) {
55
- switch (predicate.kind) {
56
- case "compare": return compareValues(readField(record, predicate.field), predicate.op, predicate.value);
57
- case "in": return predicate.values.includes(readField(record, predicate.field));
58
- case "all": return predicate.of.every((p) => evaluatePredicate(p, record));
59
- case "any": return predicate.of.some((p) => evaluatePredicate(p, record));
60
- case "not": return !evaluatePredicate(predicate.of, record);
61
- }
62
- }
63
- /** Numbers compare numerically; everything else compares as strings. */
64
- function compareValues(value, op, target) {
65
- if (op === "eq") return value === target;
66
- if (op === "ne") return value !== target;
67
- const numeric = typeof value === "number" && typeof target === "number";
68
- const left = numeric ? value : String(value);
69
- const right = numeric ? target : String(target);
70
- switch (op) {
71
- case "lt": return left < right;
72
- case "lte": return left <= right;
73
- case "gt": return left > right;
74
- case "gte": return left >= right;
75
- }
76
- }
77
- /**
78
- * Execute a selection rule.
79
- *
80
- * Round-robin walks groups in lexicographic order and takes ids in
81
- * within-group order until `take` ids are chosen. Filter-of keeps the ids of
82
- * `bases[rule.base]` whose record satisfies the predicate, in the registered
83
- * order. Ids absent from `records` are evaluated on their id alone (fields
84
- * derived from the id via `idFields`), so a sealed base outlives its source
85
- * records.
86
- */
87
- function runSelectionRule(rule, records, options) {
88
- if (rule.kind === "round-robin") {
89
- const allowed = new Set(rule.reads);
90
- for (const field of [rule.groupBy, rule.withinOrder.field]) if (!allowed.has(field)) throw new ValidationError(`runSelectionRule: round-robin reads '${field}' but its closed read set is [${rule.reads.join(", ")}]`);
91
- const byGroup = /* @__PURE__ */ new Map();
92
- for (const record of records) {
93
- const group = String(readField(record, rule.groupBy));
94
- const id = String(readField(record, rule.withinOrder.field));
95
- const bucket = byGroup.get(group);
96
- if (bucket) bucket.push(id);
97
- else byGroup.set(group, [id]);
98
- }
99
- for (const ids of byGroup.values()) {
100
- ids.sort();
101
- if (rule.withinOrder.dir === "desc") ids.reverse();
102
- }
103
- const groups = [...byGroup.keys()].sort();
104
- const chosen = [];
105
- let cursor = 0;
106
- while (chosen.length < rule.take && groups.some((g) => byGroup.get(g).length > 0)) {
107
- const group = groups[cursor % groups.length];
108
- const ids = byGroup.get(group);
109
- if (ids.length > 0) chosen.push(ids.shift());
110
- cursor += 1;
111
- }
112
- return chosen;
113
- }
114
- const base = options.bases?.[rule.base];
115
- if (!base) throw new ValidationError(`runSelectionRule: filter-of base '${rule.base}' was not provided`);
116
- const index = new Map(records.map((r) => [String(readField(r, options.idField)), r]));
117
- const sorted = [...base.filter((id) => {
118
- const record = index.get(id) ?? options.idFields?.(id);
119
- if (!record) throw new ValidationError(`runSelectionRule: base id '${id}' has no record and no idFields derivation`);
120
- return evaluatePredicate(rule.keep, record);
121
- })].sort();
122
- if (rule.order.dir === "desc") sorted.reverse();
123
- return sorted;
124
- }
125
- function evaluateSetExpr(expr, rows, armField, idField) {
126
- if (expr.kind === "rows-where") {
127
- const ids = /* @__PURE__ */ new Set();
128
- for (const row of rows) {
129
- if (String(readField(row, armField)) !== expr.arm) continue;
130
- if (evaluatePredicate(expr.event, row)) ids.add(String(readField(row, idField)));
131
- }
132
- return ids;
133
- }
134
- if (expr.of.length === 0) throw new ValidationError("computeEstimand: empty intersect");
135
- const [first, ...rest] = expr.of.map((e) => evaluateSetExpr(e, rows, armField, idField));
136
- const out = /* @__PURE__ */ new Set();
137
- for (const id of first) if (rest.every((s) => s.has(id))) out.add(id);
138
- return out;
139
- }
140
- /**
141
- * Read one registered outcome field as a number.
142
- *
143
- * A binary outcome reaches evidence as `true` or `false`. Its mean is the pass
144
- * rate and the mean of its paired differences is the risk difference, so
145
- * `true` reads as 1 and `false` as 0 — the same quantity a caller would
146
- * otherwise encode by hand, and the same reading in every interpreter here.
147
- * Every other type, and a non-finite number, is a measurement defect: it
148
- * rejects instead of poisoning the mean with `NaN` or a coerced zero.
149
- */
150
- function readNumericOutcome(raw, context, field, where) {
151
- if (typeof raw === "boolean") return raw ? 1 : 0;
152
- if (typeof raw === "number" && Number.isFinite(raw)) return raw;
153
- throw new ValidationError(`${context}: value field '${field}' is not a finite number or a boolean on ${where}`);
154
- }
155
- /** Compute an estimand over evidence rows. Pure; reads only registered fields. */
156
- function computeEstimand(estimand, rows) {
157
- switch (estimand.kind) {
158
- case "rate": {
159
- const numerator = rows.filter((r) => evaluatePredicate(estimand.event, r)).length;
160
- if (rows.length === 0) throw new ValidationError("computeEstimand: rate over zero rows");
161
- return {
162
- value: numerator / rows.length,
163
- numerator,
164
- denominator: rows.length
165
- };
166
- }
167
- case "rate-at-least-once": {
168
- const byGroup = /* @__PURE__ */ new Map();
169
- for (const row of rows) {
170
- const group = String(readField(row, estimand.groupBy));
171
- const hit = evaluatePredicate(estimand.event, row);
172
- byGroup.set(group, (byGroup.get(group) ?? false) || hit);
173
- }
174
- if (byGroup.size === 0) throw new ValidationError("computeEstimand: rate-at-least-once over zero groups");
175
- const numerator = [...byGroup.values()].filter(Boolean).length;
176
- return {
177
- value: numerator / byGroup.size,
178
- numerator,
179
- denominator: byGroup.size
180
- };
181
- }
182
- case "paired-mean-diff": {
183
- const byPair = /* @__PURE__ */ new Map();
184
- for (const row of rows) {
185
- const arm = String(readField(row, estimand.armField));
186
- if (arm !== estimand.treatment && arm !== estimand.control) continue;
187
- const pair = String(readField(row, estimand.pairBy));
188
- const value = readNumericOutcome(readField(row, estimand.value), "computeEstimand paired-mean-diff", estimand.value, `pair '${pair}'`);
189
- const slot = byPair.get(pair) ?? {};
190
- if (arm === estimand.treatment) slot.treatment = value;
191
- else slot.control = value;
192
- byPair.set(pair, slot);
193
- }
194
- if (byPair.size === 0) throw new ValidationError("computeEstimand: paired-mean-diff over zero pairs");
195
- let sum = 0;
196
- for (const slot of byPair.values()) sum += (slot.treatment ?? 0) - (slot.control ?? 0);
197
- return {
198
- value: sum / byPair.size,
199
- numerator: sum,
200
- denominator: byPair.size
201
- };
202
- }
203
- case "set-ratio": {
204
- const numeratorSet = evaluateSetExpr(estimand.numerator, rows, estimand.armField, estimand.idField);
205
- const denominatorSet = evaluateSetExpr(estimand.denominator, rows, estimand.armField, estimand.idField);
206
- if (denominatorSet.size === 0) throw new ValidationError("computeEstimand: set-ratio denominator set is empty");
207
- return {
208
- value: numeratorSet.size / denominatorSet.size,
209
- numerator: numeratorSet.size,
210
- denominator: denominatorSet.size
211
- };
212
- }
213
- }
214
- }
215
- /**
216
- * Execute an interval spec.
217
- *
218
- * Cluster-bootstrap resamples whole clusters of the per-row `value` field and
219
- * takes percentile bounds of the pooled mean. Clopper-Pearson computes the
220
- * exact binomial interval and requires `successes`/`trials` evidence instead
221
- * of rows.
222
- */
223
- function computeInterval(spec, evidence) {
224
- if (spec.kind === "cluster-bootstrap") {
225
- if (evidence.kind !== "rows") throw new ValidationError("computeInterval: cluster-bootstrap requires row evidence");
226
- const clusters = /* @__PURE__ */ new Map();
227
- for (const row of evidence.rows) {
228
- const cluster = String(readField(row, spec.clusterBy));
229
- const value = readNumericOutcome(readField(row, evidence.value), "computeInterval cluster-bootstrap", evidence.value, `cluster '${cluster}'`);
230
- const bucket = clusters.get(cluster);
231
- if (bucket) bucket.push(value);
232
- else clusters.set(cluster, [value]);
233
- }
234
- const clusterValues = [...clusters.entries()].sort(([a], [b]) => a < b ? -1 : a > b ? 1 : 0).map(([, values]) => values);
235
- if (clusterValues.length < 2) throw new ValidationError(`computeInterval: cluster-bootstrap needs >= 2 clusters, got ${clusterValues.length}`);
236
- const rng = mulberry32(spec.seed);
237
- const means = new Array(spec.resamples);
238
- for (let draw = 0; draw < spec.resamples; draw++) {
239
- let sum = 0;
240
- let count = 0;
241
- for (let pick = 0; pick < clusterValues.length; pick++) {
242
- const cluster = clusterValues[Math.floor(rng() * clusterValues.length)];
243
- for (const value of cluster) sum += value;
244
- count += cluster.length;
245
- }
246
- means[draw] = sum / count;
247
- }
248
- means.sort((a, b) => a - b);
249
- const alpha = 1 - spec.level;
250
- const lowerIndex = Math.floor(alpha / 2 * spec.resamples);
251
- const upperIndex = Math.min(spec.resamples - 1, Math.ceil((1 - alpha / 2) * spec.resamples) - 1);
252
- return {
253
- lower: means[lowerIndex],
254
- upper: means[Math.max(lowerIndex, upperIndex)],
255
- level: spec.level
256
- };
257
- }
258
- if (evidence.kind !== "binomial") throw new ValidationError("computeInterval: clopper-pearson requires binomial evidence");
259
- const { successes, trials } = evidence;
260
- if (!Number.isInteger(successes) || !Number.isInteger(trials) || trials <= 0 || successes < 0) throw new ValidationError(`computeInterval: clopper-pearson needs 0 <= successes <= trials, got ${successes}/${trials}`);
261
- if (successes > trials) throw new ValidationError(`computeInterval: clopper-pearson successes ${successes} exceed trials ${trials}`);
262
- const alpha = 1 - spec.level;
263
- return {
264
- lower: successes === 0 ? 0 : binomialQuantile(successes, trials, alpha / 2, "lower"),
265
- upper: successes === trials ? 1 : binomialQuantile(successes, trials, alpha / 2, "upper"),
266
- level: spec.level
267
- };
268
- }
269
- /**
270
- * Clopper-Pearson bound by bisection on the binomial tail. The lower bound is
271
- * the p with P(X >= successes | p) = alpha; the upper is the p with
272
- * P(X <= successes | p) = alpha. Deterministic, no special functions.
273
- */
274
- function binomialQuantile(successes, trials, alpha, side) {
275
- const tail = (p) => {
276
- let sum = 0;
277
- for (let k = 0; k <= trials; k++) {
278
- if (!(side === "lower" ? k >= successes : k <= successes)) continue;
279
- sum += Math.exp(logBinomialPmf(k, trials, p));
280
- }
281
- return sum;
282
- };
283
- let lo = 0;
284
- let hi = 1;
285
- for (let iter = 0; iter < 100; iter++) {
286
- const mid = (lo + hi) / 2;
287
- if (tail(mid) < alpha) if (side === "lower") lo = mid;
288
- else hi = mid;
289
- else if (side === "lower") hi = mid;
290
- else lo = mid;
291
- }
292
- return (lo + hi) / 2;
293
- }
294
- function logBinomialPmf(k, n, p) {
295
- if (p <= 0) return k === 0 ? 0 : Number.NEGATIVE_INFINITY;
296
- if (p >= 1) return k === n ? 0 : Number.NEGATIVE_INFINITY;
297
- return logChoose(n, k) + k * Math.log(p) + (n - k) * Math.log(1 - p);
298
- }
299
- function logChoose(n, k) {
300
- return logFactorial(n) - logFactorial(k) - logFactorial(n - k);
301
- }
302
- const LOG_FACTORIAL_CACHE = [0];
303
- function logFactorial(n) {
304
- for (let i = LOG_FACTORIAL_CACHE.length; i <= n; i++) LOG_FACTORIAL_CACHE[i] = LOG_FACTORIAL_CACHE[i - 1] + Math.log(i);
305
- return LOG_FACTORIAL_CACHE[n];
306
- }
307
- function evaluateCondition(condition, evidence) {
308
- switch (condition.kind) {
309
- case "interval-excludes-zero": {
310
- const interval = evidence.intervals[condition.interval];
311
- if (!interval) throw new ValidationError(`evaluateCondition: interval '${condition.interval}' is not in the evidence`);
312
- return (interval.lower > 0 || interval.upper < 0) && (condition.sign === "positive" ? interval.lower > 0 : interval.upper < 0);
313
- }
314
- case "interval-includes-zero": {
315
- const interval = evidence.intervals[condition.interval];
316
- if (!interval) throw new ValidationError(`evaluateCondition: interval '${condition.interval}' is not in the evidence`);
317
- return interval.lower <= 0 && interval.upper >= 0;
318
- }
319
- case "quantity-threshold": {
320
- const value = evidence.quantities[condition.quantity];
321
- if (value === void 0) throw new ValidationError(`evaluateCondition: quantity '${condition.quantity}' is not in the evidence`);
322
- return compareValues(value, condition.op, condition.value);
323
- }
324
- case "obligation-met": return evidence.obligationsMet[condition.obligation] === true;
325
- case "all": return condition.of.every((c) => evaluateCondition(c, evidence));
326
- case "any": return condition.of.some((c) => evaluateCondition(c, evidence));
327
- case "not": return !evaluateCondition(condition.of, evidence);
328
- }
329
- }
330
- function executeDecisionRule(rule, evidence) {
331
- if (rule.kind === "report-only") return {
332
- verdict: "report-only",
333
- report: [...rule.estimands, ...rule.intervals]
334
- };
335
- for (const branch of rule.branches) if (evaluateCondition(branch.when, evidence)) return {
336
- verdict: branch.verdict,
337
- report: branch.report
338
- };
339
- throw new DecisionTableNotTotalError("executeDecisionRule: decision table is not total — no branch matched the evidence");
340
- }
341
- /**
342
- * Replicate-flip counting over graded states. A state whose replicates split
343
- * between pass and fail is flipping; its flip rate is the minority share.
344
- */
345
- function evaluateOracleDeterminismGate(id, gate, repsByState) {
346
- const evidence = {};
347
- let passed = true;
348
- for (const [state, reps] of Object.entries(repsByState)) {
349
- const passes = reps.filter(Boolean).length;
350
- const flipRate = reps.length === 0 ? 0 : Math.min(passes, reps.length - passes) / reps.length;
351
- evidence[state] = {
352
- passes,
353
- replicates: reps.length,
354
- flipRate
355
- };
356
- if (flipRate > gate.maxFlipRate) passed = false;
357
- }
358
- return {
359
- id,
360
- passed,
361
- evidence
362
- };
363
- }
364
- /**
365
- * Join two population snapshots on `joinOn` and compare the registered fields.
366
- * Only rows present in both snapshots are compared; a presence change is a
367
- * different failure and needs its own gate.
368
- */
369
- function evaluatePopulationReproducibilityGate(id, gate, populations) {
370
- const rightByKey = new Map(populations.right.map((r) => [String(readField(r, gate.joinOn)), r]));
371
- const changed = [];
372
- for (const left of populations.left) {
373
- const key = String(readField(left, gate.joinOn));
374
- const right = rightByKey.get(key);
375
- if (!right) continue;
376
- const moved = gate.compare.filter((f) => readField(left, f) !== readField(right, f));
377
- if (moved.length > 0) changed.push(`${key} ${moved.map((f) => `${f}:${String(readField(left, f))}->${String(readField(right, f))}`).join(" ")}`);
378
- }
379
- return {
380
- id,
381
- passed: changed.length <= gate.maxChangedRows,
382
- evidence: changed
383
- };
384
- }
385
- /** The registered claim about provenance must hold on the provenance record. */
386
- function evaluateProvenanceGate(id, gate, provenance) {
387
- const passed = evaluatePredicate(gate.claim, provenance);
388
- return {
389
- id,
390
- passed,
391
- evidence: { claimHolds: passed }
392
- };
393
- }
394
- /** Final path segment equality between the pinned and the served identity. */
395
- function evaluateIdentityGate(id, _gate, identities) {
396
- const basename = (s) => s.split("/").pop() ?? s;
397
- const passed = basename(identities.pinned) === basename(identities.served);
398
- return {
399
- id,
400
- passed,
401
- evidence: {
402
- pinned: identities.pinned,
403
- served: identities.served,
404
- matched: passed
405
- }
406
- };
407
- }
408
- /**
409
- * The design's power curve must reach the registered target at some grid
410
- * effect. The curve must cover the registered effect grid exactly — a curve
411
- * computed on a different grid is different evidence and is refused.
412
- */
413
- function evaluatePowerFloorGate(id, gate, curve) {
414
- const byEffect = new Map(curve.map((point) => [point.effect, point.power]));
415
- const missing = gate.effectGrid.filter((effect) => !byEffect.has(effect));
416
- if (missing.length > 0) throw new ValidationError(`evaluatePowerFloorGate: curve does not cover registered effects [${missing.join(", ")}]`);
417
- const powers = gate.effectGrid.map((effect) => byEffect.get(effect));
418
- const maxPower = Math.max(...powers);
419
- return {
420
- id,
421
- passed: maxPower >= gate.target,
422
- evidence: {
423
- target: gate.target,
424
- maxPower,
425
- curve: gate.effectGrid.map((effect) => ({
426
- effect,
427
- power: byEffect.get(effect)
428
- }))
429
- }
430
- };
431
- }
432
- function evaluateHaltRule(halt, gates) {
433
- const seen = new Map(gates.map((g) => [g.id, g]));
434
- const missing = halt.when.gates.filter((id) => !seen.has(id));
435
- if (missing.length > 0) throw new ValidationError(`evaluateHaltRule: halt references gates that were not evaluated: [${missing.join(", ")}]`);
436
- const failed = halt.when.gates.filter((id) => !seen.get(id).passed);
437
- return failed.length > 0 ? {
438
- fired: true,
439
- action: halt.action,
440
- failedGates: failed
441
- } : {
442
- fired: false,
443
- action: null,
444
- failedGates: []
445
- };
446
- }
447
- /**
448
- * Execute the uniform-pass schedule against measured pass costs. Pass 1 always
449
- * runs; each later pass runs only when the cumulative spend plus the last
450
- * measured pass cost stays at or under the registered ceiling. The registered
451
- * ledger is the pre-spend the ceiling counts.
452
- */
453
- function runUniformPassBudget(rule, measuredPassCosts) {
454
- let cumulative = rule.ledger.reduce((sum, entry) => sum + entry.usd, 0);
455
- const decisions = [];
456
- let uniformN = 0;
457
- for (let pass = 1; pass <= rule.maxPasses; pass++) {
458
- if (pass === 1) {
459
- if (measuredPassCosts[0] === void 0) break;
460
- cumulative += measuredPassCosts[0];
461
- uniformN = 1;
462
- continue;
463
- }
464
- const projected = measuredPassCosts[pass - 2];
465
- if (projected === void 0) break;
466
- const go = cumulative + projected <= rule.ceilingUsd;
467
- decisions.push({
468
- pass,
469
- cumulativeBefore: cumulative,
470
- projected,
471
- go
472
- });
473
- if (!go || measuredPassCosts[pass - 1] === void 0) break;
474
- cumulative += measuredPassCosts[pass - 1];
475
- uniformN = pass;
476
- }
477
- return {
478
- decisions,
479
- uniformN
480
- };
481
- }
482
- /**
483
- * Walk the registered n-ladder and pick the first affordable step. When no
484
- * step fits the ceiling, the rule refuses and reports the projection instead
485
- * of shrinking the row set — "never subset rows" is the registered invariant.
486
- */
487
- function projectNLadderBudget(rule, measured) {
488
- const projections = rule.steps.map((n) => {
489
- const projectedUsd = measured.unitCostUsd * measured.rows * n;
490
- return {
491
- n,
492
- projectedUsd,
493
- affordable: projectedUsd <= rule.ceilingUsd
494
- };
495
- });
496
- const first = projections.find((p) => p.affordable);
497
- if (first) return {
498
- chosenN: first.n,
499
- projections,
500
- refusal: null
501
- };
502
- return {
503
- chosenN: null,
504
- projections,
505
- refusal: {
506
- onExhaust: rule.onExhaust,
507
- reason: `no ladder step fits the ${rule.ceilingUsd} USD ceiling at ${measured.rows} rows x ${measured.unitCostUsd} USD per unit`
508
- }
509
- };
510
- }
511
- /**
512
- * Classify one rollout event under the registered reissue policy. A carrier
513
- * event within the issue budget is reissued; a model outcome always stands;
514
- * a carrier event past `maxIssues` is exhausted and reported, never retried.
515
- */
516
- function classifyReissue(policy, event, issuesSoFar) {
517
- if (!policy.carrierEvents.includes(event)) return "stands";
518
- return issuesSoFar < policy.maxIssues ? "reissue" : "exhausted";
519
- }
520
- //#endregion
521
16
  //#region src/experiment/budget.ts
522
17
  /**
523
18
  * Matched-budget verification between arms, as a refusal object.
@@ -805,6 +300,7 @@ function conditionRefs(condition) {
805
300
  */
806
301
  function defineExperiment(spec) {
807
302
  const problems = [];
303
+ const claim = spec.claim === void 0 ? void 0 : defineEvaluationClaim(spec.claim);
808
304
  if (!spec.id || spec.id.trim().length === 0) problems.push("id is empty");
809
305
  if (spec.arms.length === 0) problems.push("at least one arm is required");
810
306
  const armIds = /* @__PURE__ */ new Set();
@@ -819,6 +315,15 @@ function defineExperiment(spec) {
819
315
  const gateNames = new Set(Object.keys(spec.gates ?? {}));
820
316
  const selectionNames = new Set(Object.keys(spec.selections ?? {}));
821
317
  const sealedSubsetNames = new Set(Object.keys(spec.sealedSubsets ?? {}));
318
+ for (const [name, interval] of Object.entries(spec.intervals ?? {})) problems.push(...intervalSpecProblems(interval).map((problem) => `interval '${name}': ${problem}`));
319
+ for (const [name, gate] of Object.entries(spec.gates ?? {})) {
320
+ if (gate.kind !== "power-floor") continue;
321
+ problems.push(...powerFloorProblems(gate).map((problem) => `gate '${name}': ${problem}`));
322
+ if (claim?.minimumEffect !== void 0 && gate.minimumEffect !== claim.minimumEffect) problems.push(`gate '${name}' minimumEffect differs from the evaluation claim`);
323
+ }
324
+ if (claim?.generalization === "new-units") {
325
+ for (const [name, interval] of Object.entries(spec.intervals ?? {})) if (interval?.kind === "cluster-bootstrap" && interval.clusterBy !== claim.independentUnit) problems.push(`interval '${name}' must resample '${claim.independentUnit}' from the evaluation claim`);
326
+ }
822
327
  const checkCondition = (condition, where) => {
823
328
  const refs = conditionRefs(condition);
824
329
  for (const name of refs.intervals) if (!intervalNames.has(name)) problems.push(`${where} reads unregistered interval '${name}'`);
@@ -853,7 +358,10 @@ function defineExperiment(spec) {
853
358
  for (const partition of spec.admission.partitions ?? []) if (!stageIds.has(partition.from)) problems.push(`admission partition '${partition.id}' draws from unknown stage '${partition.from}'`);
854
359
  }
855
360
  if (problems.length > 0) throw new ValidationError(`defineExperiment('${spec.id}'): ${problems.join("; ")}`);
856
- return deepFreeze(structuredClone(spec));
361
+ return deepFreeze(structuredClone({
362
+ ...spec,
363
+ ...claim ? { claim } : {}
364
+ }));
857
365
  }
858
366
  function deepFreeze(value) {
859
367
  if (value !== null && typeof value === "object") {
@@ -865,7 +373,7 @@ function deepFreeze(value) {
865
373
  /** Validate, canonicalize, and hash a spec into its registration. */
866
374
  async function sealExperiment(spec, options = {}) {
867
375
  const validated = defineExperiment(spec);
868
- const digest = specDigest(validated, "sha256-rfc8785");
376
+ const digest = specDigest(validated);
869
377
  return {
870
378
  spec: validated,
871
379
  digest,
@@ -881,46 +389,39 @@ async function sealExperiment(spec, options = {}) {
881
389
  * There is no way to change what is decided without producing a new digest.
882
390
  */
883
391
  async function amendExperiment(sealed, amendment) {
884
- await assertSealIntact(sealed);
885
- const validated = defineExperiment(amendment.spec);
886
- const digest = specDigest(validated, "sha256-rfc8785");
392
+ const captured = structuredClone(sealed);
393
+ const requested = structuredClone(amendment);
394
+ await assertSealIntact(captured);
395
+ const validated = defineExperiment(requested.spec);
396
+ const digest = specDigest(validated);
887
397
  return {
888
398
  spec: validated,
889
399
  digest,
890
400
  algo: "sha256-rfc8785",
891
- sealedAt: sealed.sealedAt,
892
- initialDigest: sealed.initialDigest,
893
- amendments: [...sealed.amendments, {
894
- at: amendment.at ?? (/* @__PURE__ */ new Date()).toISOString(),
895
- reason: amendment.reason,
896
- blind: [...amendment.blind],
401
+ sealedAt: captured.sealedAt,
402
+ initialDigest: captured.initialDigest,
403
+ amendments: [...captured.amendments, {
404
+ at: requested.at ?? (/* @__PURE__ */ new Date()).toISOString(),
405
+ reason: requested.reason,
406
+ blind: [...requested.blind],
897
407
  digest
898
408
  }]
899
409
  };
900
410
  }
901
- /** True when the sealed digest still matches the spec it carries, under the
902
- * scheme the seal declares. */
411
+ /** True when the supported seal matches its canonical spec. */
903
412
  async function verifySealedExperiment(sealed) {
904
- return specDigest(sealed.spec, sealed.algo) === sealed.digest;
905
- }
906
- /**
907
- * Serialize a spec under `algo` and digest it. `'sha256-content'` is read-only
908
- * — it exists so a seal written by an earlier release still verifies, and no
909
- * path that WRITES a digest may pass it.
910
- */
911
- function specDigest(spec, algo) {
912
- const serialized = algo === "sha256-rfc8785" ? canonicalString(spec) : JSON.stringify(sortKeysDeep(spec));
913
- return createHash("sha256").update(serialized, "utf8").digest("hex");
413
+ if (sealed.algo !== "sha256-rfc8785") return false;
414
+ try {
415
+ return specDigest(sealed.spec) === sealed.digest;
416
+ } catch {
417
+ return false;
418
+ }
914
419
  }
915
- function sortKeysDeep(value) {
916
- if (value === null || typeof value !== "object") return value;
917
- if (Array.isArray(value)) return value.map(sortKeysDeep);
918
- const out = {};
919
- for (const key of Object.keys(value).sort()) out[key] = sortKeysDeep(value[key]);
920
- return out;
420
+ function specDigest(spec) {
421
+ return hashCanonical(spec).slice(7);
921
422
  }
922
423
  async function assertSealIntact(sealed) {
923
- if (!await verifySealedExperiment(sealed)) throw new SealIntegrityError(`sealed experiment '${sealed.spec.id}' digest ${sealed.digest} does not match its spec — the registration was tampered with`);
424
+ if (!await verifySealedExperiment(sealed)) throw new SealIntegrityError(`sealed experiment '${sealed.spec.id}' has an unsupported digest scheme or its digest does not match its canonical spec`);
924
425
  }
925
426
  /**
926
427
  * Verify the seal and return executors bound to it. This is the module's only
@@ -928,14 +429,20 @@ async function assertSealIntact(sealed) {
928
429
  * rule that is cannot run differently.
929
430
  */
930
431
  async function openSealedExperiment(sealed) {
931
- await assertSealIntact(sealed);
932
- const spec = sealed.spec;
432
+ const captured = structuredClone(sealed);
433
+ await assertSealIntact(captured);
434
+ const spec = defineExperiment(captured.spec);
435
+ const frozenSeal = deepFreeze({
436
+ ...captured,
437
+ spec
438
+ });
933
439
  const need = (value, what) => {
934
440
  if (value === void 0) throw new ValidationError(`experiment '${spec.id}' registered no ${what}`);
935
441
  return value;
936
442
  };
937
443
  return {
938
- sealed,
444
+ sealed: frozenSeal,
445
+ units: (records) => summarizeEvaluationUnits(need(spec.claim, "evaluation claim"), records),
939
446
  decide: (evidence) => executeDecisionRule(spec.decision, evidence),
940
447
  admit: (records) => executeAdmissionRule(need(spec.admission, "admission rule"), records),
941
448
  select: (name, records, options) => {
@@ -968,7 +475,27 @@ async function openSealedExperiment(sealed) {
968
475
  },
969
476
  matchedBudgets: (arms) => verifyMatchedBudgets(need(spec.matchedBudget, "matched-budget rule"), arms),
970
477
  estimate: (name, rows) => computeEstimand(need(spec.estimands?.[name], `estimand '${name}'`), rows),
971
- interval: (name, evidence) => computeInterval(need(spec.intervals?.[name], `interval '${name}'`), evidence)
478
+ interval: (name, evidence) => {
479
+ const interval = need(spec.intervals?.[name], `interval '${name}'`);
480
+ const claim = spec.claim;
481
+ let units;
482
+ if (claim && evidence.kind === "rows") units = summarizeEvaluationUnits(claim, evidence.rows);
483
+ else if (claim?.generalization === "new-units" && evidence.kind === "binomial") {
484
+ if (evidence.unitIds === void 0 || evidence.unitIds.length !== evidence.trials || new Set(evidence.unitIds).size !== evidence.trials || evidence.unitIds.some((id) => typeof id !== "string" || !id.trim() || id.trim() !== id)) throw new ValidationError("claimed binomial interval needs one unique unitId per independent trial");
485
+ units = {
486
+ observations: evidence.trials,
487
+ independentUnits: evidence.trials,
488
+ units: evidence.unitIds.map((id) => ({
489
+ id,
490
+ observations: 1
491
+ }))
492
+ };
493
+ }
494
+ return {
495
+ ...computeInterval(interval, evidence),
496
+ ...units ? { units } : {}
497
+ };
498
+ }
972
499
  };
973
500
  }
974
501
  //#endregion
@@ -1173,25 +700,33 @@ function renderEvidenceIndex(raws) {
1173
700
  var DesignRefusalError = class extends ValidationError {};
1174
701
  /**
1175
702
  * Simulate the power of a whole-cluster percentile-bootstrap design and refuse
1176
- * a structure that cannot reach the target at any registered effect.
703
+ * a structure that cannot reach target power at the declared minimum effect.
1177
704
  */
1178
705
  function clusteredPower(options) {
1179
706
  const clusterSizes = options.clusterSizes;
1180
707
  if (clusterSizes.length === 0 || clusterSizes.some((n) => !Number.isInteger(n) || n <= 0)) throw new ValidationError(`clusteredPower: clusterSizes must be positive integers, got [${clusterSizes.join(", ")}]`);
1181
708
  if (options.effects.length === 0) throw new ValidationError("clusteredPower: effects grid is empty");
709
+ if (options.effects.some((effect) => !Number.isFinite(effect) || effect < 0 || effect > 1) || new Set(options.effects).size !== options.effects.length) throw new ValidationError("clusteredPower: effects must be unique finite values in [0,1]");
710
+ if (!Number.isFinite(options.minimumEffect) || options.minimumEffect <= 0 || options.minimumEffect > 1) throw new ValidationError("clusteredPower: minimumEffect must be in (0,1]");
711
+ if (!options.effects.includes(options.minimumEffect)) throw new ValidationError("clusteredPower: effects must contain minimumEffect exactly; no interpolation is assumed");
1182
712
  if (!Number.isInteger(options.seed)) throw new ValidationError(`clusteredPower: seed must be an integer, got ${options.seed}`);
1183
713
  const trials = options.trials ?? 2e3;
1184
714
  const resamples = options.resamples ?? 4e3;
1185
715
  if (!Number.isInteger(trials) || trials <= 0 || !Number.isInteger(resamples) || resamples <= 0) throw new ValidationError(`clusteredPower: trials and resamples must be positive integers, got ${trials}/${resamples}`);
1186
716
  const confidence = options.confidence ?? .95;
1187
- if (confidence <= 0 || confidence >= 1) throw new ValidationError(`clusteredPower: confidence must be in (0,1), got ${confidence}`);
717
+ if (!Number.isFinite(confidence) || confidence <= 0 || confidence >= 1) throw new ValidationError(`clusteredPower: confidence must be in (0,1), got ${confidence}`);
1188
718
  const alpha = options.alpha ?? .05;
1189
719
  const targetPower = options.targetPower ?? .8;
720
+ if (!Number.isFinite(alpha) || alpha <= 0 || alpha >= 1) throw new ValidationError("clusteredPower: alpha must be in (0,1)");
721
+ if (!Number.isFinite(targetPower) || targetPower <= 0 || targetPower > 1) throw new ValidationError("clusteredPower: targetPower must be in (0,1]");
1190
722
  const baseWinRate = options.baseWinRate ?? .1;
1191
723
  const baseLossRate = options.baseLossRate ?? .1;
724
+ if (!Number.isFinite(baseWinRate) || !Number.isFinite(baseLossRate) || baseWinRate < 0 || baseLossRate < 0 || baseWinRate + baseLossRate > 1) throw new ValidationError("clusteredPower: base win/loss rates must be nonnegative and sum to at most 1");
725
+ if (baseWinRate !== baseLossRate) throw new ValidationError("clusteredPower: the zero-effect model requires equal baseWinRate and baseLossRate");
1192
726
  const noisy = /* @__PURE__ */ new Map();
1193
727
  for (const cluster of options.noisyClusters ?? []) {
1194
- if (cluster.index < 0 || cluster.index >= clusterSizes.length) throw new ValidationError(`clusteredPower: noisy cluster index ${cluster.index} outside [0,${clusterSizes.length - 1}]`);
728
+ if (!Number.isInteger(cluster.index) || cluster.index < 0 || cluster.index >= clusterSizes.length) throw new ValidationError(`clusteredPower: noisy cluster index ${cluster.index} outside [0,${clusterSizes.length - 1}]`);
729
+ if (noisy.has(cluster.index) || !Number.isFinite(cluster.flipRate) || cluster.flipRate < 0 || cluster.flipRate > 1) throw new ValidationError("clusteredPower: noisy clusters need unique indices and flipRate in [0,1]");
1195
730
  noisy.set(cluster.index, cluster.flipRate);
1196
731
  }
1197
732
  const clusterCount = clusterSizes.length;
@@ -1209,12 +744,10 @@ function clusteredPower(options) {
1209
744
  noisy
1210
745
  }));
1211
746
  const maxPower = Math.max(...curve.map((point) => point.power));
747
+ const powerAtMinimumEffect = curve.find((point) => point.effect === options.minimumEffect).power;
1212
748
  const reasons = [];
1213
749
  if (!signFlipFloor.certifiableAtAlpha) reasons.push(`${clusterCount} clusters cannot certify any effect size, including 1.0: the exact whole-cluster sign-flip test's smallest two-sided p is 2^(1-${clusterCount}) = ${signFlipFloor.twoSidedP} > alpha ${alpha}; at least ${signFlipFloor.minClustersForAlpha} clusters are needed`);
1214
- if (maxPower < targetPower) {
1215
- const best = curve.reduce((a, b) => b.power > a.power ? b : a);
1216
- reasons.push(`simulated power tops out at ${maxPower.toFixed(3)} (effect ${best.effect}) across the registered grid — below the ${targetPower} target at every effect`);
1217
- }
750
+ if (powerAtMinimumEffect < targetPower) reasons.push(`simulated power ${powerAtMinimumEffect.toFixed(3)} at minimum worthwhile effect ${options.minimumEffect} is below target ${targetPower}; maximum grid power ${maxPower.toFixed(3)} does not establish adequacy at that effect`);
1218
751
  const adequate = reasons.length === 0;
1219
752
  return {
1220
753
  clusterCount,
@@ -1224,6 +757,8 @@ function clusteredPower(options) {
1224
757
  seed: options.seed,
1225
758
  confidence,
1226
759
  targetPower,
760
+ minimumEffect: options.minimumEffect,
761
+ powerAtMinimumEffect,
1227
762
  curve,
1228
763
  maxPower,
1229
764
  signFlipFloor,
@@ -1231,7 +766,7 @@ function clusteredPower(options) {
1231
766
  refusal: adequate ? null : {
1232
767
  verdict: "underpowered",
1233
768
  reasons,
1234
- recommendation: `Do not spend on this structure. Add independent clusters (>= ${Math.max(signFlipFloor.minClustersForAlpha, clusterCount)}) or register a design whose simulated power reaches ${targetPower}, then re-run clusteredPower.`
769
+ recommendation: `Do not spend on this structure. Add independent clusters (>= ${Math.max(signFlipFloor.minClustersForAlpha, clusterCount)}) or register a design whose simulated power reaches ${targetPower} at effect ${options.minimumEffect}, then re-run clusteredPower.`
1235
770
  }
1236
771
  };
1237
772
  }
@@ -1253,9 +788,10 @@ function computeSignFlipFloor(clusterCount, alpha) {
1253
788
  };
1254
789
  }
1255
790
  /**
1256
- * One effect point. Per row the paired contrast is +1 with probability
1257
- * min(baseWin + effect, 1), -1 with the base loss rate (capped by what
1258
- * remains), else 0. Noisy clusters draw win and loss independently at their
791
+ * One effect point. Loss probability is min(baseLoss, (1-effect)/2) and
792
+ * win probability is loss probability plus effect. Their difference equals
793
+ * effect, including near the probability boundary. Noisy clusters draw win
794
+ * and loss independently at their
1259
795
  * flip rate. Each trial computes a whole-cluster percentile bootstrap of the
1260
796
  * pooled row mean; the trial counts toward power when the interval excludes
1261
797
  * zero.
@@ -1277,8 +813,8 @@ function simulateEffect(effect, config) {
1277
813
  const loss = rng() < flipRate ? 1 : 0;
1278
814
  sum += win - loss;
1279
815
  } else {
1280
- const winRate = Math.min(1, config.baseWinRate + effect);
1281
- const lossRate = Math.min(1 - winRate, config.baseLossRate);
816
+ const lossRate = Math.min(config.baseLossRate, (1 - effect) / 2);
817
+ const winRate = lossRate + effect;
1282
818
  const u = rng();
1283
819
  sum += u < winRate ? 1 : u < winRate + lossRate ? -1 : 0;
1284
820
  }
@@ -1313,6 +849,6 @@ function mixSeed(seed, effect) {
1313
849
  return (seed ^ Math.round(effect * 1000003) * 2654435769) >>> 0 | 0;
1314
850
  }
1315
851
  //#endregion
1316
- export { BOOTSTRAP_GATE_MIN_N, DesignRefusalError, EVIDENCE_AUTHORITY_KINDS, EVIDENCE_RECEIPT_VERSION, EVIDENCE_STATES, ExperimentTracker, FunnelIntegrityError, INDEPENDENT_EVIDENCE_AUTHORITY_KINDS, MatchedBudgetError, SealIntegrityError, amendExperiment, assertDesignAdequate, assertFunnelReconciles, assertMatchedBudgets, benjaminiHochberg, bonferroni, buildEvidenceVector, buildFunnel, classifyReissue, clusteredPower, comparePairedArms, composeFunnels, computeEstimand, computeInterval, createCampaignEvidenceReceipt, createEvidenceReceipt, defineExperiment, eProcess, evaluateCondition, evaluateHaltRule, evaluateHypothesis, evaluateIdentityGate, evaluateOracleDeterminismGate, evaluatePopulationReproducibilityGate, evaluatePowerFloorGate, evaluatePredicate, evaluateProvenanceGate, evidenceRegistryRecordSchema, executeAdmissionRule, executeDecisionRule, fileExperimentStore, hashJson, heldoutSignificance, holm, inMemoryExperimentStore, isIndependentEvidence, manifestContentDigest, mcnemar, mcnemarPower, mcnemarRequiredN, mulberry32, openSealedExperiment, pairArms, pairHoldout, pairRunRecords, pairedBootstrap, pairedEvalueSequence, pairedMde, pairedRiskDifference, pairedRiskDifferenceExact, pairedRiskDifferenceScore, paretoPolicy, paretoSignificanceGate, parseEvidenceRegistryRecord, powerPreflight, projectNLadderBudget, readField, renderEvidenceIndex, renderFunnelTable, requiredPairedSampleSize, requiredSampleSize, runSelectionRule, runUniformPassBudget, sealExperiment, sequentialCrossingHorizon, sequentialDecide, sequentialPairedGate, signManifest, validateEvidenceRegistry, verifyEvidenceReceipt, verifyManifest, verifyMatchedBudgets, verifySealedExperiment, wilson };
852
+ export { BOOTSTRAP_GATE_MIN_N, DesignRefusalError, EVIDENCE_AUTHORITY_KINDS, EVIDENCE_RECEIPT_VERSION, EVIDENCE_STATES, ExperimentTracker, FinalEvidenceConflictError, FinalEvidenceError, FunnelIntegrityError, INDEPENDENT_EVIDENCE_AUTHORITY_KINDS, MatchedBudgetError, SealIntegrityError, amendExperiment, assertDesignAdequate, assertFunnelReconciles, assertMatchedBudgets, benjaminiHochberg, bonferroni, buildEvidenceVector, buildFunnel, classifyReissue, clusteredPower, comparePairedArms, composeFunnels, computeEstimand, computeInterval, createCampaignEvidenceReceipt, createEvidenceReceipt, defineEvaluationClaim, defineExperiment, eProcess, evaluateCondition, evaluateHaltRule, evaluateHypothesis, evaluateIdentityGate, evaluateOracleDeterminismGate, evaluatePopulationReproducibilityGate, evaluatePowerFloorGate, evaluatePredicate, evaluateProvenanceGate, evidenceRegistryRecordSchema, executeAdmissionRule, executeDecisionRule, fileExperimentStore, hashJson, heldoutSignificance, holm, inMemoryExperimentStore, isIndependentEvidence, manifestContentDigest, mcnemar, mcnemarPower, mcnemarRequiredN, mulberry32, openFinalEvidenceLedger, openSealedExperiment, pairArms, pairHoldout, pairRunRecords, pairedBootstrap, pairedEvalueSequence, pairedMde, pairedRiskDifference, pairedRiskDifferenceExact, pairedRiskDifferenceScore, paretoPolicy, paretoSignificanceGate, parseEvidenceRegistryRecord, powerPreflight, projectNLadderBudget, readField, renderEvidenceIndex, renderFunnelTable, requiredPairedSampleSize, requiredSampleSize, runSelectionRule, runUniformPassBudget, sealExperiment, sequentialCrossingHorizon, sequentialDecide, sequentialPairedGate, signManifest, summarizeEvaluationUnits, validateEvidenceRegistry, verifyEvidenceReceipt, verifyManifest, verifyMatchedBudgets, verifySealedExperiment, wilson };
1317
853
 
1318
854
  //# sourceMappingURL=index.js.map