@tangle-network/agent-eval 0.144.6 → 0.144.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (239) hide show
  1. package/CHANGELOG.md +24 -0
  2. package/README.md +2 -0
  3. package/dist/{benchmark-J9Qe6j2_.d.ts → agent-profile-CgDTo40f.d.ts} +91 -4
  4. package/dist/agent-profile-CgDTo40f.d.ts.map +1 -0
  5. package/dist/analyst/index.d.ts +470 -88
  6. package/dist/analyst/index.d.ts.map +1 -1
  7. package/dist/analyst/index.js +24 -5
  8. package/dist/analyst/index.js.map +1 -1
  9. package/dist/{analyze-runs-DWIvOAGk.js → analyze-runs-BNNK7irB.js} +3 -3
  10. package/dist/{analyze-runs-DWIvOAGk.js.map → analyze-runs-BNNK7irB.js.map} +1 -1
  11. package/dist/{baseline-D_fT6277.d.ts → baseline-CavEbRyH.d.ts} +2 -23
  12. package/dist/baseline-CavEbRyH.d.ts.map +1 -0
  13. package/dist/{benchmark-command-CQd78YHt.js → benchmark-command-BCafwNrf.js} +662 -605
  14. package/dist/benchmark-command-BCafwNrf.js.map +1 -0
  15. package/dist/benchmarks/index.d.ts +1 -1
  16. package/dist/benchmarks/index.js +1 -1
  17. package/dist/{benchmarks-BEOkuvIg.js → benchmarks-CDSolHq7.js} +4 -4
  18. package/dist/{benchmarks-BEOkuvIg.js.map → benchmarks-CDSolHq7.js.map} +1 -1
  19. package/dist/campaign/index.d.ts +7 -5
  20. package/dist/campaign/index.js +5 -3
  21. package/dist/{campaign-CXsdyym7.js → campaign-Tdy3h62h.js} +17 -301
  22. package/dist/campaign-Tdy3h62h.js.map +1 -0
  23. package/dist/cli.js +2 -2
  24. package/dist/{client-0JI64ovJ.d.ts → client-DjXROWpx.d.ts} +3 -3
  25. package/dist/{client-0JI64ovJ.d.ts.map → client-DjXROWpx.d.ts.map} +1 -1
  26. package/dist/{completion-verifier-CBiee74w.d.ts → completion-verifier-foUCLif_.d.ts} +5 -5
  27. package/dist/{completion-verifier-CBiee74w.d.ts.map → completion-verifier-foUCLif_.d.ts.map} +1 -1
  28. package/dist/contract/index.d.ts +9 -8
  29. package/dist/contract/index.d.ts.map +1 -1
  30. package/dist/contract/index.js +7 -6
  31. package/dist/contract/index.js.map +1 -1
  32. package/dist/control.d.ts +2 -2
  33. package/dist/control.js +1 -1
  34. package/dist/counterfactual-CWPTrMH7.js +126 -0
  35. package/dist/counterfactual-CWPTrMH7.js.map +1 -0
  36. package/dist/counterfactual-CxmxAONP.d.ts +72 -0
  37. package/dist/counterfactual-CxmxAONP.d.ts.map +1 -0
  38. package/dist/{default-registry-J9m-_tya.d.ts → default-registry-BZhStdjl.d.ts} +6 -5
  39. package/dist/default-registry-BZhStdjl.d.ts.map +1 -0
  40. package/dist/{default-registry-Dta70shL.js → default-registry-BaQXW1Ow.js} +2 -2
  41. package/dist/{default-registry-Dta70shL.js.map → default-registry-BaQXW1Ow.js.map} +1 -1
  42. package/dist/{dspy-rlm-engine-19FQEMBK.js → dspy-rlm-engine-BiN49gK6.js} +4 -3
  43. package/dist/{dspy-rlm-engine-19FQEMBK.js.map → dspy-rlm-engine-BiN49gK6.js.map} +1 -1
  44. package/dist/{tool-groups-CK0JCkqO.d.ts → engine-nB64f48I.d.ts} +18 -31
  45. package/dist/engine-nB64f48I.d.ts.map +1 -0
  46. package/dist/{eval-campaign-CfLQQs9B.js → eval-campaign-DNjCvAm-.js} +7 -6
  47. package/dist/{eval-campaign-CfLQQs9B.js.map → eval-campaign-DNjCvAm-.js.map} +1 -1
  48. package/dist/{exact-types-CBYF5MGd.d.ts → exact-types-Djvzosly.d.ts} +2 -2
  49. package/dist/{exact-types-CBYF5MGd.d.ts.map → exact-types-Djvzosly.d.ts.map} +1 -1
  50. package/dist/exec-BLtYZdWo.js +49 -0
  51. package/dist/exec-BLtYZdWo.js.map +1 -0
  52. package/dist/experiment/index.d.ts +802 -0
  53. package/dist/experiment/index.d.ts.map +1 -0
  54. package/dist/experiment/index.js +1108 -0
  55. package/dist/experiment/index.js.map +1 -0
  56. package/dist/experiment-tracker-CnRICnMl.js +500 -0
  57. package/dist/experiment-tracker-CnRICnMl.js.map +1 -0
  58. package/dist/experiment-tracker-IMntXr6J.d.ts +299 -0
  59. package/dist/experiment-tracker-IMntXr6J.d.ts.map +1 -0
  60. package/dist/{external-optimizer-contracts-Q9c0L3eY.d.ts → external-optimizer-contracts-CZuJNcT5.d.ts} +2 -2
  61. package/dist/{external-optimizer-contracts-Q9c0L3eY.d.ts.map → external-optimizer-contracts-CZuJNcT5.d.ts.map} +1 -1
  62. package/dist/{feedback-trajectory-GgoS0-MK.d.ts → feedback-trajectory-Rh280oXo.d.ts} +3 -3
  63. package/dist/{feedback-trajectory-GgoS0-MK.d.ts.map → feedback-trajectory-Rh280oXo.d.ts.map} +1 -1
  64. package/dist/fuzz.d.ts +1 -1
  65. package/dist/hosted/index.d.ts +3 -3
  66. package/dist/{index-BZUe-ODI.d.ts → index-BZ3-y4YL.d.ts} +3 -3
  67. package/dist/{index-BZUe-ODI.d.ts.map → index-BZ3-y4YL.d.ts.map} +1 -1
  68. package/dist/{index-DSC51roc.d.ts → index-C3ssXVLv.d.ts} +2 -2
  69. package/dist/{index-DSC51roc.d.ts.map → index-C3ssXVLv.d.ts.map} +1 -1
  70. package/dist/{index-4XwggC10.d.ts → index-C5HOo4ZF2.d.ts} +4 -4
  71. package/dist/index-C5HOo4ZF2.d.ts.map +1 -0
  72. package/dist/{index-B6-B0zTB.d.ts → index-CvXXlyz7.d.ts} +2 -2
  73. package/dist/{index-B6-B0zTB.d.ts.map → index-CvXXlyz7.d.ts.map} +1 -1
  74. package/dist/{index-Dx1kF3Ez.d.ts → index-CwDrUMe0.d.ts} +2 -2
  75. package/dist/{index-Dx1kF3Ez.d.ts.map → index-CwDrUMe0.d.ts.map} +1 -1
  76. package/dist/{index-BIL5vxxt.d.ts → index-Sh2I0DRc.d.ts} +11 -645
  77. package/dist/index-Sh2I0DRc.d.ts.map +1 -0
  78. package/dist/index.d.ts +214 -404
  79. package/dist/index.d.ts.map +1 -1
  80. package/dist/index.js +233 -649
  81. package/dist/index.js.map +1 -1
  82. package/dist/{insight-report-DqEsugpr.d.ts → insight-report-C6h6F_4L.d.ts} +3 -3
  83. package/dist/{insight-report-DqEsugpr.d.ts.map → insight-report-C6h6F_4L.d.ts.map} +1 -1
  84. package/dist/{integrity-CNGUaGBY.d.ts → integrity-BuqEKu-x.d.ts} +2 -2
  85. package/dist/{integrity-CNGUaGBY.d.ts.map → integrity-BuqEKu-x.d.ts.map} +1 -1
  86. package/dist/integrity-MLzHOfV9.js +141 -0
  87. package/dist/integrity-MLzHOfV9.js.map +1 -0
  88. package/dist/kind-factory-BHIgPmzS.js.map +1 -1
  89. package/dist/{llm-client-Dv5BiKLE.js → llm-client-DzvMUsS_.js} +24 -6
  90. package/dist/llm-client-DzvMUsS_.js.map +1 -0
  91. package/dist/matrix/index.d.ts +2 -2
  92. package/dist/meta-eval/index.d.ts +1 -1
  93. package/dist/{mint-DD-0oQTA.js → mint-CGEkzPLf.js} +2 -2
  94. package/dist/{mint-DD-0oQTA.js.map → mint-CGEkzPLf.js.map} +1 -1
  95. package/dist/{multi-layer-verifier-BHY1gWAc.d.ts → multi-layer-verifier-DnAqwl0h.d.ts} +2 -2
  96. package/dist/{multi-layer-verifier-BHY1gWAc.d.ts.map → multi-layer-verifier-DnAqwl0h.d.ts.map} +1 -1
  97. package/dist/multishot/index.d.ts +2 -2
  98. package/dist/openapi.json +1 -1
  99. package/dist/paired-promotion-decision-B6zJ3gYM.d.ts +114 -0
  100. package/dist/paired-promotion-decision-B6zJ3gYM.d.ts.map +1 -0
  101. package/dist/pipelines/index.d.ts +2 -1
  102. package/dist/pipelines/index.d.ts.map +1 -1
  103. package/dist/pre-registration-DakwTRXk.js +96 -0
  104. package/dist/pre-registration-DakwTRXk.js.map +1 -0
  105. package/dist/prime-bridge-transport-6feEglLf.d.ts +49 -0
  106. package/dist/prime-bridge-transport-6feEglLf.d.ts.map +1 -0
  107. package/dist/prime-protocol-BfSalTfR.js +453 -0
  108. package/dist/prime-protocol-BfSalTfR.js.map +1 -0
  109. package/dist/profile-cell.js +242 -1
  110. package/dist/profile-cell.js.map +1 -0
  111. package/dist/promotion-policy-ChWhTDBH.d.ts +289 -0
  112. package/dist/promotion-policy-ChWhTDBH.d.ts.map +1 -0
  113. package/dist/promotion-policy-CrLrmys8.js +682 -0
  114. package/dist/promotion-policy-CrLrmys8.js.map +1 -0
  115. package/dist/{propose-review-control-BciUCZoh.js → propose-review-control-qgWLA6E9.js} +2 -2
  116. package/dist/{propose-review-control-BciUCZoh.js.map → propose-review-control-qgWLA6E9.js.map} +1 -1
  117. package/dist/{release-report-ChOgpIoQ.d.ts → release-report-CI8uisI1.d.ts} +2 -2
  118. package/dist/{release-report-ChOgpIoQ.d.ts.map → release-report-CI8uisI1.d.ts.map} +1 -1
  119. package/dist/{release-report-Sl0xfkFv.js → release-report-Dy39cbFF.js} +3 -3
  120. package/dist/{release-report-Sl0xfkFv.js.map → release-report-Dy39cbFF.js.map} +1 -1
  121. package/dist/{replay-Krvb114g.d.ts → replay-DFf-teiC.d.ts} +5 -4
  122. package/dist/replay-DFf-teiC.d.ts.map +1 -0
  123. package/dist/{replay-GW61ezMW.js → replay-DyBLaKFc.js} +3 -3
  124. package/dist/{replay-GW61ezMW.js.map → replay-DyBLaKFc.js.map} +1 -1
  125. package/dist/reporting.d.ts +3 -3
  126. package/dist/reporting.js +2 -2
  127. package/dist/{researcher-xLeNcpKX.d.ts → researcher-BoaxeCzP.d.ts} +4 -4
  128. package/dist/{researcher-xLeNcpKX.d.ts.map → researcher-BoaxeCzP.d.ts.map} +1 -1
  129. package/dist/{reward-hacking-CyuzxKly.js → reward-hacking-B2tPL9a3.js} +12 -3
  130. package/dist/reward-hacking-B2tPL9a3.js.map +1 -0
  131. package/dist/{reward-hacking-RZgnGWlx.d.ts → reward-hacking-Cf1PtEOz.d.ts} +33 -3
  132. package/dist/reward-hacking-Cf1PtEOz.d.ts.map +1 -0
  133. package/dist/rl.d.ts +17 -7
  134. package/dist/rl.d.ts.map +1 -1
  135. package/dist/rl.js +16 -7
  136. package/dist/rl.js.map +1 -1
  137. package/dist/rollout/index.d.ts +2 -2
  138. package/dist/rollout/index.js +2 -2
  139. package/dist/{rollout-C2fD1cf4.js → rollout-2ECTXb2N.js} +2 -2
  140. package/dist/{rollout-C2fD1cf4.js.map → rollout-2ECTXb2N.js.map} +1 -1
  141. package/dist/{run-evidence-C6G41MSI.d.ts → run-evidence-BDFFai9R.d.ts} +2 -2
  142. package/dist/{run-evidence-C6G41MSI.d.ts.map → run-evidence-BDFFai9R.d.ts.map} +1 -1
  143. package/dist/{run-record-CWN8-VsV.js → run-record-DqOw5X6_.js} +2 -2
  144. package/dist/{run-record-CWN8-VsV.js.map → run-record-DqOw5X6_.js.map} +1 -1
  145. package/dist/{schema-Cef2cFmb.d.ts → schema-Cef2cFmb2.d.ts} +1 -1
  146. package/dist/schema-Cef2cFmb2.d.ts.map +1 -0
  147. package/dist/{semantic-concept-judge-DwF6n05O.js → semantic-concept-judge-Bmrq6yqU.js} +2 -2
  148. package/dist/{semantic-concept-judge-DwF6n05O.js.map → semantic-concept-judge-Bmrq6yqU.js.map} +1 -1
  149. package/dist/sequential-D-BLJBKU.js +299 -0
  150. package/dist/sequential-D-BLJBKU.js.map +1 -0
  151. package/dist/{server-D6XJQHw7.js → server-iu0ede49.js} +2 -2
  152. package/dist/{server-D6XJQHw7.js.map → server-iu0ede49.js.map} +1 -1
  153. package/dist/{single-run-lock-D5iN0Xzb.js → single-run-lock-BMQEv1wG.js} +7 -161
  154. package/dist/single-run-lock-BMQEv1wG.js.map +1 -0
  155. package/dist/{skill-usage-GlOphAhX.d.ts → skill-usage-CJlWEUFt.d.ts} +10 -10
  156. package/dist/{skill-usage-GlOphAhX.d.ts.map → skill-usage-CJlWEUFt.d.ts.map} +1 -1
  157. package/dist/{skillopt-optimization-method-7S43rbDB.d.ts → skillopt-optimization-method-BO7NAl3b.d.ts} +8 -294
  158. package/dist/skillopt-optimization-method-BO7NAl3b.d.ts.map +1 -0
  159. package/dist/{skillopt-optimization-method-Bfb-vBKe.js → skillopt-optimization-method-CQwZ-ZX8.js} +8 -652
  160. package/dist/skillopt-optimization-method-CQwZ-ZX8.js.map +1 -0
  161. package/dist/statistical-heldout-Dn9ruizm.d.ts +555 -0
  162. package/dist/statistical-heldout-Dn9ruizm.d.ts.map +1 -0
  163. package/dist/{statistics-C-dm-J6H.d.ts → statistics-D6Uebe_4.d.ts} +2 -2
  164. package/dist/{statistics-C-dm-J6H.d.ts.map → statistics-D6Uebe_4.d.ts.map} +1 -1
  165. package/dist/steps-BArUxhna.d.ts +51 -0
  166. package/dist/steps-BArUxhna.d.ts.map +1 -0
  167. package/dist/{integrity-fdt8XPAv.js → store-DNe_Uv1Q.js} +2 -140
  168. package/dist/store-DNe_Uv1Q.js.map +1 -0
  169. package/dist/{summary-report-B0cAyA7N.d.ts → summary-report-DuUS_i7W.d.ts} +3 -114
  170. package/dist/summary-report-DuUS_i7W.d.ts.map +1 -0
  171. package/dist/{summary-report-9A5y7EsK.js → summary-report-Lf-5I7xh.js} +2 -2
  172. package/dist/{summary-report-9A5y7EsK.js.map → summary-report-Lf-5I7xh.js.map} +1 -1
  173. package/dist/supervisor-run/index.d.ts +2 -2
  174. package/dist/supervisor-run/index.js +1 -1
  175. package/dist/{supervisor-run-DiyQVczd.js → supervisor-run-D_sokXcO.js} +32 -8
  176. package/dist/{supervisor-run-DiyQVczd.js.map → supervisor-run-D_sokXcO.js.map} +1 -1
  177. package/dist/tool-groups-DIVBnJyl.d.ts +28 -0
  178. package/dist/tool-groups-DIVBnJyl.d.ts.map +1 -0
  179. package/dist/trace-repair/index.d.ts +2102 -0
  180. package/dist/trace-repair/index.d.ts.map +1 -0
  181. package/dist/trace-repair/index.js +3878 -0
  182. package/dist/trace-repair/index.js.map +1 -0
  183. package/dist/traces.d.ts +6 -6
  184. package/dist/traces.js +3 -2
  185. package/dist/trajectory-YC15QDYQ.d.ts +24 -0
  186. package/dist/trajectory-YC15QDYQ.d.ts.map +1 -0
  187. package/dist/trajectory-replay/index.d.ts +781 -0
  188. package/dist/trajectory-replay/index.d.ts.map +1 -0
  189. package/dist/trajectory-replay/index.js +2103 -0
  190. package/dist/trajectory-replay/index.js.map +1 -0
  191. package/dist/{types-XMVEdrE_.d.ts → types-D216SgwM.d.ts} +24 -6
  192. package/dist/types-D216SgwM.d.ts.map +1 -0
  193. package/dist/{types-yLK8gXE9.d.ts → types-D4mog56g.d.ts} +2 -2
  194. package/dist/{types-yLK8gXE9.d.ts.map → types-D4mog56g.d.ts.map} +1 -1
  195. package/dist/{types-BhP9q0Fq.d.ts → types-DF_Udrp-.d.ts} +52 -3
  196. package/dist/{types-BhP9q0Fq.d.ts.map → types-DF_Udrp-.d.ts.map} +1 -1
  197. package/dist/{types-DOZyvsFU.d.ts → types-DYuNHo9R.d.ts} +3 -3
  198. package/dist/{types-DOZyvsFU.d.ts.map → types-DYuNHo9R.d.ts.map} +1 -1
  199. package/dist/usage-receipt-EVI8B8Xu.js.map +1 -1
  200. package/dist/verdict-DExhxfgR.d.ts +201 -0
  201. package/dist/verdict-DExhxfgR.d.ts.map +1 -0
  202. package/dist/verdict-cache-BCcOh0kF.js +159 -0
  203. package/dist/verdict-cache-BCcOh0kF.js.map +1 -0
  204. package/dist/wire/index.d.ts +2 -2
  205. package/dist/wire/index.js +1 -1
  206. package/docs/charter.md +112 -0
  207. package/docs/experiment.md +104 -0
  208. package/docs/prime-analyst.md +1 -0
  209. package/docs/trace-analysis.md +26 -0
  210. package/docs/trace-repair-admission.md +194 -0
  211. package/docs/trace-repair-analyst-arms.md +121 -0
  212. package/docs/trace-repair-continuation.md +107 -0
  213. package/docs/trace-repair-grader.md +163 -0
  214. package/docs/trajectory-replay.md +110 -0
  215. package/docs/verification-strategies.md +103 -0
  216. package/package.json +19 -2
  217. package/dist/agent-profile-cell-CbfBm2g6.js +0 -335
  218. package/dist/agent-profile-cell-CbfBm2g6.js.map +0 -1
  219. package/dist/baseline-D_fT6277.d.ts.map +0 -1
  220. package/dist/benchmark-J9Qe6j2_.d.ts.map +0 -1
  221. package/dist/benchmark-command-CQd78YHt.js.map +0 -1
  222. package/dist/campaign-CXsdyym7.js.map +0 -1
  223. package/dist/default-registry-J9m-_tya.d.ts.map +0 -1
  224. package/dist/index-4XwggC10.d.ts.map +0 -1
  225. package/dist/index-BIL5vxxt.d.ts.map +0 -1
  226. package/dist/integrity-fdt8XPAv.js.map +0 -1
  227. package/dist/llm-client-Dv5BiKLE.js.map +0 -1
  228. package/dist/replay-Krvb114g.d.ts.map +0 -1
  229. package/dist/reward-hacking-CyuzxKly.js.map +0 -1
  230. package/dist/reward-hacking-RZgnGWlx.d.ts.map +0 -1
  231. package/dist/schema-Cef2cFmb.d.ts.map +0 -1
  232. package/dist/single-run-lock-D5iN0Xzb.js.map +0 -1
  233. package/dist/skillopt-optimization-method-7S43rbDB.d.ts.map +0 -1
  234. package/dist/skillopt-optimization-method-Bfb-vBKe.js.map +0 -1
  235. package/dist/summary-report-B0cAyA7N.d.ts.map +0 -1
  236. package/dist/tool-groups-CK0JCkqO.d.ts.map +0 -1
  237. package/dist/types-XMVEdrE_.d.ts.map +0 -1
  238. package/dist/verdict-Dps8_okt.d.ts +0 -37
  239. package/dist/verdict-Dps8_okt.d.ts.map +0 -1
@@ -0,0 +1,1108 @@
1
+ import { c as ValidationError, n as CaptureIntegrityError } from "../errors-D-LKuDhb.js";
2
+ import { a as verifyManifest, i as signManifest, n as evaluateHypothesis, r as hashJson, t as canonicalize } from "../pre-registration-DakwTRXk.js";
3
+ import { A as pairedRiskDifference, B as requiredSampleSize, C as mulberry32, E as pairedBootstrap, G as wilson, M as pairedRiskDifferenceScore, S as mcnemarRequiredN, b as mcnemar, c as bonferroni, h as holm, j as pairedRiskDifferenceExact, k as pairedMde, m as eProcess, s as benjaminiHochberg, t as BOOTSTRAP_GATE_MIN_N, x as mcnemarPower, z as requiredPairedSampleSize } from "../statistics-ByxzSiOM.js";
4
+ import { n as pairArms, r as pairRunRecords, t as comparePairedArms } from "../paired-arms-iZ08VFMN.js";
5
+ import { o as inMemoryExperimentStore, r as fileExperimentStore, s as clusteredPairedBinary, t as ExperimentTracker } from "../experiment-tracker-CnRICnMl.js";
6
+ import { c as heldoutSignificance, i as powerPreflight, l as pairHoldout, n as paretoPolicy, r as paretoSignificanceGate, t as buildEvidenceVector } from "../promotion-policy-CrLrmys8.js";
7
+ import { n as sequentialPairedGate, t as sequentialDecide } from "../sequential-D-BLJBKU.js";
8
+ import { n as pairedEvalueSequence } from "../sequential-Br0mAPHA.js";
9
+ //#region src/experiment/ast.ts
10
+ /**
11
+ * The registered-rule AST: every rule an experiment registers is DATA.
12
+ *
13
+ * A closure cannot be canonicalized or hashed; a node tree can. `sealExperiment`
14
+ * hashes the whole tree, and every interpreter in this file takes only a node
15
+ * plus evidence records — no parameter for alpha, threshold, metric, or
16
+ * stopping rule exists on any executable surface. The registered object and
17
+ * the executed object are therefore the same object, and registered-vs-ran
18
+ * drift is unrepresentable rather than checked.
19
+ *
20
+ * Node families:
21
+ * Predicate closed-key comparisons — the only leaf
22
+ * AdmissionRule monotone funnel stages with registered waivers
23
+ * SelectionRule deterministic subsets over a closed field set
24
+ * Estimand what the experiment measures
25
+ * IntervalSpec how uncertainty is computed, seed included
26
+ * Condition decision guards over named derived quantities
27
+ * DecisionRule ordered verdict table, or a registered absence of one
28
+ * Obligation a control that must exist before a verdict class is read
29
+ * ValidityGate pre-spend design checks
30
+ * HaltRule gates as prerequisites — failure refuses the spend
31
+ * BudgetRule spend schedules with a named ledger
32
+ * MatchedBudgetRule arm budget matching as a refusal
33
+ * ReissuePolicy carrier faults are reissued; model outcomes stand
34
+ */
35
+ /** A decision rule's branches did not cover the evidence. */
36
+ var DecisionTableNotTotalError = class extends ValidationError {};
37
+ /** Read a dot-separated field path. Missing segments yield `undefined`. */
38
+ function readField(record, path) {
39
+ let current = record;
40
+ for (const key of path.split(".")) {
41
+ if (current === null || typeof current !== "object") return void 0;
42
+ current = current[key];
43
+ }
44
+ return current;
45
+ }
46
+ /** Evaluate a predicate against one evidence record. */
47
+ function evaluatePredicate(predicate, record) {
48
+ switch (predicate.kind) {
49
+ case "compare": return compareValues(readField(record, predicate.field), predicate.op, predicate.value);
50
+ case "in": return predicate.values.includes(readField(record, predicate.field));
51
+ case "all": return predicate.of.every((p) => evaluatePredicate(p, record));
52
+ case "any": return predicate.of.some((p) => evaluatePredicate(p, record));
53
+ case "not": return !evaluatePredicate(predicate.of, record);
54
+ }
55
+ }
56
+ /** Numbers compare numerically; everything else compares as strings. */
57
+ function compareValues(value, op, target) {
58
+ if (op === "eq") return value === target;
59
+ if (op === "ne") return value !== target;
60
+ const numeric = typeof value === "number" && typeof target === "number";
61
+ const left = numeric ? value : String(value);
62
+ const right = numeric ? target : String(target);
63
+ switch (op) {
64
+ case "lt": return left < right;
65
+ case "lte": return left <= right;
66
+ case "gt": return left > right;
67
+ case "gte": return left >= right;
68
+ }
69
+ }
70
+ /**
71
+ * Execute a selection rule.
72
+ *
73
+ * Round-robin walks groups in lexicographic order and takes ids in
74
+ * within-group order until `take` ids are chosen. Filter-of keeps the ids of
75
+ * `bases[rule.base]` whose record satisfies the predicate, in the registered
76
+ * order. Ids absent from `records` are evaluated on their id alone (fields
77
+ * derived from the id via `idFields`), so a sealed base outlives its source
78
+ * records.
79
+ */
80
+ function runSelectionRule(rule, records, options) {
81
+ if (rule.kind === "round-robin") {
82
+ const allowed = new Set(rule.reads);
83
+ for (const field of [rule.groupBy, rule.withinOrder.field]) if (!allowed.has(field)) throw new ValidationError(`runSelectionRule: round-robin reads '${field}' but its closed read set is [${rule.reads.join(", ")}]`);
84
+ const byGroup = /* @__PURE__ */ new Map();
85
+ for (const record of records) {
86
+ const group = String(readField(record, rule.groupBy));
87
+ const id = String(readField(record, rule.withinOrder.field));
88
+ const bucket = byGroup.get(group);
89
+ if (bucket) bucket.push(id);
90
+ else byGroup.set(group, [id]);
91
+ }
92
+ for (const ids of byGroup.values()) {
93
+ ids.sort();
94
+ if (rule.withinOrder.dir === "desc") ids.reverse();
95
+ }
96
+ const groups = [...byGroup.keys()].sort();
97
+ const chosen = [];
98
+ let cursor = 0;
99
+ while (chosen.length < rule.take && groups.some((g) => byGroup.get(g).length > 0)) {
100
+ const group = groups[cursor % groups.length];
101
+ const ids = byGroup.get(group);
102
+ if (ids.length > 0) chosen.push(ids.shift());
103
+ cursor += 1;
104
+ }
105
+ return chosen;
106
+ }
107
+ const base = options.bases?.[rule.base];
108
+ if (!base) throw new ValidationError(`runSelectionRule: filter-of base '${rule.base}' was not provided`);
109
+ const index = new Map(records.map((r) => [String(readField(r, options.idField)), r]));
110
+ const sorted = [...base.filter((id) => {
111
+ const record = index.get(id) ?? options.idFields?.(id);
112
+ if (!record) throw new ValidationError(`runSelectionRule: base id '${id}' has no record and no idFields derivation`);
113
+ return evaluatePredicate(rule.keep, record);
114
+ })].sort();
115
+ if (rule.order.dir === "desc") sorted.reverse();
116
+ return sorted;
117
+ }
118
+ function evaluateSetExpr(expr, rows, armField, idField) {
119
+ if (expr.kind === "rows-where") {
120
+ const ids = /* @__PURE__ */ new Set();
121
+ for (const row of rows) {
122
+ if (String(readField(row, armField)) !== expr.arm) continue;
123
+ if (evaluatePredicate(expr.event, row)) ids.add(String(readField(row, idField)));
124
+ }
125
+ return ids;
126
+ }
127
+ if (expr.of.length === 0) throw new ValidationError("computeEstimand: empty intersect");
128
+ const [first, ...rest] = expr.of.map((e) => evaluateSetExpr(e, rows, armField, idField));
129
+ const out = /* @__PURE__ */ new Set();
130
+ for (const id of first) if (rest.every((s) => s.has(id))) out.add(id);
131
+ return out;
132
+ }
133
+ /** Compute an estimand over evidence rows. Pure; reads only registered fields. */
134
+ function computeEstimand(estimand, rows) {
135
+ switch (estimand.kind) {
136
+ case "rate": {
137
+ const numerator = rows.filter((r) => evaluatePredicate(estimand.event, r)).length;
138
+ if (rows.length === 0) throw new ValidationError("computeEstimand: rate over zero rows");
139
+ return {
140
+ value: numerator / rows.length,
141
+ numerator,
142
+ denominator: rows.length
143
+ };
144
+ }
145
+ case "rate-at-least-once": {
146
+ const byGroup = /* @__PURE__ */ new Map();
147
+ for (const row of rows) {
148
+ const group = String(readField(row, estimand.groupBy));
149
+ const hit = evaluatePredicate(estimand.event, row);
150
+ byGroup.set(group, (byGroup.get(group) ?? false) || hit);
151
+ }
152
+ if (byGroup.size === 0) throw new ValidationError("computeEstimand: rate-at-least-once over zero groups");
153
+ const numerator = [...byGroup.values()].filter(Boolean).length;
154
+ return {
155
+ value: numerator / byGroup.size,
156
+ numerator,
157
+ denominator: byGroup.size
158
+ };
159
+ }
160
+ case "paired-mean-diff": {
161
+ const byPair = /* @__PURE__ */ new Map();
162
+ for (const row of rows) {
163
+ const arm = String(readField(row, estimand.armField));
164
+ if (arm !== estimand.treatment && arm !== estimand.control) continue;
165
+ const pair = String(readField(row, estimand.pairBy));
166
+ const value = readField(row, estimand.value);
167
+ if (typeof value !== "number") throw new ValidationError(`computeEstimand: paired-mean-diff value field '${estimand.value}' is not a number on pair '${pair}'`);
168
+ const slot = byPair.get(pair) ?? {};
169
+ if (arm === estimand.treatment) slot.treatment = value;
170
+ else slot.control = value;
171
+ byPair.set(pair, slot);
172
+ }
173
+ if (byPair.size === 0) throw new ValidationError("computeEstimand: paired-mean-diff over zero pairs");
174
+ let sum = 0;
175
+ for (const slot of byPair.values()) sum += (slot.treatment ?? 0) - (slot.control ?? 0);
176
+ return {
177
+ value: sum / byPair.size,
178
+ numerator: sum,
179
+ denominator: byPair.size
180
+ };
181
+ }
182
+ case "set-ratio": {
183
+ const numeratorSet = evaluateSetExpr(estimand.numerator, rows, estimand.armField, estimand.idField);
184
+ const denominatorSet = evaluateSetExpr(estimand.denominator, rows, estimand.armField, estimand.idField);
185
+ if (denominatorSet.size === 0) throw new ValidationError("computeEstimand: set-ratio denominator set is empty");
186
+ return {
187
+ value: numeratorSet.size / denominatorSet.size,
188
+ numerator: numeratorSet.size,
189
+ denominator: denominatorSet.size
190
+ };
191
+ }
192
+ }
193
+ }
194
+ /**
195
+ * Execute an interval spec.
196
+ *
197
+ * Cluster-bootstrap resamples whole clusters of the per-row `value` field and
198
+ * takes percentile bounds of the pooled mean. Clopper-Pearson computes the
199
+ * exact binomial interval and requires `successes`/`trials` evidence instead
200
+ * of rows.
201
+ */
202
+ function computeInterval(spec, evidence) {
203
+ if (spec.kind === "cluster-bootstrap") {
204
+ if (evidence.kind !== "rows") throw new ValidationError("computeInterval: cluster-bootstrap requires row evidence");
205
+ const clusters = /* @__PURE__ */ new Map();
206
+ for (const row of evidence.rows) {
207
+ const cluster = String(readField(row, spec.clusterBy));
208
+ const value = readField(row, evidence.value);
209
+ if (typeof value !== "number") throw new ValidationError(`computeInterval: value field '${evidence.value}' is not a number in cluster '${cluster}'`);
210
+ const bucket = clusters.get(cluster);
211
+ if (bucket) bucket.push(value);
212
+ else clusters.set(cluster, [value]);
213
+ }
214
+ const clusterValues = [...clusters.entries()].sort(([a], [b]) => a < b ? -1 : a > b ? 1 : 0).map(([, values]) => values);
215
+ if (clusterValues.length < 2) throw new ValidationError(`computeInterval: cluster-bootstrap needs >= 2 clusters, got ${clusterValues.length}`);
216
+ const rng = mulberry32(spec.seed);
217
+ const means = new Array(spec.resamples);
218
+ for (let draw = 0; draw < spec.resamples; draw++) {
219
+ let sum = 0;
220
+ let count = 0;
221
+ for (let pick = 0; pick < clusterValues.length; pick++) {
222
+ const cluster = clusterValues[Math.floor(rng() * clusterValues.length)];
223
+ for (const value of cluster) sum += value;
224
+ count += cluster.length;
225
+ }
226
+ means[draw] = sum / count;
227
+ }
228
+ means.sort((a, b) => a - b);
229
+ const alpha = 1 - spec.level;
230
+ const lowerIndex = Math.floor(alpha / 2 * spec.resamples);
231
+ const upperIndex = Math.min(spec.resamples - 1, Math.ceil((1 - alpha / 2) * spec.resamples) - 1);
232
+ return {
233
+ lower: means[lowerIndex],
234
+ upper: means[Math.max(lowerIndex, upperIndex)],
235
+ level: spec.level
236
+ };
237
+ }
238
+ if (evidence.kind !== "binomial") throw new ValidationError("computeInterval: clopper-pearson requires binomial evidence");
239
+ const { successes, trials } = evidence;
240
+ if (!Number.isInteger(successes) || !Number.isInteger(trials) || trials <= 0 || successes < 0) throw new ValidationError(`computeInterval: clopper-pearson needs 0 <= successes <= trials, got ${successes}/${trials}`);
241
+ if (successes > trials) throw new ValidationError(`computeInterval: clopper-pearson successes ${successes} exceed trials ${trials}`);
242
+ const alpha = 1 - spec.level;
243
+ return {
244
+ lower: successes === 0 ? 0 : binomialQuantile(successes, trials, alpha / 2, "lower"),
245
+ upper: successes === trials ? 1 : binomialQuantile(successes, trials, alpha / 2, "upper"),
246
+ level: spec.level
247
+ };
248
+ }
249
+ /**
250
+ * Clopper-Pearson bound by bisection on the binomial tail. The lower bound is
251
+ * the p with P(X >= successes | p) = alpha; the upper is the p with
252
+ * P(X <= successes | p) = alpha. Deterministic, no special functions.
253
+ */
254
+ function binomialQuantile(successes, trials, alpha, side) {
255
+ const tail = (p) => {
256
+ let sum = 0;
257
+ for (let k = 0; k <= trials; k++) {
258
+ if (!(side === "lower" ? k >= successes : k <= successes)) continue;
259
+ sum += Math.exp(logBinomialPmf(k, trials, p));
260
+ }
261
+ return sum;
262
+ };
263
+ let lo = 0;
264
+ let hi = 1;
265
+ for (let iter = 0; iter < 100; iter++) {
266
+ const mid = (lo + hi) / 2;
267
+ if (tail(mid) < alpha) if (side === "lower") lo = mid;
268
+ else hi = mid;
269
+ else if (side === "lower") hi = mid;
270
+ else lo = mid;
271
+ }
272
+ return (lo + hi) / 2;
273
+ }
274
+ function logBinomialPmf(k, n, p) {
275
+ if (p <= 0) return k === 0 ? 0 : Number.NEGATIVE_INFINITY;
276
+ if (p >= 1) return k === n ? 0 : Number.NEGATIVE_INFINITY;
277
+ return logChoose(n, k) + k * Math.log(p) + (n - k) * Math.log(1 - p);
278
+ }
279
+ function logChoose(n, k) {
280
+ return logFactorial(n) - logFactorial(k) - logFactorial(n - k);
281
+ }
282
+ const LOG_FACTORIAL_CACHE = [0];
283
+ function logFactorial(n) {
284
+ for (let i = LOG_FACTORIAL_CACHE.length; i <= n; i++) LOG_FACTORIAL_CACHE[i] = LOG_FACTORIAL_CACHE[i - 1] + Math.log(i);
285
+ return LOG_FACTORIAL_CACHE[n];
286
+ }
287
+ function evaluateCondition(condition, evidence) {
288
+ switch (condition.kind) {
289
+ case "interval-excludes-zero": {
290
+ const interval = evidence.intervals[condition.interval];
291
+ if (!interval) throw new ValidationError(`evaluateCondition: interval '${condition.interval}' is not in the evidence`);
292
+ return (interval.lower > 0 || interval.upper < 0) && (condition.sign === "positive" ? interval.lower > 0 : interval.upper < 0);
293
+ }
294
+ case "interval-includes-zero": {
295
+ const interval = evidence.intervals[condition.interval];
296
+ if (!interval) throw new ValidationError(`evaluateCondition: interval '${condition.interval}' is not in the evidence`);
297
+ return interval.lower <= 0 && interval.upper >= 0;
298
+ }
299
+ case "quantity-threshold": {
300
+ const value = evidence.quantities[condition.quantity];
301
+ if (value === void 0) throw new ValidationError(`evaluateCondition: quantity '${condition.quantity}' is not in the evidence`);
302
+ return compareValues(value, condition.op, condition.value);
303
+ }
304
+ case "obligation-met": return evidence.obligationsMet[condition.obligation] === true;
305
+ case "all": return condition.of.every((c) => evaluateCondition(c, evidence));
306
+ case "any": return condition.of.some((c) => evaluateCondition(c, evidence));
307
+ case "not": return !evaluateCondition(condition.of, evidence);
308
+ }
309
+ }
310
+ function executeDecisionRule(rule, evidence) {
311
+ if (rule.kind === "report-only") return {
312
+ verdict: "report-only",
313
+ report: [...rule.estimands, ...rule.intervals]
314
+ };
315
+ for (const branch of rule.branches) if (evaluateCondition(branch.when, evidence)) return {
316
+ verdict: branch.verdict,
317
+ report: branch.report
318
+ };
319
+ throw new DecisionTableNotTotalError("executeDecisionRule: decision table is not total — no branch matched the evidence");
320
+ }
321
+ /**
322
+ * Replicate-flip counting over graded states. A state whose replicates split
323
+ * between pass and fail is flipping; its flip rate is the minority share.
324
+ */
325
+ function evaluateOracleDeterminismGate(id, gate, repsByState) {
326
+ const evidence = {};
327
+ let passed = true;
328
+ for (const [state, reps] of Object.entries(repsByState)) {
329
+ const passes = reps.filter(Boolean).length;
330
+ const flipRate = reps.length === 0 ? 0 : Math.min(passes, reps.length - passes) / reps.length;
331
+ evidence[state] = {
332
+ passes,
333
+ replicates: reps.length,
334
+ flipRate
335
+ };
336
+ if (flipRate > gate.maxFlipRate) passed = false;
337
+ }
338
+ return {
339
+ id,
340
+ passed,
341
+ evidence
342
+ };
343
+ }
344
+ /**
345
+ * Join two population snapshots on `joinOn` and compare the registered fields.
346
+ * Only rows present in both snapshots are compared; a presence change is a
347
+ * different failure and needs its own gate.
348
+ */
349
+ function evaluatePopulationReproducibilityGate(id, gate, populations) {
350
+ const rightByKey = new Map(populations.right.map((r) => [String(readField(r, gate.joinOn)), r]));
351
+ const changed = [];
352
+ for (const left of populations.left) {
353
+ const key = String(readField(left, gate.joinOn));
354
+ const right = rightByKey.get(key);
355
+ if (!right) continue;
356
+ const moved = gate.compare.filter((f) => readField(left, f) !== readField(right, f));
357
+ if (moved.length > 0) changed.push(`${key} ${moved.map((f) => `${f}:${String(readField(left, f))}->${String(readField(right, f))}`).join(" ")}`);
358
+ }
359
+ return {
360
+ id,
361
+ passed: changed.length <= gate.maxChangedRows,
362
+ evidence: changed
363
+ };
364
+ }
365
+ /** The registered claim about provenance must hold on the provenance record. */
366
+ function evaluateProvenanceGate(id, gate, provenance) {
367
+ const passed = evaluatePredicate(gate.claim, provenance);
368
+ return {
369
+ id,
370
+ passed,
371
+ evidence: { claimHolds: passed }
372
+ };
373
+ }
374
+ /** Final path segment equality between the pinned and the served identity. */
375
+ function evaluateIdentityGate(id, _gate, identities) {
376
+ const basename = (s) => s.split("/").pop() ?? s;
377
+ const passed = basename(identities.pinned) === basename(identities.served);
378
+ return {
379
+ id,
380
+ passed,
381
+ evidence: {
382
+ pinned: identities.pinned,
383
+ served: identities.served,
384
+ matched: passed
385
+ }
386
+ };
387
+ }
388
+ /**
389
+ * The design's power curve must reach the registered target at some grid
390
+ * effect. The curve must cover the registered effect grid exactly — a curve
391
+ * computed on a different grid is different evidence and is refused.
392
+ */
393
+ function evaluatePowerFloorGate(id, gate, curve) {
394
+ const byEffect = new Map(curve.map((point) => [point.effect, point.power]));
395
+ const missing = gate.effectGrid.filter((effect) => !byEffect.has(effect));
396
+ if (missing.length > 0) throw new ValidationError(`evaluatePowerFloorGate: curve does not cover registered effects [${missing.join(", ")}]`);
397
+ const powers = gate.effectGrid.map((effect) => byEffect.get(effect));
398
+ const maxPower = Math.max(...powers);
399
+ return {
400
+ id,
401
+ passed: maxPower >= gate.target,
402
+ evidence: {
403
+ target: gate.target,
404
+ maxPower,
405
+ curve: gate.effectGrid.map((effect) => ({
406
+ effect,
407
+ power: byEffect.get(effect)
408
+ }))
409
+ }
410
+ };
411
+ }
412
+ function evaluateHaltRule(halt, gates) {
413
+ const seen = new Map(gates.map((g) => [g.id, g]));
414
+ const missing = halt.when.gates.filter((id) => !seen.has(id));
415
+ if (missing.length > 0) throw new ValidationError(`evaluateHaltRule: halt references gates that were not evaluated: [${missing.join(", ")}]`);
416
+ const failed = halt.when.gates.filter((id) => !seen.get(id).passed);
417
+ return failed.length > 0 ? {
418
+ fired: true,
419
+ action: halt.action,
420
+ failedGates: failed
421
+ } : {
422
+ fired: false,
423
+ action: null,
424
+ failedGates: []
425
+ };
426
+ }
427
+ /**
428
+ * Execute the uniform-pass schedule against measured pass costs. Pass 1 always
429
+ * runs; each later pass runs only when the cumulative spend plus the last
430
+ * measured pass cost stays at or under the registered ceiling. The registered
431
+ * ledger is the pre-spend the ceiling counts.
432
+ */
433
+ function runUniformPassBudget(rule, measuredPassCosts) {
434
+ let cumulative = rule.ledger.reduce((sum, entry) => sum + entry.usd, 0);
435
+ const decisions = [];
436
+ let uniformN = 0;
437
+ for (let pass = 1; pass <= rule.maxPasses; pass++) {
438
+ if (pass === 1) {
439
+ if (measuredPassCosts[0] === void 0) break;
440
+ cumulative += measuredPassCosts[0];
441
+ uniformN = 1;
442
+ continue;
443
+ }
444
+ const projected = measuredPassCosts[pass - 2];
445
+ if (projected === void 0) break;
446
+ const go = cumulative + projected <= rule.ceilingUsd;
447
+ decisions.push({
448
+ pass,
449
+ cumulativeBefore: cumulative,
450
+ projected,
451
+ go
452
+ });
453
+ if (!go || measuredPassCosts[pass - 1] === void 0) break;
454
+ cumulative += measuredPassCosts[pass - 1];
455
+ uniformN = pass;
456
+ }
457
+ return {
458
+ decisions,
459
+ uniformN
460
+ };
461
+ }
462
+ /**
463
+ * Walk the registered n-ladder and pick the first affordable step. When no
464
+ * step fits the ceiling, the rule refuses and reports the projection instead
465
+ * of shrinking the row set — "never subset rows" is the registered invariant.
466
+ */
467
+ function projectNLadderBudget(rule, measured) {
468
+ const projections = rule.steps.map((n) => {
469
+ const projectedUsd = measured.unitCostUsd * measured.rows * n;
470
+ return {
471
+ n,
472
+ projectedUsd,
473
+ affordable: projectedUsd <= rule.ceilingUsd
474
+ };
475
+ });
476
+ const first = projections.find((p) => p.affordable);
477
+ if (first) return {
478
+ chosenN: first.n,
479
+ projections,
480
+ refusal: null
481
+ };
482
+ return {
483
+ chosenN: null,
484
+ projections,
485
+ refusal: {
486
+ onExhaust: rule.onExhaust,
487
+ reason: `no ladder step fits the ${rule.ceilingUsd} USD ceiling at ${measured.rows} rows x ${measured.unitCostUsd} USD per unit`
488
+ }
489
+ };
490
+ }
491
+ /**
492
+ * Classify one rollout event under the registered reissue policy. A carrier
493
+ * event within the issue budget is reissued; a model outcome always stands;
494
+ * a carrier event past `maxIssues` is exhausted and reported, never retried.
495
+ */
496
+ function classifyReissue(policy, event, issuesSoFar) {
497
+ if (!policy.carrierEvents.includes(event)) return "stands";
498
+ return issuesSoFar < policy.maxIssues ? "reissue" : "exhausted";
499
+ }
500
+ //#endregion
501
+ //#region src/experiment/budget.ts
502
+ /**
503
+ * Matched-budget verification between arms, as a refusal object.
504
+ *
505
+ * A paired contrast is only meaningful when both arms spent comparable
506
+ * resources; "realized prompt and completion tokens must agree within 5%" is
507
+ * a registered rule, so its verification returns a verdict artifact — the
508
+ * refusal lives inside the result, never in prose beside it.
509
+ */
510
+ /** Arms whose realized budgets diverge past the registered tolerance. */
511
+ var MatchedBudgetError = class extends CaptureIntegrityError {};
512
+ /**
513
+ * Compare realized per-arm spend under the registered tolerance. Requires at
514
+ * least two arms — a single arm has nothing to match against. Negative or
515
+ * non-finite token counts are refused as evidence corruption, not compared.
516
+ */
517
+ function verifyMatchedBudgets(rule, arms) {
518
+ if (arms.length < 2) throw new MatchedBudgetError(`verifyMatchedBudgets: need >= 2 arms to match budgets, got ${arms.length}`);
519
+ for (const arm of arms) if (!Number.isFinite(arm.realizedTokens) || arm.realizedTokens < 0) throw new MatchedBudgetError(`verifyMatchedBudgets: arm '${arm.armId}' has invalid realized tokens ${arm.realizedTokens}`);
520
+ const sorted = [...arms].sort((a, b) => a.realizedTokens - b.realizedTokens);
521
+ const min = sorted[0];
522
+ const max = sorted[sorted.length - 1];
523
+ const maxRelativeGap = max.realizedTokens === 0 ? 0 : (max.realizedTokens - min.realizedTokens) / max.realizedTokens;
524
+ const matched = maxRelativeGap <= rule.tolerance;
525
+ return {
526
+ rule,
527
+ arms: [...arms],
528
+ maxRelativeGap,
529
+ widestPair: [min.armId, max.armId],
530
+ matched,
531
+ refusal: matched ? null : {
532
+ onFail: rule.onFail,
533
+ reason: `realized ${rule.measure} diverge ${(maxRelativeGap * 100).toFixed(1)}% between '${min.armId}' (${min.realizedTokens}) and '${max.armId}' (${max.realizedTokens}) — registered tolerance is ${(rule.tolerance * 100).toFixed(1)}%; the contrast is refused`
534
+ }
535
+ };
536
+ }
537
+ /** Throw the refusal for callers that gate the contrast on it. */
538
+ function assertMatchedBudgets(rule, arms) {
539
+ const verdict = verifyMatchedBudgets(rule, arms);
540
+ if (verdict.refusal) throw new MatchedBudgetError(verdict.refusal.reason);
541
+ return verdict;
542
+ }
543
+ //#endregion
544
+ //#region src/experiment/funnel.ts
545
+ /**
546
+ * The denominator chain as a first-class object.
547
+ *
548
+ * A benchmark whose denominator is not auditable is not a benchmark. Every
549
+ * stage names what entered, what it removed, and what survived, so
550
+ * `input = surviving + sum(excluded)` reads off the table instead of being
551
+ * trusted. A stage that gains rows is refused at construction — a funnel is
552
+ * monotone by definition, and a non-monotone one is a broken denominator,
553
+ * not a formatting choice.
554
+ *
555
+ * The object is its own JSON render; `renderFunnelTable` is the text render.
556
+ * `executeAdmissionRule` produces one by running a sealed {@link AdmissionRule}
557
+ * over evidence records.
558
+ */
559
+ /** A funnel stage gained rows, double-counted them, or failed to reconcile. */
560
+ var FunnelIntegrityError = class extends CaptureIntegrityError {};
561
+ /**
562
+ * Build a funnel from counts and refuse anything non-monotone.
563
+ *
564
+ * Refusals: a negative count, a stage that gains rows (excluded < 0 is the
565
+ * only way to gain — `remaining = entering - excluded` by construction, so a
566
+ * gain cannot be smuggled in through `remaining`), named exclusions that do
567
+ * not sum to the stage total, and a partition drawing from an unknown stage
568
+ * or exceeding what that stage excluded.
569
+ */
570
+ function buildFunnel(input) {
571
+ if (!Number.isInteger(input.input) || input.input < 0) throw new FunnelIntegrityError(`funnel input must be a non-negative integer, got ${input.input}`);
572
+ let entering = input.input;
573
+ const stages = [];
574
+ for (const stage of input.stages) {
575
+ if (!Number.isInteger(stage.excluded)) throw new FunnelIntegrityError(`funnel stage '${stage.id}' excluded count must be an integer, got ${stage.excluded}`);
576
+ if (stage.excluded < 0) throw new FunnelIntegrityError(`funnel stage '${stage.id}' gains ${-stage.excluded} rows — a funnel stage can only remove rows`);
577
+ if (stage.excluded > entering) throw new FunnelIntegrityError(`funnel stage '${stage.id}' excludes ${stage.excluded} rows but only ${entering} entered`);
578
+ if (stage.exclusions) {
579
+ const sum = Object.values(stage.exclusions).reduce((a, b) => a + b, 0);
580
+ if (sum !== stage.excluded) throw new FunnelIntegrityError(`funnel stage '${stage.id}' names exclusions summing to ${sum} but excludes ${stage.excluded}`);
581
+ }
582
+ const remaining = entering - stage.excluded;
583
+ stages.push({
584
+ id: stage.id,
585
+ entering,
586
+ excluded: stage.excluded,
587
+ remaining,
588
+ ...stage.exclusions ? { exclusions: { ...stage.exclusions } } : {},
589
+ ...stage.waives ? { waives: [...stage.waives] } : {}
590
+ });
591
+ entering = remaining;
592
+ }
593
+ const byStage = new Map(stages.map((s) => [s.id, s]));
594
+ const partitions = (input.partitions ?? []).map((partition) => {
595
+ const source = byStage.get(partition.from);
596
+ if (!source) throw new FunnelIntegrityError(`funnel partition '${partition.id}' draws from unknown stage '${partition.from}'`);
597
+ if (partition.count < 0 || partition.count > source.excluded) throw new FunnelIntegrityError(`funnel partition '${partition.id}' counts ${partition.count} rows but stage '${partition.from}' excluded ${source.excluded}`);
598
+ return {
599
+ id: partition.id,
600
+ from: partition.from,
601
+ count: partition.count,
602
+ pooling: "never"
603
+ };
604
+ });
605
+ const funnel = {
606
+ population: input.population,
607
+ input: input.input,
608
+ stages,
609
+ surviving: entering,
610
+ partitions
611
+ };
612
+ assertFunnelReconciles(funnel);
613
+ return funnel;
614
+ }
615
+ /** A chain that does not add up is a broken denominator, so this throws. */
616
+ function assertFunnelReconciles(funnel) {
617
+ const excluded = funnel.stages.reduce((sum, stage) => sum + stage.excluded, 0);
618
+ if (funnel.input !== funnel.surviving + excluded) throw new FunnelIntegrityError(`funnel '${funnel.population}' does not reconcile: input ${funnel.input} != surviving ${funnel.surviving} + excluded ${excluded}`);
619
+ let entering = funnel.input;
620
+ for (const stage of funnel.stages) {
621
+ if (stage.entering !== entering || stage.remaining !== stage.entering - stage.excluded) throw new FunnelIntegrityError(`funnel '${funnel.population}' stage '${stage.id}' does not chain: entering ${stage.entering} (expected ${entering}), remaining ${stage.remaining}`);
622
+ if (stage.remaining > stage.entering) throw new FunnelIntegrityError(`funnel '${funnel.population}' stage '${stage.id}' gains rows: ${stage.entering} -> ${stage.remaining}`);
623
+ entering = stage.remaining;
624
+ }
625
+ }
626
+ /**
627
+ * Run a sealed admission rule over evidence records. Each stage keeps the rows
628
+ * its predicate accepts; partitions draw from the rows their source stage
629
+ * dropped. The result embeds the funnel, so the denominator chain and the
630
+ * surviving rows can never disagree.
631
+ */
632
+ function executeAdmissionRule(rule, records) {
633
+ let current = [...records];
634
+ const droppedAt = /* @__PURE__ */ new Map();
635
+ const stageInputs = [];
636
+ for (const stage of rule.stages) {
637
+ const kept = [];
638
+ const dropped = [];
639
+ for (const record of current) if (evaluatePredicate(stage.keep, record)) kept.push(record);
640
+ else dropped.push(record);
641
+ droppedAt.set(stage.id, dropped);
642
+ stageInputs.push({
643
+ id: stage.id,
644
+ excluded: dropped.length,
645
+ ...stage.waives ? { waives: stage.waives } : {}
646
+ });
647
+ current = kept;
648
+ }
649
+ const partitionRows = {};
650
+ const partitionCounts = [];
651
+ for (const partition of rule.partitions ?? []) {
652
+ const source = droppedAt.get(partition.from);
653
+ if (!source) throw new FunnelIntegrityError(`admission partition '${partition.id}' draws from unknown stage '${partition.from}'`);
654
+ const rows = source.filter((record) => evaluatePredicate(partition.keep, record));
655
+ partitionRows[partition.id] = rows;
656
+ partitionCounts.push({
657
+ id: partition.id,
658
+ from: partition.from,
659
+ count: rows.length
660
+ });
661
+ }
662
+ return {
663
+ funnel: buildFunnel({
664
+ population: rule.population,
665
+ input: records.length,
666
+ stages: stageInputs,
667
+ partitions: partitionCounts
668
+ }),
669
+ survivors: current,
670
+ partitionRows
671
+ };
672
+ }
673
+ /**
674
+ * Chain two funnels whose boundary agrees: the second funnel's input must be
675
+ * exactly the first funnel's survivors. Anything else is a gap or an
676
+ * injection, and both are refused.
677
+ */
678
+ function composeFunnels(first, second) {
679
+ if (second.input !== first.surviving) throw new FunnelIntegrityError(`cannot compose funnels: '${first.population}' survives ${first.surviving} rows but '${second.population}' starts from ${second.input}`);
680
+ return buildFunnel({
681
+ population: first.population,
682
+ input: first.input,
683
+ stages: [...first.stages, ...second.stages].map((stage) => ({
684
+ id: stage.id,
685
+ excluded: stage.excluded,
686
+ ...stage.exclusions ? { exclusions: stage.exclusions } : {},
687
+ ...stage.waives ? { waives: stage.waives } : {}
688
+ })),
689
+ partitions: [...first.partitions, ...second.partitions].map((partition) => ({
690
+ id: partition.id,
691
+ from: partition.from,
692
+ count: partition.count
693
+ }))
694
+ });
695
+ }
696
+ /**
697
+ * Text render of the chain. One row per stage; the reconciliation line at the
698
+ * bottom restates `input = surviving + excluded` so a reader can check the
699
+ * arithmetic without a tool.
700
+ */
701
+ function renderFunnelTable(funnel) {
702
+ const rows = funnel.stages.map((stage) => [
703
+ stage.id + (stage.waives?.length ? ` (waives: ${stage.waives.join(", ")})` : ""),
704
+ String(stage.entering),
705
+ String(stage.excluded),
706
+ String(stage.remaining)
707
+ ]);
708
+ const header = [
709
+ "stage",
710
+ "entering",
711
+ "excluded",
712
+ "remaining"
713
+ ];
714
+ const widths = header.map((h, col) => Math.max(h.length, ...rows.map((r) => r[col].length)));
715
+ const line = (cells) => cells.map((cell, col) => cell.padEnd(widths[col])).join(" ");
716
+ const out = [
717
+ `population: ${funnel.population}`,
718
+ `input: ${funnel.input}`,
719
+ line(header),
720
+ line(widths.map((w) => "-".repeat(w))),
721
+ ...rows.map((r) => line(r))
722
+ ];
723
+ const excluded = funnel.stages.reduce((sum, stage) => sum + stage.excluded, 0);
724
+ out.push(`surviving: ${funnel.surviving} (input ${funnel.input} = surviving ${funnel.surviving} + excluded ${excluded})`);
725
+ for (const partition of funnel.partitions) out.push(`partition ${partition.id}: ${partition.count} rows from '${partition.from}' — reported separately, never pooled`);
726
+ return out.join("\n");
727
+ }
728
+ //#endregion
729
+ //#region src/experiment/define.ts
730
+ /**
731
+ * Define, seal, and execute experiments whose every rule is registered data.
732
+ *
733
+ * `defineExperiment` validates the cross-references inside a spec and freezes
734
+ * it. `sealExperiment` canonicalizes and hashes the whole tree — arms, row
735
+ * admission, selection, estimands, intervals, decision rule, gates, halt,
736
+ * budget — into one digest that lands on every downstream artifact. Changing
737
+ * what is decided requires a new digest: an amendment is a re-seal with a
738
+ * reason and blindness attestations, and the digest history is the audit
739
+ * trail.
740
+ *
741
+ * `openSealedExperiment` is the only execution surface. It verifies the seal
742
+ * and returns executors bound to the sealed spec; none of them takes a
743
+ * parameter for alpha, threshold, metric, or stopping rule, so
744
+ * registered-vs-ran drift is unrepresentable rather than checked.
745
+ */
746
+ /** A sealed experiment whose digest no longer matches its spec. */
747
+ var SealIntegrityError = class extends ValidationError {};
748
+ function conditionRefs(condition) {
749
+ switch (condition.kind) {
750
+ case "interval-excludes-zero":
751
+ case "interval-includes-zero": return {
752
+ intervals: [condition.interval],
753
+ quantities: [],
754
+ obligations: []
755
+ };
756
+ case "quantity-threshold": return {
757
+ intervals: [],
758
+ quantities: [condition.quantity],
759
+ obligations: []
760
+ };
761
+ case "obligation-met": return {
762
+ intervals: [],
763
+ quantities: [],
764
+ obligations: [condition.obligation]
765
+ };
766
+ case "all":
767
+ case "any": {
768
+ const nested = condition.of.map(conditionRefs);
769
+ return {
770
+ intervals: nested.flatMap((r) => r.intervals),
771
+ quantities: nested.flatMap((r) => r.quantities),
772
+ obligations: nested.flatMap((r) => r.obligations)
773
+ };
774
+ }
775
+ case "not": return conditionRefs(condition.of);
776
+ }
777
+ }
778
+ /**
779
+ * Validate every cross-reference inside a spec and freeze it.
780
+ *
781
+ * A decision condition may only read a registered interval, a registered
782
+ * estimand, or a registered obligation; a halt rule may only reference
783
+ * registered gates; a filter-of base must resolve to a registered selection
784
+ * or sealed subset. Anything else is refused here, before sealing.
785
+ */
786
+ function defineExperiment(spec) {
787
+ const problems = [];
788
+ if (!spec.id || spec.id.trim().length === 0) problems.push("id is empty");
789
+ if (spec.arms.length === 0) problems.push("at least one arm is required");
790
+ const armIds = /* @__PURE__ */ new Set();
791
+ for (const arm of spec.arms) {
792
+ if (armIds.has(arm.id)) problems.push(`duplicate arm id '${arm.id}'`);
793
+ armIds.add(arm.id);
794
+ }
795
+ if (!spec.arms.some((arm) => arm.role === "treatment")) problems.push("at least one arm must have role treatment");
796
+ const intervalNames = new Set(Object.keys(spec.intervals ?? {}));
797
+ const estimandNames = new Set(Object.keys(spec.estimands ?? {}));
798
+ const obligationIds = new Set((spec.obligations ?? []).map((o) => o.id));
799
+ const gateNames = new Set(Object.keys(spec.gates ?? {}));
800
+ const selectionNames = new Set(Object.keys(spec.selections ?? {}));
801
+ const sealedSubsetNames = new Set(Object.keys(spec.sealedSubsets ?? {}));
802
+ const checkCondition = (condition, where) => {
803
+ const refs = conditionRefs(condition);
804
+ for (const name of refs.intervals) if (!intervalNames.has(name)) problems.push(`${where} reads unregistered interval '${name}'`);
805
+ for (const name of refs.quantities) if (!estimandNames.has(name)) problems.push(`${where} reads unregistered quantity '${name}'`);
806
+ for (const name of refs.obligations) if (!obligationIds.has(name)) problems.push(`${where} reads unregistered obligation '${name}'`);
807
+ };
808
+ if (spec.decision.kind === "table") {
809
+ if (spec.decision.branches.length === 0) problems.push("decision table has no branches");
810
+ spec.decision.branches.forEach((branch, index) => {
811
+ checkCondition(branch.when, `decision branch ${index} ('${branch.verdict}')`);
812
+ });
813
+ } else {
814
+ for (const name of spec.decision.estimands) if (!estimandNames.has(name)) problems.push(`report-only decision names unregistered estimand '${name}'`);
815
+ for (const name of spec.decision.intervals) if (!intervalNames.has(name)) problems.push(`report-only decision names unregistered interval '${name}'`);
816
+ }
817
+ if (spec.decision.kind === "table" && spec.obligations) {
818
+ const verdicts = new Set(spec.decision.branches.map((b) => b.verdict));
819
+ for (const obligation of spec.obligations) for (const verdict of obligation.appliesToVerdicts) if (!verdicts.has(verdict)) problems.push(`obligation '${obligation.id}' applies to verdict '${verdict}' which no branch produces`);
820
+ }
821
+ if (spec.halt) {
822
+ for (const gate of spec.halt.when.gates) if (!gateNames.has(gate)) problems.push(`halt rule references unregistered gate '${gate}'`);
823
+ }
824
+ for (const [name, selection] of Object.entries(spec.selections ?? {})) if (selection.kind === "filter-of") {
825
+ if (!selectionNames.has(selection.base) && !sealedSubsetNames.has(selection.base)) problems.push(`selection '${name}' filters unregistered base '${selection.base}' (not a selection or sealed subset)`);
826
+ }
827
+ if (spec.admission) {
828
+ const stageIds = /* @__PURE__ */ new Set();
829
+ for (const stage of spec.admission.stages) {
830
+ if (stageIds.has(stage.id)) problems.push(`duplicate admission stage id '${stage.id}'`);
831
+ stageIds.add(stage.id);
832
+ }
833
+ for (const partition of spec.admission.partitions ?? []) if (!stageIds.has(partition.from)) problems.push(`admission partition '${partition.id}' draws from unknown stage '${partition.from}'`);
834
+ }
835
+ if (problems.length > 0) throw new ValidationError(`defineExperiment('${spec.id}'): ${problems.join("; ")}`);
836
+ return deepFreeze(structuredClone(spec));
837
+ }
838
+ function deepFreeze(value) {
839
+ if (value !== null && typeof value === "object") {
840
+ for (const key of Object.keys(value)) deepFreeze(value[key]);
841
+ Object.freeze(value);
842
+ }
843
+ return value;
844
+ }
845
+ /** Validate, canonicalize, and hash a spec into its registration. */
846
+ async function sealExperiment(spec, options = {}) {
847
+ const validated = defineExperiment(spec);
848
+ const digest = await hashJson(validated);
849
+ return {
850
+ spec: validated,
851
+ digest,
852
+ algo: "sha256-content",
853
+ sealedAt: options.sealedAt ?? (/* @__PURE__ */ new Date()).toISOString(),
854
+ initialDigest: digest,
855
+ amendments: []
856
+ };
857
+ }
858
+ /**
859
+ * Amend a sealed experiment. The current seal is verified first, the new spec
860
+ * is validated and re-hashed, and the amendment appends to the digest chain.
861
+ * There is no way to change what is decided without producing a new digest.
862
+ */
863
+ async function amendExperiment(sealed, amendment) {
864
+ await assertSealIntact(sealed);
865
+ const validated = defineExperiment(amendment.spec);
866
+ const digest = await hashJson(validated);
867
+ return {
868
+ spec: validated,
869
+ digest,
870
+ algo: "sha256-content",
871
+ sealedAt: sealed.sealedAt,
872
+ initialDigest: sealed.initialDigest,
873
+ amendments: [...sealed.amendments, {
874
+ at: amendment.at ?? (/* @__PURE__ */ new Date()).toISOString(),
875
+ reason: amendment.reason,
876
+ blind: [...amendment.blind],
877
+ digest
878
+ }]
879
+ };
880
+ }
881
+ /** True when the sealed digest still matches the spec it carries. */
882
+ async function verifySealedExperiment(sealed) {
883
+ return await hashJson(sealed.spec) === sealed.digest;
884
+ }
885
+ async function assertSealIntact(sealed) {
886
+ if (!await verifySealedExperiment(sealed)) throw new SealIntegrityError(`sealed experiment '${sealed.spec.id}' digest ${sealed.digest} does not match its spec — the registration was tampered with`);
887
+ }
888
+ /**
889
+ * Verify the seal and return executors bound to it. This is the module's only
890
+ * execution surface: a rule that is not in the sealed spec cannot run, and a
891
+ * rule that is cannot run differently.
892
+ */
893
+ async function openSealedExperiment(sealed) {
894
+ await assertSealIntact(sealed);
895
+ const spec = sealed.spec;
896
+ const need = (value, what) => {
897
+ if (value === void 0) throw new ValidationError(`experiment '${spec.id}' registered no ${what}`);
898
+ return value;
899
+ };
900
+ return {
901
+ sealed,
902
+ decide: (evidence) => executeDecisionRule(spec.decision, evidence),
903
+ admit: (records) => executeAdmissionRule(need(spec.admission, "admission rule"), records),
904
+ select: (name, records, options) => {
905
+ return runSelectionRule(need(spec.selections?.[name], `selection '${name}'`), records, {
906
+ idField: options.idField,
907
+ bases: spec.sealedSubsets
908
+ });
909
+ },
910
+ gate: (name, evidence) => {
911
+ const gate = need(spec.gates?.[name], `gate '${name}'`);
912
+ if (gate.kind !== evidence.kind) throw new ValidationError(`gate '${name}' is registered as ${gate.kind} but received ${evidence.kind} evidence`);
913
+ switch (gate.kind) {
914
+ case "oracle-determinism": return evaluateOracleDeterminismGate(name, gate, evidence.repsByState);
915
+ case "population-reproducibility": return evaluatePopulationReproducibilityGate(name, gate, evidence);
916
+ case "provenance-assertion": return evaluateProvenanceGate(name, gate, evidence.provenance);
917
+ case "identity": return evaluateIdentityGate(name, gate, evidence);
918
+ case "power-floor": return evaluatePowerFloorGate(name, gate, evidence.curve);
919
+ }
920
+ },
921
+ halt: (gates) => evaluateHaltRule(need(spec.halt, "halt rule"), gates),
922
+ runUniformPassBudget: (measuredPassCosts) => {
923
+ const budget = need(spec.budget, "budget rule");
924
+ if (budget.kind !== "uniform-pass") throw new ValidationError(`experiment '${spec.id}' registered a ${budget.kind} budget, not uniform-pass`);
925
+ return runUniformPassBudget(budget, measuredPassCosts);
926
+ },
927
+ projectNLadderBudget: (measured) => {
928
+ const budget = need(spec.budget, "budget rule");
929
+ if (budget.kind !== "n-ladder") throw new ValidationError(`experiment '${spec.id}' registered a ${budget.kind} budget, not n-ladder`);
930
+ return projectNLadderBudget(budget, measured);
931
+ },
932
+ matchedBudgets: (arms) => verifyMatchedBudgets(need(spec.matchedBudget, "matched-budget rule"), arms),
933
+ estimate: (name, rows) => computeEstimand(need(spec.estimands?.[name], `estimand '${name}'`), rows),
934
+ interval: (name, evidence) => computeInterval(need(spec.intervals?.[name], `interval '${name}'`), evidence)
935
+ };
936
+ }
937
+ //#endregion
938
+ //#region src/experiment/power.ts
939
+ /**
940
+ * Design-time power for task-clustered paired designs, with a refusal verdict.
941
+ *
942
+ * The failure this prevents (measured): a pre-registered kill test fixed a
943
+ * task-clustered bootstrap over 14 rows in 4 task clusters. Simulated at the
944
+ * registered seed, the design's power topped out at 0.69 — at a per-row
945
+ * effect of 1.0. "Four clusters cannot certify any effect size, including
946
+ * 1.0" was learned by running the experiment; this module computes it before
947
+ * a dollar is spent.
948
+ *
949
+ * Two floors, one simulation:
950
+ * - Closed form, zero spend: with C independent clusters the exact
951
+ * whole-cluster sign-flip test can never produce a two-sided p below
952
+ * 2^(1-C). C=4 gives 0.125; C=3 gives 0.25 — both above a 0.05 alpha, so
953
+ * those designs are refused at ANY effect size, before simulation.
954
+ * - Seeded simulation: per-row paired contrasts drawn under a registered
955
+ * effect model, a whole-cluster percentile bootstrap on each trial, power =
956
+ * the fraction of trials whose interval excludes zero.
957
+ *
958
+ * The refusal is a verdict INSIDE the returned artifact (the powerPreflight
959
+ * shape, made cluster-aware); `assertDesignAdequate` turns it into a throw for
960
+ * callers that want configuration-time failure.
961
+ */
962
+ /** A design refused at configuration time, before any spend. */
963
+ var DesignRefusalError = class extends ValidationError {};
964
+ /**
965
+ * Simulate the power of a whole-cluster percentile-bootstrap design and refuse
966
+ * a structure that cannot reach the target at any registered effect.
967
+ */
968
+ function clusteredPower(options) {
969
+ const clusterSizes = options.clusterSizes;
970
+ if (clusterSizes.length === 0 || clusterSizes.some((n) => !Number.isInteger(n) || n <= 0)) throw new ValidationError(`clusteredPower: clusterSizes must be positive integers, got [${clusterSizes.join(", ")}]`);
971
+ if (options.effects.length === 0) throw new ValidationError("clusteredPower: effects grid is empty");
972
+ if (!Number.isInteger(options.seed)) throw new ValidationError(`clusteredPower: seed must be an integer, got ${options.seed}`);
973
+ const trials = options.trials ?? 2e3;
974
+ const resamples = options.resamples ?? 4e3;
975
+ if (!Number.isInteger(trials) || trials <= 0 || !Number.isInteger(resamples) || resamples <= 0) throw new ValidationError(`clusteredPower: trials and resamples must be positive integers, got ${trials}/${resamples}`);
976
+ const confidence = options.confidence ?? .95;
977
+ if (confidence <= 0 || confidence >= 1) throw new ValidationError(`clusteredPower: confidence must be in (0,1), got ${confidence}`);
978
+ const alpha = options.alpha ?? .05;
979
+ const targetPower = options.targetPower ?? .8;
980
+ const baseWinRate = options.baseWinRate ?? .1;
981
+ const baseLossRate = options.baseLossRate ?? .1;
982
+ const noisy = /* @__PURE__ */ new Map();
983
+ for (const cluster of options.noisyClusters ?? []) {
984
+ if (cluster.index < 0 || cluster.index >= clusterSizes.length) throw new ValidationError(`clusteredPower: noisy cluster index ${cluster.index} outside [0,${clusterSizes.length - 1}]`);
985
+ noisy.set(cluster.index, cluster.flipRate);
986
+ }
987
+ const clusterCount = clusterSizes.length;
988
+ const totalRows = clusterSizes.reduce((a, b) => a + b, 0);
989
+ const signFlipFloor = computeSignFlipFloor(clusterCount, alpha);
990
+ const curve = [];
991
+ for (const effect of options.effects) curve.push(simulateEffect(effect, {
992
+ clusterSizes,
993
+ seed: options.seed,
994
+ trials,
995
+ resamples,
996
+ confidence,
997
+ baseWinRate,
998
+ baseLossRate,
999
+ noisy
1000
+ }));
1001
+ const maxPower = Math.max(...curve.map((point) => point.power));
1002
+ const reasons = [];
1003
+ if (!signFlipFloor.certifiableAtAlpha) reasons.push(`${clusterCount} clusters cannot certify any effect size, including 1.0: the exact whole-cluster sign-flip test's smallest two-sided p is 2^(1-${clusterCount}) = ${signFlipFloor.twoSidedP} > alpha ${alpha}; at least ${signFlipFloor.minClustersForAlpha} clusters are needed`);
1004
+ if (maxPower < targetPower) {
1005
+ const best = curve.reduce((a, b) => b.power > a.power ? b : a);
1006
+ reasons.push(`simulated power tops out at ${maxPower.toFixed(3)} (effect ${best.effect}) across the registered grid — below the ${targetPower} target at every effect`);
1007
+ }
1008
+ const adequate = reasons.length === 0;
1009
+ return {
1010
+ clusterCount,
1011
+ totalRows,
1012
+ trials,
1013
+ resamples,
1014
+ seed: options.seed,
1015
+ confidence,
1016
+ targetPower,
1017
+ curve,
1018
+ maxPower,
1019
+ signFlipFloor,
1020
+ adequate,
1021
+ refusal: adequate ? null : {
1022
+ verdict: "underpowered",
1023
+ reasons,
1024
+ recommendation: `Do not spend on this structure. Add independent clusters (>= ${Math.max(signFlipFloor.minClustersForAlpha, clusterCount)}) or register a design whose simulated power reaches ${targetPower}, then re-run clusteredPower.`
1025
+ }
1026
+ };
1027
+ }
1028
+ /** Throw the refusal for callers that want configuration-time failure. */
1029
+ function assertDesignAdequate(result) {
1030
+ if (result.refusal) throw new DesignRefusalError(`design refused (underpowered): ${result.refusal.reasons.join("; ")}`);
1031
+ }
1032
+ function computeSignFlipFloor(clusterCount, alpha) {
1033
+ const twoSidedP = 2 ** (1 - clusterCount);
1034
+ const oneSidedP = 2 ** -clusterCount;
1035
+ let minClusters = 1;
1036
+ while (2 ** (1 - minClusters) > alpha) minClusters += 1;
1037
+ return {
1038
+ twoSidedP,
1039
+ oneSidedP,
1040
+ alpha,
1041
+ certifiableAtAlpha: twoSidedP <= alpha,
1042
+ minClustersForAlpha: minClusters
1043
+ };
1044
+ }
1045
+ /**
1046
+ * One effect point. Per row the paired contrast is +1 with probability
1047
+ * min(baseWin + effect, 1), -1 with the base loss rate (capped by what
1048
+ * remains), else 0. Noisy clusters draw win and loss independently at their
1049
+ * flip rate. Each trial computes a whole-cluster percentile bootstrap of the
1050
+ * pooled row mean; the trial counts toward power when the interval excludes
1051
+ * zero.
1052
+ */
1053
+ function simulateEffect(effect, config) {
1054
+ const rng = mulberry32(mixSeed(config.seed, effect));
1055
+ const clusterCount = config.clusterSizes.length;
1056
+ let excludes = 0;
1057
+ const widths = new Array(config.trials);
1058
+ const sums = new Array(clusterCount);
1059
+ const means = new Array(config.resamples);
1060
+ for (let trial = 0; trial < config.trials; trial++) {
1061
+ for (let cluster = 0; cluster < clusterCount; cluster++) {
1062
+ const size = config.clusterSizes[cluster];
1063
+ const flipRate = config.noisy.get(cluster);
1064
+ let sum = 0;
1065
+ for (let row = 0; row < size; row++) if (flipRate !== void 0) {
1066
+ const win = rng() < flipRate ? 1 : 0;
1067
+ const loss = rng() < flipRate ? 1 : 0;
1068
+ sum += win - loss;
1069
+ } else {
1070
+ const winRate = Math.min(1, config.baseWinRate + effect);
1071
+ const lossRate = Math.min(1 - winRate, config.baseLossRate);
1072
+ const u = rng();
1073
+ sum += u < winRate ? 1 : u < winRate + lossRate ? -1 : 0;
1074
+ }
1075
+ sums[cluster] = sum;
1076
+ }
1077
+ for (let draw = 0; draw < config.resamples; draw++) {
1078
+ let pooledSum = 0;
1079
+ let pooledRows = 0;
1080
+ for (let pick = 0; pick < clusterCount; pick++) {
1081
+ const index = Math.floor(rng() * clusterCount);
1082
+ pooledSum += sums[index];
1083
+ pooledRows += config.clusterSizes[index];
1084
+ }
1085
+ means[draw] = pooledSum / pooledRows;
1086
+ }
1087
+ means.sort((a, b) => a - b);
1088
+ const tail = 1 - config.confidence;
1089
+ const lower = means[Math.floor(tail / 2 * config.resamples)];
1090
+ const upper = means[Math.max(Math.floor(tail / 2 * config.resamples), Math.min(config.resamples - 1, Math.ceil((1 - tail / 2) * config.resamples) - 1))];
1091
+ widths[trial] = upper - lower;
1092
+ if (lower > 0 || upper < 0) excludes += 1;
1093
+ }
1094
+ widths.sort((a, b) => a - b);
1095
+ return {
1096
+ effect,
1097
+ power: excludes / config.trials,
1098
+ medianCiWidth: widths[Math.floor(config.trials / 2)]
1099
+ };
1100
+ }
1101
+ /** Fold the effect into the seed so every grid point draws an independent stream. */
1102
+ function mixSeed(seed, effect) {
1103
+ return (seed ^ Math.round(effect * 1000003) * 2654435769) >>> 0 | 0;
1104
+ }
1105
+ //#endregion
1106
+ export { BOOTSTRAP_GATE_MIN_N, DecisionTableNotTotalError, DesignRefusalError, ExperimentTracker, FunnelIntegrityError, MatchedBudgetError, SealIntegrityError, amendExperiment, assertDesignAdequate, assertFunnelReconciles, assertMatchedBudgets, benjaminiHochberg, bonferroni, buildEvidenceVector, buildFunnel, canonicalize, classifyReissue, clusteredPairedBinary, clusteredPower, comparePairedArms, composeFunnels, computeEstimand, computeInterval, defineExperiment, eProcess, evaluateCondition, evaluateHaltRule, evaluateHypothesis, evaluateIdentityGate, evaluateOracleDeterminismGate, evaluatePopulationReproducibilityGate, evaluatePowerFloorGate, evaluatePredicate, evaluateProvenanceGate, executeAdmissionRule, executeDecisionRule, fileExperimentStore, hashJson, heldoutSignificance, holm, inMemoryExperimentStore, mcnemar, mcnemarPower, mcnemarRequiredN, mulberry32, openSealedExperiment, pairArms, pairHoldout, pairRunRecords, pairedBootstrap, pairedEvalueSequence, pairedMde, pairedRiskDifference, pairedRiskDifferenceExact, pairedRiskDifferenceScore, paretoPolicy, paretoSignificanceGate, powerPreflight, projectNLadderBudget, readField, renderFunnelTable, requiredPairedSampleSize, requiredSampleSize, runSelectionRule, runUniformPassBudget, sealExperiment, sequentialDecide, sequentialPairedGate, signManifest, verifyManifest, verifyMatchedBudgets, verifySealedExperiment, wilson };
1107
+
1108
+ //# sourceMappingURL=index.js.map