@tangle-network/agent-eval 0.144.5 → 0.144.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (257) hide show
  1. package/CHANGELOG.md +30 -0
  2. package/README.md +2 -0
  3. package/dist/{benchmark-Fmo42QVE.d.ts → agent-profile-CgDTo40f.d.ts} +91 -4
  4. package/dist/agent-profile-CgDTo40f.d.ts.map +1 -0
  5. package/dist/{agent-profile-cell-Cw0PVwDr.d.ts → agent-profile-cell-BOP-iA9Q.d.ts} +2 -2
  6. package/dist/{agent-profile-cell-Cw0PVwDr.d.ts.map → agent-profile-cell-BOP-iA9Q.d.ts.map} +1 -1
  7. package/dist/analyst/index.d.ts +471 -89
  8. package/dist/analyst/index.d.ts.map +1 -1
  9. package/dist/analyst/index.js +24 -5
  10. package/dist/analyst/index.js.map +1 -1
  11. package/dist/{analyze-runs-DWIvOAGk.js → analyze-runs-BNNK7irB.js} +3 -3
  12. package/dist/{analyze-runs-DWIvOAGk.js.map → analyze-runs-BNNK7irB.js.map} +1 -1
  13. package/dist/{baseline-D_fT6277.d.ts → baseline-CavEbRyH.d.ts} +2 -23
  14. package/dist/baseline-CavEbRyH.d.ts.map +1 -0
  15. package/dist/{benchmark-command-CA_NFOmy.js → benchmark-command-BCafwNrf.js} +664 -605
  16. package/dist/benchmark-command-BCafwNrf.js.map +1 -0
  17. package/dist/benchmarks/index.d.ts +1 -1
  18. package/dist/benchmarks/index.js +1 -1
  19. package/dist/{benchmarks-CWbsj0t4.js → benchmarks-CDSolHq7.js} +4 -4
  20. package/dist/{benchmarks-CWbsj0t4.js.map → benchmarks-CDSolHq7.js.map} +1 -1
  21. package/dist/campaign/index.d.ts +8 -6
  22. package/dist/campaign/index.js +5 -3
  23. package/dist/{campaign-ClpnD7Ug.js → campaign-Tdy3h62h.js} +17 -301
  24. package/dist/campaign-Tdy3h62h.js.map +1 -0
  25. package/dist/cli.js +2 -2
  26. package/dist/{client-DAb7MWtL.d.ts → client-DjXROWpx.d.ts} +4 -4
  27. package/dist/{client-DAb7MWtL.d.ts.map → client-DjXROWpx.d.ts.map} +1 -1
  28. package/dist/{completion-verifier-VvpHRu78.d.ts → completion-verifier-foUCLif_.d.ts} +6 -6
  29. package/dist/{completion-verifier-VvpHRu78.d.ts.map → completion-verifier-foUCLif_.d.ts.map} +1 -1
  30. package/dist/contract/index.d.ts +11 -10
  31. package/dist/contract/index.d.ts.map +1 -1
  32. package/dist/contract/index.js +7 -6
  33. package/dist/contract/index.js.map +1 -1
  34. package/dist/control.d.ts +2 -2
  35. package/dist/control.js +1 -1
  36. package/dist/{cost-ledger-FuQvHxPm.d.ts → cost-ledger-Bv_e8XHY.d.ts} +2 -2
  37. package/dist/{cost-ledger-FuQvHxPm.d.ts.map → cost-ledger-Bv_e8XHY.d.ts.map} +1 -1
  38. package/dist/counterfactual-CWPTrMH7.js +126 -0
  39. package/dist/counterfactual-CWPTrMH7.js.map +1 -0
  40. package/dist/counterfactual-CxmxAONP.d.ts +72 -0
  41. package/dist/counterfactual-CxmxAONP.d.ts.map +1 -0
  42. package/dist/{dataset-v_Y5902-.d.ts → dataset-C8xaLXdY.d.ts} +2 -2
  43. package/dist/{dataset-v_Y5902-.d.ts.map → dataset-C8xaLXdY.d.ts.map} +1 -1
  44. package/dist/{default-registry-BwbZ9N9v.d.ts → default-registry-BZhStdjl.d.ts} +6 -5
  45. package/dist/default-registry-BZhStdjl.d.ts.map +1 -0
  46. package/dist/{default-registry-RLNNoeEP.js → default-registry-BaQXW1Ow.js} +2 -2
  47. package/dist/{default-registry-RLNNoeEP.js.map → default-registry-BaQXW1Ow.js.map} +1 -1
  48. package/dist/{dspy-rlm-engine-19FQEMBK.js → dspy-rlm-engine-BiN49gK6.js} +4 -3
  49. package/dist/{dspy-rlm-engine-19FQEMBK.js.map → dspy-rlm-engine-BiN49gK6.js.map} +1 -1
  50. package/dist/{tool-groups-ByZiqpVk.d.ts → engine-nB64f48I.d.ts} +21 -34
  51. package/dist/engine-nB64f48I.d.ts.map +1 -0
  52. package/dist/{errors-DkfjIDvD.d.ts → errors-CKPfb2aH.d.ts} +2 -2
  53. package/dist/{errors-DkfjIDvD.d.ts.map → errors-CKPfb2aH.d.ts.map} +1 -1
  54. package/dist/errors-D-LKuDhb.js.map +1 -1
  55. package/dist/{eval-campaign-lZcDIwQM.js → eval-campaign-DNjCvAm-.js} +7 -6
  56. package/dist/{eval-campaign-lZcDIwQM.js.map → eval-campaign-DNjCvAm-.js.map} +1 -1
  57. package/dist/{exact-types-BQ7W90C4.d.ts → exact-types-Djvzosly.d.ts} +2 -2
  58. package/dist/{exact-types-BQ7W90C4.d.ts.map → exact-types-Djvzosly.d.ts.map} +1 -1
  59. package/dist/exec-BLtYZdWo.js +49 -0
  60. package/dist/exec-BLtYZdWo.js.map +1 -0
  61. package/dist/experiment/index.d.ts +802 -0
  62. package/dist/experiment/index.d.ts.map +1 -0
  63. package/dist/experiment/index.js +1108 -0
  64. package/dist/experiment/index.js.map +1 -0
  65. package/dist/experiment-tracker-CnRICnMl.js +500 -0
  66. package/dist/experiment-tracker-CnRICnMl.js.map +1 -0
  67. package/dist/experiment-tracker-IMntXr6J.d.ts +299 -0
  68. package/dist/experiment-tracker-IMntXr6J.d.ts.map +1 -0
  69. package/dist/{external-optimizer-contracts-CdmX2K2S.d.ts → external-optimizer-contracts-CZuJNcT5.d.ts} +3 -3
  70. package/dist/{external-optimizer-contracts-CdmX2K2S.d.ts.map → external-optimizer-contracts-CZuJNcT5.d.ts.map} +1 -1
  71. package/dist/{feedback-trajectory-WK7x4mhy.d.ts → feedback-trajectory-Rh280oXo.d.ts} +4 -4
  72. package/dist/{feedback-trajectory-WK7x4mhy.d.ts.map → feedback-trajectory-Rh280oXo.d.ts.map} +1 -1
  73. package/dist/fuzz.d.ts +2 -2
  74. package/dist/hosted/index.d.ts +3 -3
  75. package/dist/{index-BZUe-ODI.d.ts → index-BZ3-y4YL.d.ts} +3 -3
  76. package/dist/{index-BZUe-ODI.d.ts.map → index-BZ3-y4YL.d.ts.map} +1 -1
  77. package/dist/{index-DSC51roc.d.ts → index-C3ssXVLv.d.ts} +2 -2
  78. package/dist/{index-DSC51roc.d.ts.map → index-C3ssXVLv.d.ts.map} +1 -1
  79. package/dist/{index-CsuAo2-J.d.ts → index-C5HOo4ZF2.d.ts} +5 -5
  80. package/dist/index-C5HOo4ZF2.d.ts.map +1 -0
  81. package/dist/{index-DEb46kc6.d.ts → index-CvSN3IG1.d.ts} +2 -2
  82. package/dist/{index-DEb46kc6.d.ts.map → index-CvSN3IG1.d.ts.map} +1 -1
  83. package/dist/{index-CNOCxBLh.d.ts → index-CvXXlyz7.d.ts} +3 -3
  84. package/dist/{index-CNOCxBLh.d.ts.map → index-CvXXlyz7.d.ts.map} +1 -1
  85. package/dist/{index-DGIzNtRv.d.ts → index-CwDrUMe0.d.ts} +3 -3
  86. package/dist/{index-DGIzNtRv.d.ts.map → index-CwDrUMe0.d.ts.map} +1 -1
  87. package/dist/{index-CpxZSlB7.d.ts → index-Sh2I0DRc.d.ts} +15 -649
  88. package/dist/index-Sh2I0DRc.d.ts.map +1 -0
  89. package/dist/index.d.ts +252 -451
  90. package/dist/index.d.ts.map +1 -1
  91. package/dist/index.js +271 -751
  92. package/dist/index.js.map +1 -1
  93. package/dist/{insight-report-D5m1z0_n.d.ts → insight-report-C6h6F_4L.d.ts} +4 -4
  94. package/dist/{insight-report-D5m1z0_n.d.ts.map → insight-report-C6h6F_4L.d.ts.map} +1 -1
  95. package/dist/{integrity-DRXobPEs.d.ts → integrity-BuqEKu-x.d.ts} +3 -3
  96. package/dist/{integrity-DRXobPEs.d.ts.map → integrity-BuqEKu-x.d.ts.map} +1 -1
  97. package/dist/integrity-MLzHOfV9.js +141 -0
  98. package/dist/integrity-MLzHOfV9.js.map +1 -0
  99. package/dist/kind-factory-BHIgPmzS.js.map +1 -1
  100. package/dist/ledger-core/index.d.ts +1 -1
  101. package/dist/{llm-client-D3EoChAU.js → llm-client-DzvMUsS_.js} +310 -10
  102. package/dist/llm-client-DzvMUsS_.js.map +1 -0
  103. package/dist/matrix/index.d.ts +2 -2
  104. package/dist/meta-eval/index.d.ts +2 -2
  105. package/dist/{mint-DD-0oQTA.js → mint-CGEkzPLf.js} +2 -2
  106. package/dist/{mint-DD-0oQTA.js.map → mint-CGEkzPLf.js.map} +1 -1
  107. package/dist/{multi-layer-verifier-BHY1gWAc.d.ts → multi-layer-verifier-DnAqwl0h.d.ts} +2 -2
  108. package/dist/{multi-layer-verifier-BHY1gWAc.d.ts.map → multi-layer-verifier-DnAqwl0h.d.ts.map} +1 -1
  109. package/dist/multishot/index.d.ts +3 -3
  110. package/dist/openapi.json +1 -1
  111. package/dist/paired-promotion-decision-B6zJ3gYM.d.ts +114 -0
  112. package/dist/paired-promotion-decision-B6zJ3gYM.d.ts.map +1 -0
  113. package/dist/pipelines/index.d.ts +2 -1
  114. package/dist/pipelines/index.d.ts.map +1 -1
  115. package/dist/pre-registration-DakwTRXk.js +96 -0
  116. package/dist/pre-registration-DakwTRXk.js.map +1 -0
  117. package/dist/prime-bridge-transport-6feEglLf.d.ts +49 -0
  118. package/dist/prime-bridge-transport-6feEglLf.d.ts.map +1 -0
  119. package/dist/prime-protocol-BfSalTfR.js +453 -0
  120. package/dist/prime-protocol-BfSalTfR.js.map +1 -0
  121. package/dist/profile-cell.d.ts +1 -1
  122. package/dist/profile-cell.js +242 -1
  123. package/dist/profile-cell.js.map +1 -0
  124. package/dist/promotion-policy-ChWhTDBH.d.ts +289 -0
  125. package/dist/promotion-policy-ChWhTDBH.d.ts.map +1 -0
  126. package/dist/promotion-policy-CrLrmys8.js +682 -0
  127. package/dist/promotion-policy-CrLrmys8.js.map +1 -0
  128. package/dist/{propose-review-control-BciUCZoh.js → propose-review-control-qgWLA6E9.js} +2 -2
  129. package/dist/{propose-review-control-BciUCZoh.js.map → propose-review-control-qgWLA6E9.js.map} +1 -1
  130. package/dist/{release-report-BZRWdq_t.d.ts → release-report-CI8uisI1.d.ts} +4 -4
  131. package/dist/{release-report-BZRWdq_t.d.ts.map → release-report-CI8uisI1.d.ts.map} +1 -1
  132. package/dist/{release-report-Sl0xfkFv.js → release-report-Dy39cbFF.js} +3 -3
  133. package/dist/{release-report-Sl0xfkFv.js.map → release-report-Dy39cbFF.js.map} +1 -1
  134. package/dist/{replay-B7S7Pdbw.d.ts → replay-DFf-teiC.d.ts} +8 -7
  135. package/dist/replay-DFf-teiC.d.ts.map +1 -0
  136. package/dist/{replay-GW61ezMW.js → replay-DyBLaKFc.js} +3 -3
  137. package/dist/{replay-GW61ezMW.js.map → replay-DyBLaKFc.js.map} +1 -1
  138. package/dist/reporting.d.ts +4 -4
  139. package/dist/reporting.js +2 -2
  140. package/dist/{researcher-BSiCoM1s.d.ts → researcher-BoaxeCzP.d.ts} +6 -6
  141. package/dist/{researcher-BSiCoM1s.d.ts.map → researcher-BoaxeCzP.d.ts.map} +1 -1
  142. package/dist/{reward-hacking-CyuzxKly.js → reward-hacking-B2tPL9a3.js} +12 -3
  143. package/dist/reward-hacking-B2tPL9a3.js.map +1 -0
  144. package/dist/{reward-hacking-DFgkEY4p.d.ts → reward-hacking-Cf1PtEOz.d.ts} +34 -4
  145. package/dist/reward-hacking-Cf1PtEOz.d.ts.map +1 -0
  146. package/dist/rl.d.ts +19 -9
  147. package/dist/rl.d.ts.map +1 -1
  148. package/dist/rl.js +16 -7
  149. package/dist/rl.js.map +1 -1
  150. package/dist/rollout/index.d.ts +2 -2
  151. package/dist/rollout/index.js +2 -2
  152. package/dist/{rollout-C2fD1cf4.js → rollout-2ECTXb2N.js} +2 -2
  153. package/dist/{rollout-C2fD1cf4.js.map → rollout-2ECTXb2N.js.map} +1 -1
  154. package/dist/{rubric-predictive-validity-C4r4Y-q8.d.ts → rubric-predictive-validity-9qAwzkZm.d.ts} +2 -2
  155. package/dist/{rubric-predictive-validity-C4r4Y-q8.d.ts.map → rubric-predictive-validity-9qAwzkZm.d.ts.map} +1 -1
  156. package/dist/{run-evidence-j5Ynww6L.d.ts → run-evidence-BDFFai9R.d.ts} +3 -3
  157. package/dist/{run-evidence-j5Ynww6L.d.ts.map → run-evidence-BDFFai9R.d.ts.map} +1 -1
  158. package/dist/{run-record-ooo9FWns.d.ts → run-record-DdSa93_W.d.ts} +4 -4
  159. package/dist/{run-record-ooo9FWns.d.ts.map → run-record-DdSa93_W.d.ts.map} +1 -1
  160. package/dist/{run-record-CWN8-VsV.js → run-record-DqOw5X6_.js} +2 -2
  161. package/dist/{run-record-CWN8-VsV.js.map → run-record-DqOw5X6_.js.map} +1 -1
  162. package/dist/{schema-Cef2cFmb.d.ts → schema-Cef2cFmb2.d.ts} +1 -1
  163. package/dist/schema-Cef2cFmb2.d.ts.map +1 -0
  164. package/dist/{semantic-concept-judge-DKRtp2sY.js → semantic-concept-judge-Bmrq6yqU.js} +2 -2
  165. package/dist/{semantic-concept-judge-DKRtp2sY.js.map → semantic-concept-judge-Bmrq6yqU.js.map} +1 -1
  166. package/dist/sequential-D-BLJBKU.js +299 -0
  167. package/dist/sequential-D-BLJBKU.js.map +1 -0
  168. package/dist/{server-Df00sdwz.js → server-iu0ede49.js} +2 -2
  169. package/dist/{server-Df00sdwz.js.map → server-iu0ede49.js.map} +1 -1
  170. package/dist/{single-run-lock-D5iN0Xzb.js → single-run-lock-BMQEv1wG.js} +7 -161
  171. package/dist/single-run-lock-BMQEv1wG.js.map +1 -0
  172. package/dist/{skill-usage-DUvvudWR.d.ts → skill-usage-CJlWEUFt.d.ts} +11 -11
  173. package/dist/{skill-usage-DUvvudWR.d.ts.map → skill-usage-CJlWEUFt.d.ts.map} +1 -1
  174. package/dist/{skillopt-optimization-method-CGz9ywhM.d.ts → skillopt-optimization-method-BO7NAl3b.d.ts} +11 -297
  175. package/dist/skillopt-optimization-method-BO7NAl3b.d.ts.map +1 -0
  176. package/dist/{skillopt-optimization-method-C4FX42dy.js → skillopt-optimization-method-CQwZ-ZX8.js} +8 -652
  177. package/dist/skillopt-optimization-method-CQwZ-ZX8.js.map +1 -0
  178. package/dist/statistical-heldout-Dn9ruizm.d.ts +555 -0
  179. package/dist/statistical-heldout-Dn9ruizm.d.ts.map +1 -0
  180. package/dist/{statistics-B4u_CiFd.d.ts → statistics-D6Uebe_4.d.ts} +2 -2
  181. package/dist/{statistics-B4u_CiFd.d.ts.map → statistics-D6Uebe_4.d.ts.map} +1 -1
  182. package/dist/steps-BArUxhna.d.ts +51 -0
  183. package/dist/steps-BArUxhna.d.ts.map +1 -0
  184. package/dist/{integrity-fdt8XPAv.js → store-DNe_Uv1Q.js} +2 -140
  185. package/dist/store-DNe_Uv1Q.js.map +1 -0
  186. package/dist/{summary-report-BOM6dfP7.d.ts → summary-report-DuUS_i7W.d.ts} +4 -115
  187. package/dist/summary-report-DuUS_i7W.d.ts.map +1 -0
  188. package/dist/{summary-report-9A5y7EsK.js → summary-report-Lf-5I7xh.js} +2 -2
  189. package/dist/{summary-report-9A5y7EsK.js.map → summary-report-Lf-5I7xh.js.map} +1 -1
  190. package/dist/supervisor-run/index.d.ts +2 -2
  191. package/dist/supervisor-run/index.js +1 -1
  192. package/dist/{supervisor-run-DiyQVczd.js → supervisor-run-D_sokXcO.js} +32 -8
  193. package/dist/{supervisor-run-DiyQVczd.js.map → supervisor-run-D_sokXcO.js.map} +1 -1
  194. package/dist/tool-groups-DIVBnJyl.d.ts +28 -0
  195. package/dist/tool-groups-DIVBnJyl.d.ts.map +1 -0
  196. package/dist/trace-repair/index.d.ts +2102 -0
  197. package/dist/trace-repair/index.d.ts.map +1 -0
  198. package/dist/trace-repair/index.js +3878 -0
  199. package/dist/trace-repair/index.js.map +1 -0
  200. package/dist/traces.d.ts +6 -6
  201. package/dist/traces.js +3 -2
  202. package/dist/trajectory-YC15QDYQ.d.ts +24 -0
  203. package/dist/trajectory-YC15QDYQ.d.ts.map +1 -0
  204. package/dist/trajectory-replay/index.d.ts +781 -0
  205. package/dist/trajectory-replay/index.d.ts.map +1 -0
  206. package/dist/trajectory-replay/index.js +2103 -0
  207. package/dist/trajectory-replay/index.js.map +1 -0
  208. package/dist/{types-BjMFz88h.d.ts → types-D216SgwM.d.ts} +228 -9
  209. package/dist/types-D216SgwM.d.ts.map +1 -0
  210. package/dist/{types-yLK8gXE9.d.ts → types-D4mog56g.d.ts} +2 -2
  211. package/dist/{types-yLK8gXE9.d.ts.map → types-D4mog56g.d.ts.map} +1 -1
  212. package/dist/{types-CZt1PBIk.d.ts → types-DF_Udrp-.d.ts} +54 -5
  213. package/dist/{types-CZt1PBIk.d.ts.map → types-DF_Udrp-.d.ts.map} +1 -1
  214. package/dist/{types-Dcoaqcsc.d.ts → types-DYuNHo9R.d.ts} +5 -5
  215. package/dist/{types-Dcoaqcsc.d.ts.map → types-DYuNHo9R.d.ts.map} +1 -1
  216. package/dist/usage-receipt-EVI8B8Xu.js.map +1 -1
  217. package/dist/verdict-DExhxfgR.d.ts +201 -0
  218. package/dist/verdict-DExhxfgR.d.ts.map +1 -0
  219. package/dist/verdict-cache-BCcOh0kF.js +159 -0
  220. package/dist/verdict-cache-BCcOh0kF.js.map +1 -0
  221. package/dist/wire/index.d.ts +3 -3
  222. package/dist/wire/index.js +1 -1
  223. package/docs/building-doctrine.md +15 -0
  224. package/docs/charter.md +112 -0
  225. package/docs/experiment.md +104 -0
  226. package/docs/prime-analyst.md +1 -0
  227. package/docs/trace-analysis.md +26 -0
  228. package/docs/trace-repair-admission.md +194 -0
  229. package/docs/trace-repair-analyst-arms.md +121 -0
  230. package/docs/trace-repair-continuation.md +107 -0
  231. package/docs/trace-repair-grader.md +163 -0
  232. package/docs/trajectory-replay.md +110 -0
  233. package/docs/verification-strategies.md +103 -0
  234. package/package.json +21 -4
  235. package/dist/agent-profile-cell-CbfBm2g6.js +0 -335
  236. package/dist/agent-profile-cell-CbfBm2g6.js.map +0 -1
  237. package/dist/baseline-D_fT6277.d.ts.map +0 -1
  238. package/dist/benchmark-Fmo42QVE.d.ts.map +0 -1
  239. package/dist/benchmark-command-CA_NFOmy.js.map +0 -1
  240. package/dist/campaign-ClpnD7Ug.js.map +0 -1
  241. package/dist/default-registry-BwbZ9N9v.d.ts.map +0 -1
  242. package/dist/index-CpxZSlB7.d.ts.map +0 -1
  243. package/dist/index-CsuAo2-J.d.ts.map +0 -1
  244. package/dist/integrity-fdt8XPAv.js.map +0 -1
  245. package/dist/llm-client-D3EoChAU.js.map +0 -1
  246. package/dist/replay-B7S7Pdbw.d.ts.map +0 -1
  247. package/dist/reward-hacking-CyuzxKly.js.map +0 -1
  248. package/dist/reward-hacking-DFgkEY4p.d.ts.map +0 -1
  249. package/dist/schema-Cef2cFmb.d.ts.map +0 -1
  250. package/dist/single-run-lock-D5iN0Xzb.js.map +0 -1
  251. package/dist/skillopt-optimization-method-C4FX42dy.js.map +0 -1
  252. package/dist/skillopt-optimization-method-CGz9ywhM.d.ts.map +0 -1
  253. package/dist/summary-report-BOM6dfP7.d.ts.map +0 -1
  254. package/dist/tool-groups-ByZiqpVk.d.ts.map +0 -1
  255. package/dist/types-BjMFz88h.d.ts.map +0 -1
  256. package/dist/verdict-Dps8_okt.d.ts +0 -37
  257. package/dist/verdict-Dps8_okt.d.ts.map +0 -1
@@ -0,0 +1,682 @@
1
+ import { E as pairedBootstrap, M as pairedRiskDifferenceScore, N as pairedSignTest, O as pairedDeltaTieFraction, T as pairedBinaryScale, j as pairedRiskDifferenceExact } from "./statistics-ByxzSiOM.js";
2
+ //#region src/paired-delta-test.ts
3
+ /** Smallest all-positive sample that can clear a one-sided exact sign test. */
4
+ function minimumPairsForPairedDeltaTest(confidence = .95) {
5
+ if (!Number.isFinite(confidence) || confidence <= 0 || confidence >= 1) throw new Error(`minimumPairsForPairedDeltaTest: confidence must be in (0,1), got ${confidence}`);
6
+ const oneSidedAlpha = (1 - confidence) / 2;
7
+ return Math.ceil(Math.log2(1 / oneSidedAlpha));
8
+ }
9
+ /**
10
+ * Tests whether a paired candidate-minus-baseline delta clears a threshold.
11
+ *
12
+ * At 20 or more pairs, the percentile bootstrap lower bound carries the
13
+ * decision. Below that point the interval is descriptive only, so the function
14
+ * switches to a pre-registered one-sided exact sign test. The exact path is
15
+ * deliberately conservative: it requires both a point estimate above the
16
+ * threshold and enough consistently positive paired differences.
17
+ *
18
+ * ## A zero-width interval is never significant
19
+ *
20
+ * When every paired delta is identical the resample distribution is a point
21
+ * mass and the interval collapses: `[0, 0]` when all pairs tie, `[g, g]` on n
22
+ * identical deltas of g. Neither says the effect is certain — both say the
23
+ * sample carries no information about how far the estimate could be wrong, and
24
+ * `low > threshold` then answers on the point estimate alone. It fails in both
25
+ * directions: `[0, 0]` clears every NEGATIVE threshold, which is how a
26
+ * tie-dominated pass/fail comparison laundered a regression into a
27
+ * noninferiority pass, and `[g, g]` clears every threshold below g with no
28
+ * spread behind it. Under a bounded asymmetric null whose true mean paired
29
+ * delta is exactly 0 — 2 % of pairs dropping by 1.0, the rest gaining 0.0204 —
30
+ * every sample that misses the drop is exactly that shape, and deciding on
31
+ * `low > 0` promoted 65.65 % of samples at n = 20 against a nominal 5 %.
32
+ *
33
+ * So `indeterminate` is reported and `significant` is false whenever the
34
+ * interval has zero width, on BOTH paths: at small n the exact sign test is a
35
+ * test of the MEDIAN and a zero-spread sample is precisely where it stops
36
+ * saying anything about the mean the caller is thresholding.
37
+ *
38
+ * `threshold` may be negative — that is a noninferiority margin, and it is the
39
+ * regime the zero-width hole is worst in. For a two-point (pass/fail) outcome
40
+ * the percentile bootstrap is not a valid interval at a nonzero margin at all;
41
+ * use {@link decidePairedPromotion}, which routes those to Tango's score
42
+ * interval, rather than thresholding this function's bootstrap directly.
43
+ */
44
+ function pairedDeltaTest(before, after, options = {}) {
45
+ const threshold = options.threshold ?? 0;
46
+ if (!Number.isFinite(threshold)) throw new Error(`pairedDeltaTest: threshold must be finite, got ${threshold}`);
47
+ const exactMinimum = minimumPairsForPairedDeltaTest(options.confidence ?? .95);
48
+ const requestedMinimum = options.minPairs ?? exactMinimum;
49
+ if (!Number.isInteger(requestedMinimum) || requestedMinimum < 1) throw new Error(`pairedDeltaTest: minPairs must be a positive integer, got ${requestedMinimum}`);
50
+ const minimumPairs = Math.max(requestedMinimum, exactMinimum);
51
+ const bootstrap = pairedBootstrap(before, after, options);
52
+ const sufficient = bootstrap.n >= minimumPairs;
53
+ const indeterminate = bootstrap.n > 0 && (!Number.isFinite(bootstrap.low) || !Number.isFinite(bootstrap.high) || bootstrap.low === bootstrap.high);
54
+ if (bootstrap.gateEligible) return {
55
+ bootstrap,
56
+ method: "bootstrap-ci",
57
+ pValue: null,
58
+ minimumPairs,
59
+ sufficient,
60
+ indeterminate,
61
+ significant: sufficient && !indeterminate && bootstrap.low > threshold
62
+ };
63
+ const exact = pairedSignTest(before.map((value, index) => after[index] - value - threshold), "greater");
64
+ const estimate = options.statistic === "mean" ? bootstrap.mean : bootstrap.median;
65
+ return {
66
+ bootstrap,
67
+ method: "exact-sign",
68
+ pValue: exact.pValue,
69
+ minimumPairs,
70
+ sufficient,
71
+ indeterminate,
72
+ significant: sufficient && !indeterminate && estimate > threshold && exact.pValue <= (1 - bootstrap.confidence) / 2
73
+ };
74
+ }
75
+ //#endregion
76
+ //#region src/paired-promotion-decision.ts
77
+ /**
78
+ * @module
79
+ * ONE rule for "does this paired interval clear a promotion threshold".
80
+ *
81
+ * The rule below was derived on `HeldOutGate` (#479) after the same estimator
82
+ * bug shipped twice. It then turned out that a SECOND gate — the composable
83
+ * `heldOutGate`, plus everything else routed through `heldoutSignificance` —
84
+ * still carried the original defect, because the rule had been written into one
85
+ * gate's method body rather than into a shared function. Two copies of a
86
+ * statistical rule is how a defect survives in one of them, so there is now
87
+ * exactly one copy and both gates call it.
88
+ *
89
+ * Three things the rule does that a bare `pairedBootstrap(...).low > threshold`
90
+ * does not:
91
+ *
92
+ * 1. **Two-point (pass/fail) outcomes decide on Tango's SCORE interval.** On a
93
+ * pass/fail eval the paired delta vector is dominated by ties, so the
94
+ * bootstrap of the mean is a resample of a lattice with three atoms and its
95
+ * percentile interval is not valid at a nonzero margin. The score interval
96
+ * (`pairedRiskDifferenceScore`) re-estimates the nuisance loss rate under
97
+ * each hypothesised margin instead of fixing it at the observed value, which
98
+ * is the only construction that stays a confidence interval as the margin
99
+ * moves off zero — the regime every noninferiority threshold lives in.
100
+ * Measured on the composable gate before this change, at a true risk
101
+ * difference sitting exactly on the production caller's -0.05 margin and a
102
+ * nominal 5 %: 14.60 % false promotion at n = 40 and 10.10 % at n = 76.
103
+ * 2. **McNemar's exact test holds a VETO at every non-negative threshold.**
104
+ * Redundant with the interval by construction and kept anyway, so that
105
+ * swapping the estimator for one without that duality cannot silently
106
+ * reintroduce "promotes what the exact test refuses". Witness: n = 6, b = 5,
107
+ * c = 0 — no exact argument reaches alpha = 0.05 with 5 discordant pairs
108
+ * (two-sided floor 2/2^5 = 0.0625), whatever an interval says. A NEGATIVE
109
+ * threshold is a noninferiority question, which McNemar's test of "no
110
+ * difference" is not the right test for, so the veto does not apply there.
111
+ * 3. **A ZERO-WIDTH interval is refused, wherever it sits.** At [0, 0] it
112
+ * cannot tell a gain from a regression and clears every negative threshold.
113
+ * Away from zero it fails the opposite way: n identical positive deltas give
114
+ * [g, g], which clears threshold 0 on no spread at all. Both are an absence
115
+ * of evidence. Measured on the composable gate before this change, under a
116
+ * bounded asymmetric null whose true mean paired delta is exactly 0: 88.50 %
117
+ * false promotion at n = 6 and 65.65 % at n = 20 against a nominal 5 %.
118
+ *
119
+ * Orthogonal to the small-sample switch inside {@link pairedDeltaTest}: that
120
+ * picks the TEST from the sample size (bootstrap CI at n >= 20, pre-registered
121
+ * exact sign test below it), this picks the ESTIMATOR from the outcome's shape.
122
+ * Both are needed — an exact sign test applied to a tie-pinned median is still
123
+ * blind, and a mean bootstrap CI at n = 6 is still not a valid test.
124
+ */
125
+ /**
126
+ * Which estimator {@link decidePairedPromotion} would use on this data, and the
127
+ * shape facts behind it — for callers that must report the shape on a path
128
+ * where no interval is computed at all (an early rejection, or zero pairs).
129
+ * Cheap: no bootstrap, no interval.
130
+ */
131
+ function pairedDecisionShape(before, after, statistic = "mean") {
132
+ const tieFraction = before.length === 0 ? null : pairedDeltaTieFraction(before, after);
133
+ if (statistic === "median") return {
134
+ statistic: "median_bootstrap",
135
+ binaryScale: null,
136
+ tieFraction
137
+ };
138
+ const binaryScale = pairedBinaryScale(before, after);
139
+ if (binaryScale !== null) return {
140
+ statistic: "paired_risk_difference",
141
+ binaryScale,
142
+ tieFraction
143
+ };
144
+ return {
145
+ statistic: "mean_bootstrap",
146
+ binaryScale: null,
147
+ tieFraction
148
+ };
149
+ }
150
+ /**
151
+ * Decide whether a paired candidate-minus-baseline delta clears a promotion
152
+ * threshold. `before` is the baseline arm, `after` the candidate arm, paired by
153
+ * position. Throws on unequal lengths.
154
+ */
155
+ function decidePairedPromotion(before, after, options = {}) {
156
+ if (before.length !== after.length) throw new Error(`decidePairedPromotion: unequal sample sizes (${before.length} vs ${after.length})`);
157
+ const threshold = options.threshold ?? 0;
158
+ if (!Number.isFinite(threshold)) throw new Error(`decidePairedPromotion: threshold must be finite, got ${threshold}`);
159
+ const confidence = options.confidence ?? .95;
160
+ const exactMinimum = minimumPairsForPairedDeltaTest(confidence);
161
+ const requestedMinimum = options.minPairs ?? exactMinimum;
162
+ if (!Number.isInteger(requestedMinimum) || requestedMinimum < 1) throw new Error(`decidePairedPromotion: minPairs must be a positive integer, got ${requestedMinimum}`);
163
+ const minimumPairs = Math.max(requestedMinimum, exactMinimum);
164
+ const n = before.length;
165
+ const sufficient = n >= minimumPairs;
166
+ const { binaryScale, tieFraction } = pairedDecisionShape(before, after, options.statistic);
167
+ let core;
168
+ if (binaryScale !== null) {
169
+ const unitControl = before.map((v) => v / binaryScale);
170
+ const unitTreatment = after.map((v) => v / binaryScale);
171
+ const exact = pairedRiskDifferenceExact(unitControl, unitTreatment, confidence);
172
+ const score = pairedRiskDifferenceScore(unitControl, unitTreatment, confidence);
173
+ const low = score.lower * binaryScale;
174
+ core = {
175
+ statistic: "paired_risk_difference",
176
+ method: "score-interval",
177
+ delta: score.riskDifference * binaryScale,
178
+ low,
179
+ high: score.upper * binaryScale,
180
+ bootstrap: null,
181
+ mcnemar: {
182
+ b: exact.b,
183
+ c: exact.c,
184
+ nDiscordant: exact.nDiscordant,
185
+ pValue: exact.pValue
186
+ },
187
+ pValue: null,
188
+ clearsThreshold: low > threshold,
189
+ label: "success-rate",
190
+ methodDetail: ""
191
+ };
192
+ } else {
193
+ const bootstrapStatistic = options.statistic === "median" ? "median" : "mean";
194
+ const test = pairedDeltaTest(before, after, {
195
+ confidence,
196
+ resamples: options.resamples,
197
+ statistic: bootstrapStatistic,
198
+ seed: options.seed,
199
+ threshold,
200
+ minPairs: options.minPairs
201
+ });
202
+ const ci = test.bootstrap;
203
+ core = {
204
+ statistic: bootstrapStatistic === "mean" ? "mean_bootstrap" : "median_bootstrap",
205
+ method: test.method,
206
+ delta: bootstrapStatistic === "mean" ? ci.mean : ci.median,
207
+ low: ci.low,
208
+ high: ci.high,
209
+ bootstrap: ci,
210
+ mcnemar: null,
211
+ pValue: test.pValue,
212
+ clearsThreshold: test.significant,
213
+ label: bootstrapStatistic,
214
+ methodDetail: test.method === "exact-sign" ? ` Below ${test.minimumPairs} pairs the interval is descriptive only; the decision is the exact one-sided sign test, p=${fmt(test.pValue ?? 1)}.` : ""
215
+ };
216
+ }
217
+ const indeterminate = !Number.isFinite(core.low) || !Number.isFinite(core.high) || core.low === core.high;
218
+ const indeterminateCause = !indeterminate ? "" : tieFraction === 1 ? "every paired delta is an exact tie" : core.mcnemar !== null && core.mcnemar.nDiscordant === 0 ? "every pair is concordant (0 discordant pairs)" : `the ${core.label} CI collapsed to a point at ${fmt(core.low)}`;
219
+ const exactTestVetoes = core.mcnemar !== null && threshold >= 0 && !(core.mcnemar.pValue < 1 - confidence);
220
+ return {
221
+ n,
222
+ threshold,
223
+ confidence,
224
+ binaryScale,
225
+ tieFraction,
226
+ minimumPairs,
227
+ sufficient,
228
+ indeterminate,
229
+ indeterminateCause,
230
+ exactTestVetoes,
231
+ promote: sufficient && !indeterminate && core.clearsThreshold && !exactTestVetoes,
232
+ ...core
233
+ };
234
+ }
235
+ function fmt(x) {
236
+ return x.toFixed(4);
237
+ }
238
+ //#endregion
239
+ //#region src/campaign/gates/statistical-heldout.ts
240
+ /**
241
+ * Statistical held-out promotion machinery — the trustworthy core the
242
+ * point-estimate `heldout-delta` gate lacked.
243
+ *
244
+ * The shipped false positive it prevents: a winner re-scored against the
245
+ * baseline on the holdout read run-to-run model NOISE (e.g. 91 vs 95) as a
246
+ * "+4 lift" and shipped, because the gate compared point estimates with no
247
+ * confidence interval. Here we pair candidate vs baseline holdout observations
248
+ * and bootstrap a CI on the paired delta — a candidate ships only when the CI
249
+ * lower bound clears the effect-size threshold (the gain is real at the
250
+ * confidence level, not noise), and is blocked when a critical dimension
251
+ * (e.g. `hallucination_free` for a legal agent) significantly regresses even if
252
+ * the net composite rose (anti-Goodhart).
253
+ *
254
+ * Two traps this module is built around (both produce a NEW false positive if
255
+ * gotten wrong):
256
+ * 1. PAIRING GRANULARITY — pairs by FULL `cellId` (`scenario:rep`), never by
257
+ * `scenarioId` (which averages reps away and destroys the within-pair
258
+ * variance reduction that makes a paired bootstrap tighter than unpaired).
259
+ * One paired observation per cell ⇒ reps multiply n.
260
+ * 2. SCALE — a judge may emit composites/dimensions on [0,1] or 0-100. The
261
+ * threshold + tolerance are interpreted in the judge's NATIVE scale; the
262
+ * per-dimension tolerance auto-scales off the observed baseline magnitudes
263
+ * so `-0.10` on [0,1] doesn't silently become a no-op on a 0-100 dimension.
264
+ */
265
+ /** Tie fraction at/above which a gate annotates its verdict with the tie share.
266
+ * Tie-domination of the median bites structurally at >= 0.5 (the median is then
267
+ * 0 by construction); 0.4 is a softer warn threshold that flags a run APPROACHING
268
+ * that regime, so an operator sees it before the median goes fully blind. */
269
+ const TIE_WARN_FRACTION = .4;
270
+ /**
271
+ * Pair candidate vs baseline holdout observations by FULL cellId. `select`
272
+ * pulls the scalar from a cell's judge reports (composite, or a named
273
+ * dimension); a cell contributes the mean of `select` across its judges. Cells
274
+ * whose scenario is not in `scenarioIds`, or where `select` is undefined for
275
+ * every judge on either side, are skipped on BOTH sides so the arrays stay
276
+ * paired. Throws when the two maps disagree on which holdout cells exist — a
277
+ * load-bearing invariant: the baseline + winner holdout campaigns run the same
278
+ * scenarios with the same seed base, so their cellIds MUST align; a mismatch
279
+ * means a silent pairing bug, not a soft fallback.
280
+ */
281
+ function pairHoldout(candidate, baseline, scenarioIds, select) {
282
+ const cellValue = (byCell, cellId) => {
283
+ const scores = byCell.get(cellId);
284
+ if (!scores) return void 0;
285
+ const vals = [];
286
+ for (const s of Object.values(scores)) {
287
+ if (s.failed === true) throw new Error(`pairHoldout: cell '${cellId}' contains a failed judge score`);
288
+ const v = select(s);
289
+ if (typeof v === "number" && !Number.isFinite(v)) throw new Error(`pairHoldout: cell '${cellId}' contains a non-finite selected score`);
290
+ if (typeof v === "number") vals.push(v);
291
+ }
292
+ if (vals.length === 0) return void 0;
293
+ return vals.reduce((a, b) => a + b, 0) / vals.length;
294
+ };
295
+ const inScope = (cellId) => scenarioIds.has(cellId.split(":")[0] ?? "");
296
+ const candCells = [...candidate.keys()].filter(inScope).sort();
297
+ const baseCells = [...baseline.keys()].filter(inScope).sort();
298
+ if (candCells.length !== baseCells.length || candCells.some((c, i) => c !== baseCells[i])) throw new Error(`pairHoldout: candidate/baseline holdout cells do not align — candidate=[${candCells.join(",")}] baseline=[${baseCells.join(",")}]. Both holdout campaigns must run the same scenarios with the same seed base.`);
299
+ const before = [];
300
+ const after = [];
301
+ const cellIds = [];
302
+ for (const cellId of candCells) {
303
+ const b = cellValue(baseline, cellId);
304
+ const a = cellValue(candidate, cellId);
305
+ if (b === void 0 && a === void 0) continue;
306
+ if (b === void 0 || a === void 0) throw new Error(`pairHoldout: cell '${cellId}' has a selected score on only one arm`);
307
+ before.push(b);
308
+ after.push(a);
309
+ cellIds.push(cellId);
310
+ }
311
+ return {
312
+ before,
313
+ after,
314
+ cellIds
315
+ };
316
+ }
317
+ /**
318
+ * Significance of the held-out composite lift: ship only when the lower bound
319
+ * of the interval the outcome's shape ADMITS exceeds `deltaThreshold` (default
320
+ * 0 ⇒ "confidently positive"). Interpret `deltaThreshold` in the judge's native
321
+ * scale.
322
+ *
323
+ * The decision is delegated whole to {@link decidePairedPromotion}, the one
324
+ * copy of the rule (`src/paired-promotion-decision.ts`), which `HeldOutGate`
325
+ * also calls. That module's header carries the measurements; the short version
326
+ * is three guards a bare `bootstrap.low > threshold` does not have:
327
+ *
328
+ * - a two-point (pass/fail) outcome decides on Tango's SCORE interval, the
329
+ * only paired-binary construction that stays valid at a nonzero margin;
330
+ * - McNemar's exact test VETOES at any non-negative threshold;
331
+ * - a ZERO-WIDTH interval is refused rather than promoted, in either
332
+ * direction — [0,0] clears every negative threshold and [g,g] clears every
333
+ * threshold below g, and both are an absence of evidence, not a result.
334
+ *
335
+ * Measured on this function before those guards landed, at a nominal 5 %:
336
+ * 14.60 % false promotion at n = 40 on a paired-binary noninferiority boundary,
337
+ * and 88.50 % at n = 6 under a bounded asymmetric null whose true mean paired
338
+ * delta is exactly 0.
339
+ *
340
+ * At small n, where the percentile bootstrap is descriptive only, a
341
+ * pre-registered exact sign test still carries the bootstrap path.
342
+ */
343
+ function heldoutSignificance(paired, opts = {}) {
344
+ const deltaThreshold = opts.deltaThreshold ?? 0;
345
+ const confidence = opts.confidence ?? .95;
346
+ const resamples = opts.resamples ?? 2e3;
347
+ const seed = opts.seed ?? 1337;
348
+ const statistic = opts.statistic ?? "mean";
349
+ const decision = decidePairedPromotion(paired.before, paired.after, {
350
+ confidence,
351
+ resamples,
352
+ statistic,
353
+ seed,
354
+ threshold: deltaThreshold,
355
+ minPairs: opts.minProductiveRuns
356
+ });
357
+ const bootstrap = decision.bootstrap ?? pairedBootstrap(paired.before, paired.after, {
358
+ confidence,
359
+ resamples,
360
+ statistic,
361
+ seed
362
+ });
363
+ const medianBootstrap = statistic === "median" ? bootstrap : pairedBootstrap(paired.before, paired.after, {
364
+ confidence,
365
+ resamples,
366
+ statistic: "median",
367
+ seed
368
+ });
369
+ const n = paired.before.length;
370
+ let ties = 0;
371
+ for (let i = 0; i < n; i += 1) {
372
+ const after = paired.after[i] ?? 0;
373
+ const before = paired.before[i] ?? 0;
374
+ if (Math.abs(after - before) < 1e-9) ties += 1;
375
+ }
376
+ const tieFraction = n === 0 ? 0 : ties / n;
377
+ return {
378
+ paired,
379
+ bootstrap,
380
+ medianBootstrap,
381
+ decision,
382
+ decisionStatistic: decision.statistic,
383
+ mcnemar: decision.mcnemar,
384
+ tieFraction,
385
+ n,
386
+ minimumRequired: decision.minimumPairs,
387
+ decisionMethod: decision.method,
388
+ pValue: decision.pValue,
389
+ significant: decision.promote,
390
+ fewRuns: !decision.sufficient
391
+ };
392
+ }
393
+ /** Detect the native scale of a set of scores: 0-100 when any magnitude clears
394
+ * 1.5, else [0,1]. Used to auto-scale the regression tolerance so a default
395
+ * expressed for [0,1] is not silently a no-op on a 0-100 dimension. */
396
+ function detectScale(values) {
397
+ return values.some((v) => Math.abs(v) > 1.5) ? 100 : 1;
398
+ }
399
+ /** Per-critical-dimension regression guard. For each dimension, pair the
400
+ * candidate vs baseline values by full cellId and bootstrap the paired delta;
401
+ * a dimension is "regressed" when the CI lower bound < −tolerance (conservative
402
+ * — blocks if the credible worst case exceeds tolerance, which is the right
403
+ * posture for safety dimensions like `hallucination_free`). When `tolerance`
404
+ * is omitted it auto-scales: 0.05 on [0,1], 5 on 0-100.
405
+ *
406
+ * The interval comes from {@link decidePairedPromotion}, so a pass/fail
407
+ * dimension is judged on Tango's score interval rather than a percentile
408
+ * bootstrap of the mean — `tolerance` is a NONZERO margin, and the bootstrap
409
+ * is not a valid interval at one. That matters most here because this guard
410
+ * fails OPEN by construction: `tolerance` is positive, so an interval pinned at
411
+ * [0,0] never satisfies `low < −tolerance` and a real regression on a safety
412
+ * dimension would be reported as `regressed: false`. On the median it fails the
413
+ * same way for the same reason — when most pairs tie, which is automatic for a
414
+ * pass/fail dimension on {0,1} and on the 0-100 encoding `detectScale` exists
415
+ * to support, the median CI collapses to [0,0]. Pass `statistic: 'median'` to
416
+ * restore the pre-0.134 behaviour. */
417
+ function dimensionRegressions(candidate, baseline, scenarioIds, criticalDimensions, opts = {}) {
418
+ const out = [];
419
+ for (const dim of criticalDimensions) {
420
+ const paired = pairHoldout(candidate, baseline, scenarioIds, (s) => s.dimensions[dim]);
421
+ if (paired.before.length === 0) continue;
422
+ const tolerance = opts.tolerance ?? .05 * detectScale([...paired.before, ...paired.after]);
423
+ const bootstrapStatistic = opts.statistic ?? "mean";
424
+ const shared = {
425
+ confidence: opts.confidence ?? .95,
426
+ resamples: opts.resamples ?? 2e3,
427
+ statistic: bootstrapStatistic,
428
+ seed: opts.seed ?? 1337
429
+ };
430
+ const guard = decidePairedPromotion(paired.before, paired.after, shared);
431
+ const regression = decidePairedPromotion(paired.after, paired.before, {
432
+ ...shared,
433
+ threshold: tolerance
434
+ });
435
+ const bootstrap = guard.bootstrap ?? pairedBootstrap(paired.before, paired.after, shared);
436
+ out.push({
437
+ dimension: dim,
438
+ bootstrap,
439
+ bootstrapStatistic,
440
+ ci: {
441
+ low: guard.low,
442
+ high: guard.high
443
+ },
444
+ decisionStatistic: guard.statistic,
445
+ mcnemar: guard.mcnemar,
446
+ indeterminate: guard.indeterminate,
447
+ regressed: bootstrap.low < -tolerance || regression.promote,
448
+ tolerance,
449
+ n: paired.before.length
450
+ });
451
+ }
452
+ return out;
453
+ }
454
+ //#endregion
455
+ //#region src/campaign/gates/power-preflight.ts
456
+ /** Two-sided z for the common confidence levels; interpolation is overkill here. */
457
+ function zFor(confidence) {
458
+ if (confidence >= .99) return 2.576;
459
+ if (confidence >= .95) return 1.96;
460
+ if (confidence >= .9) return 1.645;
461
+ return 1.282;
462
+ }
463
+ /** Estimate the minimum detectable lift a paired-holdout improvement run can
464
+ * ship at a given budget, from the baseline holdout composites — call it BEFORE
465
+ * spending a search to learn whether the effect you are hunting is even
466
+ * observable at this holdout size and worker variance. */
467
+ function powerPreflight(opts) {
468
+ const composites = opts.baselineComposites.filter((v) => Number.isFinite(v));
469
+ if (composites.length < 3) throw new Error(`powerPreflight: need >= 3 finite baseline composites to estimate variance, got ${composites.length}`);
470
+ const deltaThreshold = opts.deltaThreshold ?? .05;
471
+ const confidence = opts.confidence ?? .95;
472
+ const n = opts.pairedN ?? composites.length;
473
+ if (n < 2) throw new Error(`powerPreflight: pairedN must be >= 2, got ${n}`);
474
+ const mean = composites.reduce((a, b) => a + b, 0) / composites.length;
475
+ const variance = composites.reduce((a, b) => a + (b - mean) * (b - mean), 0) / (composites.length - 1);
476
+ const sd = Math.sqrt(variance);
477
+ const z = zFor(confidence);
478
+ const mde = deltaThreshold + z * Math.SQRT2 * sd / Math.sqrt(n);
479
+ const scaleAssumed = composites.every((v) => v >= -.001 && v <= 1.5);
480
+ const headroom = Math.max(0, 1 - mean);
481
+ const underpowered = scaleAssumed && mde > headroom;
482
+ const sharedChannelCaveat = opts.sharedScorerChannel ? "Holdout and gate share one scoring channel: raising n/reps reduces only idiosyncratic noise — systematic judge bias remains and this MDE is a lower bound. Full debiasing needs an independent second scoring channel (different judge/benchmark family)." : void 0;
483
+ const recommendation = underpowered ? `UNDERPOWERED: minimum detectable lift ${mde.toFixed(3)} exceeds the ${headroom.toFixed(3)} headroom above the baseline (${mean.toFixed(3)}) — no achievable effect can ship at this budget. Raise paired n (scenarios x reps) to ~${Math.ceil((z * Math.SQRT2 * sd / Math.max(headroom - deltaThreshold, .01)) ** 2)} or reduce worker variance before searching.` : `Minimum detectable lift at n=${n}: ${mde.toFixed(3)} (baseline sd ${sd.toFixed(3)}). Effects smaller than this cannot clear the gate; budget the search for effects you believe exceed it.`;
484
+ return {
485
+ n,
486
+ sd,
487
+ mde,
488
+ baselineMean: mean,
489
+ headroom,
490
+ underpowered,
491
+ scaleAssumed,
492
+ deltaThreshold,
493
+ confidence,
494
+ ...sharedChannelCaveat ? { sharedChannelCaveat } : {},
495
+ recommendation: sharedChannelCaveat ? `${recommendation} ${sharedChannelCaveat}` : recommendation
496
+ };
497
+ }
498
+ //#endregion
499
+ //#region src/campaign/gates/promotion-policy.ts
500
+ /**
501
+ * Promotion policy over the evidence VECTOR — the substrate's answer to "never
502
+ * collapse the multi-objective promotion decision into one scalar." A
503
+ * `defaultProductionGate` is one opinionated composition; this module factors
504
+ * the decision into two reusable pieces so MANY policies can compete over the
505
+ * SAME evidence (the quant-desk pattern: one evidence bus, plural strategies):
506
+ *
507
+ * buildEvidenceVector(ctx, objectives, opts) -> EvidenceVector // the bus
508
+ * PromotionPolicy = (ev: EvidenceVector) => GateResult // a strategy
509
+ * paretoPolicy(ev) // the default strategy
510
+ * paretoSignificanceGate(options): Gate // bus + policy as a Gate
511
+ *
512
+ * The Pareto policy is SYMMETRIC multi-objective: every objective is BOTH a
513
+ * potential gain source AND a safety floor (unlike `defaultProductionGate`,
514
+ * where only `composite` can win and `criticalDimensions` are pure floors). A
515
+ * candidate ships iff it weakly DOMINATES the baseline at the confidence level —
516
+ * no objective credibly worse (CI floor breach) AND at least one objective
517
+ * credibly better (CI gain). Insufficient evidence on ANY axis -> need_more_work
518
+ * (NOT folded into hold: "gather more reps" and "reject" are different actions).
519
+ *
520
+ * Cost/latency are NOT CI axes here — `GateContext` carries only an aggregate
521
+ * per-side cost, no per-cell observation vector to bootstrap. Treat them as hard
522
+ * constraints (compose with a budget gate via `composeGate`), not faked CIs.
523
+ */
524
+ /**
525
+ * The Evidence Bus. For each objective, pair candidate vs baseline by full
526
+ * cellId and bootstrap a CI on the good-direction paired delta. Reuses the
527
+ * exact `pairHoldout` + `pairedBootstrap` machinery the held-out gate uses, so
528
+ * a single source of truth governs pairing granularity + scale handling.
529
+ */
530
+ function buildEvidenceVector(ctx, objectives, opts = {}) {
531
+ if (objectives.length === 0) throw new Error("buildEvidenceVector: at least 1 objective required");
532
+ const confidence = opts.confidence ?? .95;
533
+ const resamples = opts.resamples ?? 2e3;
534
+ const seed = opts.seed ?? 1337;
535
+ const baseline = ctx.baselineJudgeScores ?? ctx.judgeScores;
536
+ const scenarioIds = new Set(ctx.scenarios.map((s) => s.id));
537
+ const axes = [];
538
+ for (const obj of objectives) {
539
+ let select;
540
+ if (obj.source.kind === "composite") select = (s) => s.composite;
541
+ else {
542
+ const dim = obj.source.dimension;
543
+ select = (s) => s.dimensions[dim];
544
+ }
545
+ const paired = pairHoldout(ctx.judgeScores, baseline, scenarioIds, select);
546
+ const before = obj.direction === "maximize" ? paired.before : paired.after;
547
+ const after = obj.direction === "maximize" ? paired.after : paired.before;
548
+ const n = paired.before.length;
549
+ const floorTolerance = obj.floorTolerance ?? .05 * detectScale([...paired.before, ...paired.after]);
550
+ const gainThreshold = obj.gainThreshold ?? 0;
551
+ const bootstrapStatistic = opts.statistic ?? "mean";
552
+ const improvement = decidePairedPromotion(before, after, {
553
+ confidence,
554
+ resamples,
555
+ statistic: bootstrapStatistic,
556
+ seed,
557
+ threshold: gainThreshold,
558
+ minPairs: opts.minProductiveRuns
559
+ });
560
+ const regression = decidePairedPromotion(after, before, {
561
+ confidence,
562
+ resamples,
563
+ statistic: bootstrapStatistic,
564
+ seed,
565
+ threshold: floorTolerance,
566
+ minPairs: opts.minProductiveRuns
567
+ });
568
+ const bootstrap = improvement.bootstrap ?? pairedBootstrap(before, after, {
569
+ confidence,
570
+ resamples,
571
+ statistic: bootstrapStatistic,
572
+ seed
573
+ });
574
+ const floorBreached = bootstrap.low < -floorTolerance || regression.promote;
575
+ const verdict = !improvement.sufficient ? "few_runs" : floorBreached ? "regressed" : improvement.promote ? "improved" : "flat";
576
+ axes.push({
577
+ name: obj.name,
578
+ source: obj.source,
579
+ direction: obj.direction,
580
+ bootstrap,
581
+ bootstrapStatistic,
582
+ ci: {
583
+ low: improvement.low,
584
+ high: improvement.high
585
+ },
586
+ decisionStatistic: improvement.statistic,
587
+ mcnemar: improvement.mcnemar,
588
+ indeterminate: improvement.indeterminate,
589
+ n,
590
+ minimumRequired: improvement.minimumPairs,
591
+ decisionMethod: improvement.method,
592
+ gainThreshold,
593
+ floorTolerance,
594
+ verdict
595
+ });
596
+ }
597
+ const ns = axes.map((a) => a.n).filter((n) => n > 0);
598
+ return {
599
+ axes,
600
+ minN: ns.length > 0 ? Math.min(...ns) : 0,
601
+ cost: {
602
+ candidate: ctx.cost.candidate,
603
+ baseline: ctx.cost.baseline
604
+ }
605
+ };
606
+ }
607
+ /**
608
+ * The default strategy: symmetric multi-objective Pareto significance. Ship iff
609
+ * the candidate weakly dominates the baseline at the confidence level — no axis
610
+ * credibly worse AND ≥1 axis credibly better. Floor breach on any axis → hold
611
+ * (anti-Goodhart, dominates everything). Insufficient evidence on any axis →
612
+ * need_more_work. Statistically equivalent → hold (never ship noise).
613
+ */
614
+ const paretoPolicy = (ev) => {
615
+ const contributingGates = ev.axes.map((ax) => ({
616
+ name: `objective:${ax.name}`,
617
+ status: ax.verdict === "regressed" ? "fail" : ax.verdict === "few_runs" ? "not_evaluated" : "pass",
618
+ detail: {
619
+ direction: ax.direction,
620
+ source: ax.source,
621
+ verdict: ax.verdict,
622
+ n: ax.n,
623
+ deltaMedian: ax.bootstrap.median,
624
+ ciLow: ax.ci.low,
625
+ ciHigh: ax.ci.high,
626
+ decisionStatistic: ax.decisionStatistic,
627
+ decisionMethod: ax.decisionMethod,
628
+ mcnemar: ax.mcnemar,
629
+ indeterminate: ax.indeterminate,
630
+ bootstrapCiLow: ax.bootstrap.low,
631
+ bootstrapCiHigh: ax.bootstrap.high,
632
+ confidence: ax.bootstrap.confidence,
633
+ gainThreshold: ax.gainThreshold,
634
+ floorTolerance: ax.floorTolerance
635
+ }
636
+ }));
637
+ const regressed = ev.axes.filter((a) => a.verdict === "regressed");
638
+ const fewRuns = ev.axes.filter((a) => a.verdict === "few_runs");
639
+ const improved = ev.axes.filter((a) => a.verdict === "improved");
640
+ let decision;
641
+ const reasons = [];
642
+ if (regressed.length > 0) {
643
+ decision = "hold";
644
+ for (const a of regressed) reasons.push(`objective '${a.name}' regressed: good-direction CI.low ${a.ci.low.toFixed(3)} < -${a.floorTolerance} (n=${a.n})`);
645
+ } else if (fewRuns.length > 0) {
646
+ decision = "need_more_work";
647
+ for (const a of fewRuns) reasons.push(`objective '${a.name}' has only n=${a.n} paired runs — insufficient evidence to claim significance`);
648
+ } else if (improved.length > 0) {
649
+ decision = "ship";
650
+ reasons.push(`Pareto improvement at the confidence level: ${improved.map((a) => `'${a.name}' +${a.ci.low > 0 ? a.ci.low.toFixed(3) : a.bootstrap.mean.toFixed(3)} (CI.low ${a.ci.low.toFixed(3)})`).join(", ")}; no objective regressed`);
651
+ } else {
652
+ decision = "hold";
653
+ reasons.push("no Pareto improvement: candidate statistically equivalent to baseline on every objective");
654
+ }
655
+ const composite = ev.axes.find((a) => a.source.kind === "composite") ?? ev.axes[0];
656
+ return {
657
+ decision,
658
+ reasons,
659
+ contributingGates,
660
+ delta: composite?.bootstrap.median
661
+ };
662
+ };
663
+ /**
664
+ * Wrap the bus + a policy as a `Gate`. Plugs into the existing
665
+ * `runImprovementLoop({ gate })` slot and composes via `composeGate`; default
666
+ * loop behavior is unchanged because consumers opt in by passing this gate.
667
+ */
668
+ function paretoSignificanceGate(options) {
669
+ if (options.objectives.length === 0) throw new Error("paretoSignificanceGate: at least 1 objective required");
670
+ const policy = options.policy ?? paretoPolicy;
671
+ return {
672
+ name: options.name ?? "paretoSignificanceGate",
673
+ async decide(ctx) {
674
+ const ev = buildEvidenceVector(ctx, options.objectives, options);
675
+ return policy(ev);
676
+ }
677
+ };
678
+ }
679
+ //#endregion
680
+ export { TIE_WARN_FRACTION as a, heldoutSignificance as c, pairedDecisionShape as d, minimumPairsForPairedDeltaTest as f, powerPreflight as i, pairHoldout as l, paretoPolicy as n, detectScale as o, pairedDeltaTest as p, paretoSignificanceGate as r, dimensionRegressions as s, buildEvidenceVector as t, decidePairedPromotion as u };
681
+
682
+ //# sourceMappingURL=promotion-policy-CrLrmys8.js.map