@tangle-network/agent-eval 0.144.11 → 0.144.13

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (378) hide show
  1. package/CHANGELOG.md +11 -0
  2. package/dist/agent-profile-B9_GGsG8.d.ts +84 -0
  3. package/dist/agent-profile-B9_GGsG8.d.ts.map +1 -0
  4. package/dist/{agent-profile-cell-BOP-iA9Q.d.ts → agent-profile-cell-BkcRDikH.d.ts} +2 -2
  5. package/dist/{agent-profile-cell-BOP-iA9Q.d.ts.map → agent-profile-cell-BkcRDikH.d.ts.map} +1 -1
  6. package/dist/analyst/index.d.ts +134 -16
  7. package/dist/analyst/index.d.ts.map +1 -1
  8. package/dist/analyst/index.js +364 -10
  9. package/dist/analyst/index.js.map +1 -1
  10. package/dist/backend-integrity-CsVin_Wb.d.ts +280 -0
  11. package/dist/backend-integrity-CsVin_Wb.d.ts.map +1 -0
  12. package/dist/{baseline-C-GocmIW.js → baseline-BhPRQBVn.js} +3 -2
  13. package/dist/{baseline-C-GocmIW.js.map → baseline-BhPRQBVn.js.map} +1 -1
  14. package/dist/{benchmark-CWeqGl7x.js → benchmark-BhT16ep9.js} +2 -2
  15. package/dist/{benchmark-CWeqGl7x.js.map → benchmark-BhT16ep9.js.map} +1 -1
  16. package/dist/{agent-profile-DPi7IZg7.d.ts → benchmark-Ceoan7vk.d.ts} +4 -91
  17. package/dist/benchmark-Ceoan7vk.d.ts.map +1 -0
  18. package/dist/{benchmark-command-BteMFN62.js → benchmark-command-CY6cbhOB.js} +36 -79
  19. package/dist/benchmark-command-CY6cbhOB.js.map +1 -0
  20. package/dist/benchmarks/index.d.ts +244 -2
  21. package/dist/benchmarks/index.d.ts.map +1 -0
  22. package/dist/benchmarks/index.js +733 -1
  23. package/dist/benchmarks/index.js.map +1 -0
  24. package/dist/builder-eval/index.d.ts +23 -2
  25. package/dist/builder-eval/index.d.ts.map +1 -1
  26. package/dist/builder-eval/index.js +227 -3
  27. package/dist/builder-eval/index.js.map +1 -1
  28. package/dist/campaign/index.d.ts +10 -8
  29. package/dist/campaign/index.js +9 -6
  30. package/dist/{campaign-C2TTzQII.js → campaign-BYjBAypg.js} +21 -634
  31. package/dist/campaign-BYjBAypg.js.map +1 -0
  32. package/dist/{canonical-D011XM8r.js → canonical-D-XsTQ6_.js} +2 -2
  33. package/dist/{canonical-D011XM8r.js.map → canonical-D-XsTQ6_.js.map} +1 -1
  34. package/dist/{index-CQTZ-4XN.d.ts → capture-fetch-DDvpjVRU.d.ts} +3 -3
  35. package/dist/capture-fetch-DDvpjVRU.d.ts.map +1 -0
  36. package/dist/{default-registry-BmktKy8r.js → chat-client-Bp4Ebuuc.js} +1276 -1276
  37. package/dist/chat-client-Bp4Ebuuc.js.map +1 -0
  38. package/dist/cli.js +2 -2
  39. package/dist/{client-C9gzZE59.d.ts → client-KUbGulm_.d.ts} +4 -4
  40. package/dist/{client-C9gzZE59.d.ts.map → client-KUbGulm_.d.ts.map} +1 -1
  41. package/dist/contract/index.d.ts +13 -390
  42. package/dist/contract/index.d.ts.map +1 -1
  43. package/dist/contract/index.js +18 -542
  44. package/dist/contract/index.js.map +1 -1
  45. package/dist/{cost-ledger-DMFxsLKr.js → cost-ledger-BSe92yAV.js} +3 -3
  46. package/dist/{cost-ledger-DMFxsLKr.js.map → cost-ledger-BSe92yAV.js.map} +1 -1
  47. package/dist/{cost-ledger-Bv_e8XHY.d.ts → cost-ledger-DbQdN3nO.d.ts} +2 -2
  48. package/dist/{cost-ledger-Bv_e8XHY.d.ts.map → cost-ledger-DbQdN3nO.d.ts.map} +1 -1
  49. package/dist/{counterfactual-CxmxAONP.d.ts → counterfactual--bpysZF0.d.ts} +2 -14
  50. package/dist/{counterfactual-CxmxAONP.d.ts.map → counterfactual--bpysZF0.d.ts.map} +1 -1
  51. package/dist/{counterfactual-CWPTrMH7.js → counterfactual-lDfCx0Uz.js} +3 -31
  52. package/dist/{counterfactual-CWPTrMH7.js.map → counterfactual-lDfCx0Uz.js.map} +1 -1
  53. package/dist/{dataset-C8xaLXdY.d.ts → dataset-CJjKqQfA.d.ts} +2 -8
  54. package/dist/dataset-CJjKqQfA.d.ts.map +1 -0
  55. package/dist/{default-registry-Bf8Woqmq.d.ts → default-registry-Di6HP6pG.d.ts} +7 -12
  56. package/dist/default-registry-Di6HP6pG.d.ts.map +1 -0
  57. package/dist/{analyze-runs-C30yljDJ.js → define-agent-eval-iqjT--ZZ.js} +542 -130
  58. package/dist/define-agent-eval-iqjT--ZZ.js.map +1 -0
  59. package/dist/define-agent-eval-sH24zBfM.d.ts +388 -0
  60. package/dist/define-agent-eval-sH24zBfM.d.ts.map +1 -0
  61. package/dist/descriptive-B5MwKfbf.js +144 -0
  62. package/dist/descriptive-B5MwKfbf.js.map +1 -0
  63. package/dist/{dspy-rlm-engine-DbTk4JdR.js → dspy-rlm-engine-B_qhSc21.js} +4 -4
  64. package/dist/{dspy-rlm-engine-DbTk4JdR.js.map → dspy-rlm-engine-B_qhSc21.js.map} +1 -1
  65. package/dist/effect-sizes-DiH8MGOH.js +82 -0
  66. package/dist/effect-sizes-DiH8MGOH.js.map +1 -0
  67. package/dist/{engine-3hL-XqwJ.d.ts → engine-otFpE2gF.d.ts} +10 -38
  68. package/dist/engine-otFpE2gF.d.ts.map +1 -0
  69. package/dist/{errors-CKPfb2aH.d.ts → errors-DEE6u6ot.d.ts} +2 -14
  70. package/dist/{errors-CKPfb2aH.d.ts.map → errors-DEE6u6ot.d.ts.map} +1 -1
  71. package/dist/{errors-D-LKuDhb.js → errors-Dngq5h35.js} +2 -8
  72. package/dist/{errors-D-LKuDhb.js.map → errors-Dngq5h35.js.map} +1 -1
  73. package/dist/{eval-campaign-B_7wcnav.js → eval-campaign-UB-usSQ2.js} +6 -6
  74. package/dist/{eval-campaign-B_7wcnav.js.map → eval-campaign-UB-usSQ2.js.map} +1 -1
  75. package/dist/{exact-types-DSFFpLLI.d.ts → exact-types-BH1twmAJ.d.ts} +2 -2
  76. package/dist/{exact-types-DSFFpLLI.d.ts.map → exact-types-BH1twmAJ.d.ts.map} +1 -1
  77. package/dist/experiment/index.d.ts +9 -6
  78. package/dist/experiment/index.d.ts.map +1 -1
  79. package/dist/experiment/index.js +11 -7
  80. package/dist/experiment/index.js.map +1 -1
  81. package/dist/experiment-tracker-C29gXM4B.js +269 -0
  82. package/dist/experiment-tracker-C29gXM4B.js.map +1 -0
  83. package/dist/{experiment-tracker-IMntXr6J.d.ts → experiment-tracker-DWHZBAYL.d.ts} +77 -121
  84. package/dist/experiment-tracker-DWHZBAYL.d.ts.map +1 -0
  85. package/dist/{external-optimizer-contracts-CZuJNcT5.d.ts → external-optimizer-contracts-lixrOZdX.d.ts} +3 -3
  86. package/dist/{external-optimizer-contracts-CZuJNcT5.d.ts.map → external-optimizer-contracts-lixrOZdX.d.ts.map} +1 -1
  87. package/dist/external-optimizer-process-BTiNB-RH.js +301 -0
  88. package/dist/external-optimizer-process-BTiNB-RH.js.map +1 -0
  89. package/dist/{single-run-lock-DFWHEB09.js → external-optimizer-subprocess-DrJ9hR8u.js} +144 -438
  90. package/dist/external-optimizer-subprocess-DrJ9hR8u.js.map +1 -0
  91. package/dist/extract-usage-BrQ8mCLX.js +155 -0
  92. package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
  93. package/dist/{failure-cluster-CqcvCcdR.d.ts → failure-cluster-BLURuWG4.d.ts} +2 -3
  94. package/dist/failure-cluster-BLURuWG4.d.ts.map +1 -0
  95. package/dist/{feedback-trajectory-CacHpxsp.d.ts → feedback-trajectory-juOozjAc.d.ts} +5 -36
  96. package/dist/feedback-trajectory-juOozjAc.d.ts.map +1 -0
  97. package/dist/fuzz.d.ts +2 -2
  98. package/dist/fuzz.js +2 -2
  99. package/dist/hosted/index.d.ts +3 -3
  100. package/dist/{index-CvSN3IG1.d.ts → index-B8Ui1mr1.d.ts} +2 -2
  101. package/dist/{index-CvSN3IG1.d.ts.map → index-B8Ui1mr1.d.ts.map} +1 -1
  102. package/dist/{skill-usage-CO9OLRBx.d.ts → index-BWDrSVfw.d.ts} +11 -136
  103. package/dist/index-BWDrSVfw.d.ts.map +1 -0
  104. package/dist/index-Ba3YrbAL.d.ts +1 -0
  105. package/dist/{index-BnEuDAK2.d.ts → index-COQYtuRF.d.ts} +3 -3
  106. package/dist/{index-BnEuDAK2.d.ts.map → index-COQYtuRF.d.ts.map} +1 -1
  107. package/dist/{index-C3ssXVLv.d.ts → index-CvjYbU0D.d.ts} +2 -2
  108. package/dist/{index-C3ssXVLv.d.ts.map → index-CvjYbU0D.d.ts.map} +1 -1
  109. package/dist/{index-DPPGNJ_R.d.ts → index-DSmEylT9.d.ts} +17 -115
  110. package/dist/index-DSmEylT9.d.ts.map +1 -0
  111. package/dist/index.d.ts +2397 -5308
  112. package/dist/index.d.ts.map +1 -1
  113. package/dist/index.js +5914 -10496
  114. package/dist/index.js.map +1 -1
  115. package/dist/{insight-report-CgX_s0Ez.d.ts → insight-report-CRi-Ufrj.d.ts} +4 -4
  116. package/dist/{insight-report-CgX_s0Ez.d.ts.map → insight-report-CRi-Ufrj.d.ts.map} +1 -1
  117. package/dist/{integrity-MLzHOfV9.js → integrity-Cy9WHAtb.js} +2 -2
  118. package/dist/{integrity-MLzHOfV9.js.map → integrity-Cy9WHAtb.js.map} +1 -1
  119. package/dist/{integrity-DY6tIbl0.js → integrity-DysDBWDu.js} +2 -2
  120. package/dist/{integrity-DY6tIbl0.js.map → integrity-DysDBWDu.js.map} +1 -1
  121. package/dist/{integrity-BuqEKu-x.d.ts → integrity-OrcI9Nau.d.ts} +3 -3
  122. package/dist/{integrity-BuqEKu-x.d.ts.map → integrity-OrcI9Nau.d.ts.map} +1 -1
  123. package/dist/internal-BDHPCnjk.js +230 -0
  124. package/dist/internal-BDHPCnjk.js.map +1 -0
  125. package/dist/judge-calibration-C5CbMYce.d.ts +117 -0
  126. package/dist/judge-calibration-C5CbMYce.d.ts.map +1 -0
  127. package/dist/judge-calibration-DZkWrm5H.js +317 -0
  128. package/dist/judge-calibration-DZkWrm5H.js.map +1 -0
  129. package/dist/{kind-factory-B8-r8-y8.js → kind-factory-CPmSd58s.js} +208 -208
  130. package/dist/{kind-factory-B8-r8-y8.js.map → kind-factory-CPmSd58s.js.map} +1 -1
  131. package/dist/ledger-core/index.d.ts +1 -1
  132. package/dist/ledger-core/index.js +2 -2
  133. package/dist/{ledger-core-DXZIqu17.js → ledger-core-BmZt19oQ.js} +110 -110
  134. package/dist/{ledger-core-DXZIqu17.js.map → ledger-core-BmZt19oQ.js.map} +1 -1
  135. package/dist/{llm-client-DzvMUsS_.js → llm-client-d0-2TT1g.js} +4 -64
  136. package/dist/{llm-client-DzvMUsS_.js.map → llm-client-d0-2TT1g.js.map} +1 -1
  137. package/dist/{skillopt-optimization-method-CQdVeM8k.js → llm-judge-dZ8P6nGI.js} +3349 -5401
  138. package/dist/llm-judge-dZ8P6nGI.js.map +1 -0
  139. package/dist/matrix/index.d.ts +2 -2
  140. package/dist/meta-eval/index.d.ts +3 -3
  141. package/dist/meta-eval/index.js +3 -3
  142. package/dist/{metrics-C9YY1OcL.js → metrics-Cl0L1KUy.js} +2 -108
  143. package/dist/{metrics-C9YY1OcL.js.map → metrics-Cl0L1KUy.js.map} +1 -1
  144. package/dist/{mint-B2O60ACG.js → mint-Dj9Ww_3I.js} +4 -4
  145. package/dist/{mint-B2O60ACG.js.map → mint-Dj9Ww_3I.js.map} +1 -1
  146. package/dist/{multi-layer-verifier-DnAqwl0h.d.ts → multi-layer-verifier-DIguZc8Z.d.ts} +3 -3
  147. package/dist/{multi-layer-verifier-DnAqwl0h.d.ts.map → multi-layer-verifier-DIguZc8Z.d.ts.map} +1 -1
  148. package/dist/multiplicity-DIWHvysC.d.ts +43 -0
  149. package/dist/multiplicity-DIWHvysC.d.ts.map +1 -0
  150. package/dist/multishot/index.d.ts +3 -3
  151. package/dist/multishot/index.js +1 -1
  152. package/dist/openapi.json +1 -1
  153. package/dist/{opencode-sqlite-8r6WUfHc.js → opencode-sqlite-DJWAXLms.js} +2 -2
  154. package/dist/{opencode-sqlite-8r6WUfHc.js.map → opencode-sqlite-DJWAXLms.js.map} +1 -1
  155. package/dist/package-version-D7lQHt_-.js +34 -0
  156. package/dist/package-version-D7lQHt_-.js.map +1 -0
  157. package/dist/paired-arms-D-XRF_fy.js +1045 -0
  158. package/dist/paired-arms-D-XRF_fy.js.map +1 -0
  159. package/dist/paired-promotion-decision-CGzg0cI_.d.ts +251 -0
  160. package/dist/paired-promotion-decision-CGzg0cI_.d.ts.map +1 -0
  161. package/dist/paired-tests-BHIhYVdu.js +213 -0
  162. package/dist/paired-tests-BHIhYVdu.js.map +1 -0
  163. package/dist/pareto-BqNW3LJR.d.ts +117 -0
  164. package/dist/pareto-BqNW3LJR.d.ts.map +1 -0
  165. package/dist/pipelines/index.d.ts +3 -64
  166. package/dist/pipelines/index.d.ts.map +1 -1
  167. package/dist/pipelines/index.js +4 -284
  168. package/dist/pipelines/index.js.map +1 -1
  169. package/dist/power-and-mde-CHIrXJll.js +195 -0
  170. package/dist/power-and-mde-CHIrXJll.js.map +1 -0
  171. package/dist/{promotion-policy-CrLrmys8.js → power-preflight-DEw-uC7q.js} +4 -184
  172. package/dist/power-preflight-DEw-uC7q.js.map +1 -0
  173. package/dist/pre-registration-CZwSFQS4.d.ts +577 -0
  174. package/dist/pre-registration-CZwSFQS4.d.ts.map +1 -0
  175. package/dist/{prime-protocol-BfSalTfR.js → prime-protocol-6tZTVsWm.js} +72 -21
  176. package/dist/prime-protocol-6tZTVsWm.js.map +1 -0
  177. package/dist/produced-state-DU79a81m.js +586 -0
  178. package/dist/produced-state-DU79a81m.js.map +1 -0
  179. package/dist/profile-cell.d.ts +1 -1
  180. package/dist/profile-cell.js +1 -1
  181. package/dist/promotion-policy-D0nPhkSy.d.ts +134 -0
  182. package/dist/promotion-policy-D0nPhkSy.d.ts.map +1 -0
  183. package/dist/promotion-policy-xzA40Evo.js +186 -0
  184. package/dist/promotion-policy-xzA40Evo.js.map +1 -0
  185. package/dist/{skillopt-optimization-method-USDKhxSA.d.ts → provenance-CCdxgLDT.d.ts} +57 -478
  186. package/dist/provenance-CCdxgLDT.d.ts.map +1 -0
  187. package/dist/registry-oJeeI4-a.d.ts +178 -0
  188. package/dist/registry-oJeeI4-a.d.ts.map +1 -0
  189. package/dist/{release-report-DA2BCu5a.d.ts → release-confidence-BFRE5WSp.d.ts} +180 -112
  190. package/dist/release-confidence-BFRE5WSp.d.ts.map +1 -0
  191. package/dist/{release-report-BUYmoKo2.js → release-confidence-CxDuiAev.js} +125 -217
  192. package/dist/release-confidence-CxDuiAev.js.map +1 -0
  193. package/dist/reporting.d.ts +6 -5
  194. package/dist/reporting.js +7 -5
  195. package/dist/{researcher-BchpD55R.d.ts → researcher-DaN4GST-.d.ts} +7 -40
  196. package/dist/{researcher-BchpD55R.d.ts.map → researcher-DaN4GST-.d.ts.map} +1 -1
  197. package/dist/{reward-hacking-BDToousL.js → reward-hacking-DNgjilrV.js} +3 -3
  198. package/dist/reward-hacking-DNgjilrV.js.map +1 -0
  199. package/dist/{reward-hacking-BI0OMAlo.d.ts → reward-hacking-DSSTuI9r.d.ts} +5 -5
  200. package/dist/{reward-hacking-BI0OMAlo.d.ts.map → reward-hacking-DSSTuI9r.d.ts.map} +1 -1
  201. package/dist/rl.d.ts +7 -7
  202. package/dist/rl.js +11 -10
  203. package/dist/rl.js.map +1 -1
  204. package/dist/rollout/index.d.ts +2 -2
  205. package/dist/rollout/index.js +4 -4
  206. package/dist/{rollout-DRrksrcV.js → rollout-BWtw0I_6.js} +3 -3
  207. package/dist/{rollout-DRrksrcV.js.map → rollout-BWtw0I_6.js.map} +1 -1
  208. package/dist/{rubric-predictive-validity-DvkjPYCe.d.ts → rubric-predictive-validity-BgxtKe4G.d.ts} +2 -2
  209. package/dist/{rubric-predictive-validity-DvkjPYCe.d.ts.map → rubric-predictive-validity-BgxtKe4G.d.ts.map} +1 -1
  210. package/dist/{rubric-predictive-validity-BRR632r1.js → rubric-predictive-validity-Cwwyd7ah.js} +2 -2
  211. package/dist/{rubric-predictive-validity-BRR632r1.js.map → rubric-predictive-validity-Cwwyd7ah.js.map} +1 -1
  212. package/dist/{run-record-BmSPWXJR.js → run-record-BvHPVS-i.js} +3 -3
  213. package/dist/{run-record-BmSPWXJR.js.map → run-record-BvHPVS-i.js.map} +1 -1
  214. package/dist/{run-record-CF4Dwpxr.d.ts → run-record-CKiihE6f.d.ts} +4 -4
  215. package/dist/{run-record-CF4Dwpxr.d.ts.map → run-record-CKiihE6f.d.ts.map} +1 -1
  216. package/dist/{proposal-findings-2GIUo1et.js → run-score-lDzV0X8j.js} +31 -31
  217. package/dist/run-score-lDzV0X8j.js.map +1 -0
  218. package/dist/sandbox-harness-BlSOu4LX.d.ts +70 -0
  219. package/dist/sandbox-harness-BlSOu4LX.d.ts.map +1 -0
  220. package/dist/{schema-Cef2cFmb2.d.ts → schema-Cef2cFmb.d.ts} +1 -1
  221. package/dist/schema-Cef2cFmb.d.ts.map +1 -0
  222. package/dist/semantic-concept-judge-BiJxScqe.js +406 -0
  223. package/dist/semantic-concept-judge-BiJxScqe.js.map +1 -0
  224. package/dist/{sequential-D-BLJBKU.js → sequential-C458DXNf.js} +4 -3
  225. package/dist/{sequential-D-BLJBKU.js.map → sequential-C458DXNf.js.map} +1 -1
  226. package/dist/sequential-eprocess-CbUt2htw.js +83 -0
  227. package/dist/sequential-eprocess-CbUt2htw.js.map +1 -0
  228. package/dist/series-convergence-D1cL1f-4.d.ts +129 -0
  229. package/dist/series-convergence-D1cL1f-4.d.ts.map +1 -0
  230. package/dist/{server-iu0ede49.js → server-ulsOdrTI.js} +5 -21
  231. package/dist/server-ulsOdrTI.js.map +1 -0
  232. package/dist/skillopt-optimization-method-C5cotF4E.d.ts +427 -0
  233. package/dist/skillopt-optimization-method-C5cotF4E.d.ts.map +1 -0
  234. package/dist/skillopt-optimization-method-jjdnc3YK.js +1999 -0
  235. package/dist/skillopt-optimization-method-jjdnc3YK.js.map +1 -0
  236. package/dist/{statistical-heldout-TQ-4CYiN.d.ts → statistical-heldout-DoFF5KJX.d.ts} +5 -278
  237. package/dist/statistical-heldout-DoFF5KJX.d.ts.map +1 -0
  238. package/dist/{store-otlp-Dw8PPIlL.js → store-otlp-CsptLYpN.js} +3 -3
  239. package/dist/{store-otlp-Dw8PPIlL.js.map → store-otlp-CsptLYpN.js.map} +1 -1
  240. package/dist/store-tool-spans-Cq9mFd-q.js +667 -0
  241. package/dist/store-tool-spans-Cq9mFd-q.js.map +1 -0
  242. package/dist/store-tool-spans-DbDLOBOb.d.ts +342 -0
  243. package/dist/store-tool-spans-DbDLOBOb.d.ts.map +1 -0
  244. package/dist/student-t-CvBq2mve.js +38 -0
  245. package/dist/student-t-CvBq2mve.js.map +1 -0
  246. package/dist/{summary-report-Lf-5I7xh.js → summary-report-Blysd6Z2.js} +6 -3
  247. package/dist/{summary-report-Lf-5I7xh.js.map → summary-report-Blysd6Z2.js.map} +1 -1
  248. package/dist/{summary-report-BNR7DWTj.d.ts → summary-report-CaL-Hnxt.d.ts} +5 -5
  249. package/dist/{summary-report-BNR7DWTj.d.ts.map → summary-report-CaL-Hnxt.d.ts.map} +1 -1
  250. package/dist/supervisor-run/index.d.ts +391 -3
  251. package/dist/supervisor-run/index.d.ts.map +1 -0
  252. package/dist/supervisor-run/index.js +1689 -2
  253. package/dist/{supervisor-run-D_sokXcO.js.map → supervisor-run/index.js.map} +1 -1
  254. package/dist/{extract-usage-CdZdoj1s.js → task-failure-attributes-CpQ4y5RD.js} +5 -157
  255. package/dist/task-failure-attributes-CpQ4y5RD.js.map +1 -0
  256. package/dist/{tool-groups-CZPGGlHf.d.ts → tool-groups-Cteb03Ps.d.ts} +3 -3
  257. package/dist/tool-groups-Cteb03Ps.d.ts.map +1 -0
  258. package/dist/tool-waste-BDdBZG1F.js +803 -0
  259. package/dist/tool-waste-BDdBZG1F.js.map +1 -0
  260. package/dist/tool-waste-DjRDEsuI.d.ts +128 -0
  261. package/dist/tool-waste-DjRDEsuI.d.ts.map +1 -0
  262. package/dist/trace-repair/index.d.ts +14 -5
  263. package/dist/trace-repair/index.d.ts.map +1 -1
  264. package/dist/trace-repair/index.js +35 -7
  265. package/dist/trace-repair/index.js.map +1 -1
  266. package/dist/traces.d.ts +406 -7
  267. package/dist/traces.d.ts.map +1 -0
  268. package/dist/traces.js +1011 -10
  269. package/dist/traces.js.map +1 -0
  270. package/dist/trajectory-replay/index.d.ts +16 -3
  271. package/dist/trajectory-replay/index.d.ts.map +1 -1
  272. package/dist/trajectory-replay/index.js +52 -5
  273. package/dist/trajectory-replay/index.js.map +1 -1
  274. package/dist/types-BEPZc6eo.d.ts +93 -0
  275. package/dist/types-BEPZc6eo.d.ts.map +1 -0
  276. package/dist/{usage-receipt-t7vAzCRQ.js → types-BI4fT3HN.js} +67 -67
  277. package/dist/types-BI4fT3HN.js.map +1 -0
  278. package/dist/{types-BnjdJ70P.d.ts → types-BZ59Ahr8.d.ts} +5 -5
  279. package/dist/{types-BnjdJ70P.d.ts.map → types-BZ59Ahr8.d.ts.map} +1 -1
  280. package/dist/{types-D216SgwM.d.ts → types-Cx3YUh2r.d.ts} +4 -241
  281. package/dist/types-Cx3YUh2r.d.ts.map +1 -0
  282. package/dist/{types-BvDKaULh.d.ts → types-DLQx4mKU.d.ts} +4 -4
  283. package/dist/{types-BvDKaULh.d.ts.map → types-DLQx4mKU.d.ts.map} +1 -1
  284. package/dist/{types-D4mog56g.d.ts → types-yLK8gXE9.d.ts} +2 -2
  285. package/dist/{types-D4mog56g.d.ts.map → types-yLK8gXE9.d.ts.map} +1 -1
  286. package/dist/verdict-BndeTAh_.js +61 -0
  287. package/dist/verdict-BndeTAh_.js.map +1 -0
  288. package/dist/verdict-E4eRNf7-.d.ts +392 -0
  289. package/dist/verdict-E4eRNf7-.d.ts.map +1 -0
  290. package/dist/{verdict-cache-BCcOh0kF.js → verdict-cache-mZf5FEiY.js} +3 -55
  291. package/dist/{verdict-cache-BCcOh0kF.js.map → verdict-cache-mZf5FEiY.js.map} +1 -1
  292. package/dist/wire/index.d.ts +3 -3
  293. package/dist/wire/index.d.ts.map +1 -1
  294. package/dist/wire/index.js +1 -1
  295. package/docs/charter.md +3 -3
  296. package/docs/control-runtime.md +3 -42
  297. package/docs/experiment.md +0 -1
  298. package/docs/feature-guide.md +2 -2
  299. package/docs/trace-repair-grader.md +1 -0
  300. package/docs/trajectory-replay.md +1 -0
  301. package/docs/verdicts.md +43 -0
  302. package/docs/verification-strategies.md +3 -2
  303. package/package.json +6 -11
  304. package/dist/agent-profile-DPi7IZg7.d.ts.map +0 -1
  305. package/dist/analyze-runs-C30yljDJ.js.map +0 -1
  306. package/dist/baseline-CavEbRyH.d.ts +0 -136
  307. package/dist/baseline-CavEbRyH.d.ts.map +0 -1
  308. package/dist/benchmark-command-BteMFN62.js.map +0 -1
  309. package/dist/benchmarks-Dzs8CKb1.js +0 -755
  310. package/dist/benchmarks-Dzs8CKb1.js.map +0 -1
  311. package/dist/campaign-C2TTzQII.js.map +0 -1
  312. package/dist/completion-verifier-DJA5BhPb.d.ts +0 -414
  313. package/dist/completion-verifier-DJA5BhPb.d.ts.map +0 -1
  314. package/dist/control.d.ts +0 -3
  315. package/dist/control.js +0 -2
  316. package/dist/dataset-C8xaLXdY.d.ts.map +0 -1
  317. package/dist/default-registry-Bf8Woqmq.d.ts.map +0 -1
  318. package/dist/default-registry-BmktKy8r.js.map +0 -1
  319. package/dist/engine-3hL-XqwJ.d.ts.map +0 -1
  320. package/dist/experiment-tracker-CnRICnMl.js +0 -500
  321. package/dist/experiment-tracker-CnRICnMl.js.map +0 -1
  322. package/dist/experiment-tracker-IMntXr6J.d.ts.map +0 -1
  323. package/dist/extract-usage-CdZdoj1s.js.map +0 -1
  324. package/dist/failure-cluster-CqcvCcdR.d.ts.map +0 -1
  325. package/dist/feedback-trajectory-CacHpxsp.d.ts.map +0 -1
  326. package/dist/index-BZ3-y4YL.d.ts +0 -391
  327. package/dist/index-BZ3-y4YL.d.ts.map +0 -1
  328. package/dist/index-CQTZ-4XN.d.ts.map +0 -1
  329. package/dist/index-DPPGNJ_R.d.ts.map +0 -1
  330. package/dist/index-YE4KdKbO2.d.ts +0 -335
  331. package/dist/index-YE4KdKbO2.d.ts.map +0 -1
  332. package/dist/paired-arms-iZ08VFMN.js +0 -260
  333. package/dist/paired-arms-iZ08VFMN.js.map +0 -1
  334. package/dist/paired-promotion-decision-B6zJ3gYM.d.ts +0 -114
  335. package/dist/paired-promotion-decision-B6zJ3gYM.d.ts.map +0 -1
  336. package/dist/prime-protocol-BfSalTfR.js.map +0 -1
  337. package/dist/promotion-policy-Ckjhzg_4.d.ts +0 -289
  338. package/dist/promotion-policy-Ckjhzg_4.d.ts.map +0 -1
  339. package/dist/promotion-policy-CrLrmys8.js.map +0 -1
  340. package/dist/proposal-findings-2GIUo1et.js.map +0 -1
  341. package/dist/propose-review-control-dSNPjFUH.js +0 -1458
  342. package/dist/propose-review-control-dSNPjFUH.js.map +0 -1
  343. package/dist/release-report-BUYmoKo2.js.map +0 -1
  344. package/dist/release-report-DA2BCu5a.d.ts.map +0 -1
  345. package/dist/replay-CohS93nE.js +0 -1859
  346. package/dist/replay-CohS93nE.js.map +0 -1
  347. package/dist/replay-DbhZ4Ked.d.ts +0 -834
  348. package/dist/replay-DbhZ4Ked.d.ts.map +0 -1
  349. package/dist/reward-hacking-BDToousL.js.map +0 -1
  350. package/dist/run-evidence-Dhi3C81V.d.ts +0 -225
  351. package/dist/run-evidence-Dhi3C81V.d.ts.map +0 -1
  352. package/dist/schema-Cef2cFmb2.d.ts.map +0 -1
  353. package/dist/semantic-concept-judge-D1z-KepS.js +0 -767
  354. package/dist/semantic-concept-judge-D1z-KepS.js.map +0 -1
  355. package/dist/series-convergence-ofsqPWhs.d.ts +0 -35
  356. package/dist/series-convergence-ofsqPWhs.d.ts.map +0 -1
  357. package/dist/server-iu0ede49.js.map +0 -1
  358. package/dist/single-run-lock-DFWHEB09.js.map +0 -1
  359. package/dist/skill-usage-CO9OLRBx.d.ts.map +0 -1
  360. package/dist/skillopt-optimization-method-CQdVeM8k.js.map +0 -1
  361. package/dist/skillopt-optimization-method-USDKhxSA.d.ts.map +0 -1
  362. package/dist/statistical-heldout-TQ-4CYiN.d.ts.map +0 -1
  363. package/dist/statistics-ByxzSiOM.js +0 -2212
  364. package/dist/statistics-ByxzSiOM.js.map +0 -1
  365. package/dist/statistics-D6Uebe_4.d.ts +0 -968
  366. package/dist/statistics-D6Uebe_4.d.ts.map +0 -1
  367. package/dist/supervisor-run-D_sokXcO.js +0 -1690
  368. package/dist/test-graded-scenario-D1TaI2va.d.ts +0 -141
  369. package/dist/test-graded-scenario-D1TaI2va.d.ts.map +0 -1
  370. package/dist/test-graded-scenario-JHcKQNpq.js +0 -318
  371. package/dist/test-graded-scenario-JHcKQNpq.js.map +0 -1
  372. package/dist/tool-groups-CZPGGlHf.d.ts.map +0 -1
  373. package/dist/tool-use-metrics-DEGMKycK.js +0 -370
  374. package/dist/tool-use-metrics-DEGMKycK.js.map +0 -1
  375. package/dist/types-D216SgwM.d.ts.map +0 -1
  376. package/dist/usage-receipt-t7vAzCRQ.js.map +0 -1
  377. package/dist/verdict-DExhxfgR.d.ts +0 -201
  378. package/dist/verdict-DExhxfgR.d.ts.map +0 -1
@@ -0,0 +1,1045 @@
1
+ import { s as ValidationError } from "./errors-Dngq5h35.js";
2
+ import { i as makeRng, l as lnGamma, o as symmetricTwoSampleSeed, r as binomialSignTwoSided, s as zQuantile, t as assertFiniteSample, u as regularizedIncompleteBeta } from "./internal-BDHPCnjk.js";
3
+ import { r as pairedBootstrap } from "./paired-tests-BHIhYVdu.js";
4
+ //#region src/statistics/paired-binary.ts
5
+ /**
6
+ * Wilson score interval for a binomial proportion. Correct at small n and near
7
+ * 0/1, where the normal (Wald) approximation produces bounds outside [0, 1] and
8
+ * understates coverage. Use this for any pass-rate / hit-rate / realness-rate
9
+ * CI — the continuous `confidenceInterval` assumes the wrong distribution for a
10
+ * proportion. `n = 0 ⇒ {0, 0, 0}`.
11
+ */
12
+ function wilson(successes, n, confidence = .95) {
13
+ if (n <= 0) return {
14
+ estimate: 0,
15
+ lower: 0,
16
+ upper: 0
17
+ };
18
+ if (successes < 0 || successes > n) throw new Error(`wilson: successes (${successes}) must be in [0, ${n}]`);
19
+ const z = zQuantile(1 - (1 - confidence) / 2);
20
+ const p = successes / n;
21
+ const z2 = z * z;
22
+ const denom = 1 + z2 / n;
23
+ const center = (p + z2 / (2 * n)) / denom;
24
+ const half = z * Math.sqrt((p * (1 - p) + z2 / (4 * n)) / n) / denom;
25
+ return {
26
+ estimate: p,
27
+ lower: Math.max(0, center - half),
28
+ upper: Math.min(1, center + half)
29
+ };
30
+ }
31
+ /**
32
+ * Are these per-item outcomes binary (every value exactly 0 or 1)?
33
+ *
34
+ * The discriminator a promotion gate needs before choosing a paired statistic.
35
+ * On binary outcomes the paired delta vector lives in {-1, 0, +1} and is
36
+ * normally dominated by zeros (both arms solve, or both arms miss, most items),
37
+ * so its MEDIAN is pinned at exactly 0 no matter how large the real shift in
38
+ * success rate is — and a bootstrap CI on that median collapses to [0, 0].
39
+ * A gate keying on `ci.low > threshold` is then structurally unable to see
40
+ * either a gain or a regression. Detect this shape and switch to the
41
+ * paired-binary estimators ({@link mcnemar}, {@link pairedRiskDifference})
42
+ * instead of silently answering "no" forever.
43
+ *
44
+ * Empty input is NOT binary: there is no evidence of the outcome's shape, and
45
+ * defaulting an empty vector into the binary branch would pick a statistic on
46
+ * no data at all.
47
+ *
48
+ * NOT the right discriminator for a gate. It recognises the literal {0, 1}
49
+ * encoding and nothing else, so a pass/fail dimension emitted on 0-100 — which
50
+ * judges in this codebase do routinely — reads as non-binary, and a single
51
+ * partial-credit score in an otherwise pass/fail vector flips it to false while
52
+ * leaving the median just as blind. Gates want {@link pairedBinaryScale} (any
53
+ * two-point encoding). This predicate remains for callers that specifically
54
+ * mean "literally 0/1".
55
+ */
56
+ function isBinaryOutcomeVector(values) {
57
+ if (values.length === 0) return false;
58
+ for (let i = 0; i < values.length; i++) {
59
+ const v = values[i];
60
+ if (v !== 0 && v !== 1) return false;
61
+ }
62
+ return true;
63
+ }
64
+ /**
65
+ * McNemar's test for paired binary outcomes — the correct significance test for
66
+ * "does treatment change the success rate vs control on the SAME items". Only
67
+ * discordant pairs (one arm right, the other wrong) carry information; concordant
68
+ * pairs are uninformative, so a paired t-test / two-proportion z-test on the raw
69
+ * rates is wrong here. The p-value is exact: under H0 the b "treatment-wins" are
70
+ * Binomial(b + c, 0.5), so the two-sided p is the doubled binomial tail — correct
71
+ * at the small discordant counts typical of eval runs (no continuity-corrected
72
+ * chi-square approximation needed, though it is returned as `statistic` for
73
+ * reference). Inputs are paired 0/1 (or boolean) arrays, control first to match
74
+ * the module's (before, after) convention. Throws on unequal lengths.
75
+ */
76
+ function mcnemar(control, treatment) {
77
+ if (control.length !== treatment.length) throw new Error(`mcnemar: unequal sample sizes (${control.length} vs ${treatment.length})`);
78
+ const n = control.length;
79
+ let b = 0;
80
+ let c = 0;
81
+ for (let i = 0; i < n; i++) {
82
+ const ctrl = control[i] ? 1 : 0;
83
+ const treat = treatment[i] ? 1 : 0;
84
+ if (treat === 1 && ctrl === 0) b++;
85
+ else if (treat === 0 && ctrl === 1) c++;
86
+ }
87
+ const nDiscordant = b + c;
88
+ const statistic = nDiscordant === 0 ? 0 : (Math.abs(b - c) - 1) ** 2 / nDiscordant;
89
+ return {
90
+ n,
91
+ nDiscordant,
92
+ b,
93
+ c,
94
+ statistic,
95
+ pValue: binomialSignTwoSided(b, c)
96
+ };
97
+ }
98
+ /**
99
+ * Paired risk difference (the effect-size companion to {@link mcnemar}): the
100
+ * change in success rate p(treatment) − p(control) on matched items, which for
101
+ * paired binary data equals (b − c) / n. The CI uses the paired variance from
102
+ * the discordant counts, not the independent-samples formula (which overstates
103
+ * the interval by ignoring the pairing). Inputs are paired 0/1 (or boolean)
104
+ * arrays, control first. Throws on unequal lengths.
105
+ *
106
+ * REPORTING ONLY — do NOT decide a promotion on this interval. The CI is a Wald
107
+ * normal approximation, which badly UNDERCOVERS when only a handful of pairs are
108
+ * discordant: at n = 3 with b = 2, c = 0 it returns [0.133, 1.000], excluding 0,
109
+ * while McNemar's exact test on the same data gives p = 0.50. A gate keying on
110
+ * `lower > 0` would promote noise. Use {@link pairedRiskDifferenceExact}, whose
111
+ * interval is dual to the exact test by construction, for any decision.
112
+ */
113
+ function pairedRiskDifference(control, treatment, confidence = .95) {
114
+ if (control.length !== treatment.length) throw new Error(`pairedRiskDifference: unequal sample sizes (${control.length} vs ${treatment.length})`);
115
+ const n = control.length;
116
+ if (n === 0) return {
117
+ n: 0,
118
+ b: 0,
119
+ c: 0,
120
+ riskDifference: 0,
121
+ lower: 0,
122
+ upper: 0,
123
+ confidence
124
+ };
125
+ let b = 0;
126
+ let c = 0;
127
+ for (let i = 0; i < n; i++) {
128
+ const ctrl = control[i] ? 1 : 0;
129
+ const treat = treatment[i] ? 1 : 0;
130
+ if (treat === 1 && ctrl === 0) b++;
131
+ else if (treat === 0 && ctrl === 1) c++;
132
+ }
133
+ const rd = (b - c) / n;
134
+ const variance = (b + c - (b - c) ** 2 / n) / (n * n);
135
+ const half = zQuantile(1 - (1 - confidence) / 2) * Math.sqrt(Math.max(0, variance));
136
+ return {
137
+ n,
138
+ b,
139
+ c,
140
+ riskDifference: rd,
141
+ lower: Math.max(-1, rd - half),
142
+ upper: Math.min(1, rd + half),
143
+ confidence
144
+ };
145
+ }
146
+ /**
147
+ * Paired risk difference with the EXACT CONDITIONAL interval — the estimator a
148
+ * promotion gate may decide on.
149
+ *
150
+ * Conditional on the number of discordant pairs m = b + c, the treatment-win
151
+ * count b is Binomial(m, π) with π = P(treatment wins | discordant), and the
152
+ * risk difference is an exact reparameterisation: RD = (2π − 1)·m/n. So a
153
+ * Clopper-Pearson exact interval for π maps straight onto RD. This buys the
154
+ * property the Wald interval in {@link pairedRiskDifference} does not have:
155
+ *
156
+ * **`lower > 0` ⟺ McNemar's exact test rejects at α = 1 − confidence.**
157
+ *
158
+ * Clopper-Pearson excludes π = 0.5 exactly when the two-sided exact binomial
159
+ * test of π = 0.5 rejects, and that test IS {@link mcnemar}'s p-value — so the
160
+ * interval and the test can never disagree, and a gate keyed on `lower` cannot
161
+ * promote what the exact test refuses. The exact p is returned in the same
162
+ * object so the two are impossible to compute apart.
163
+ *
164
+ * The interval is conservative (exact intervals over-cover; conditioning on m
165
+ * discards the concordant pairs' information about m itself). That is the
166
+ * correct direction for a promotion gate: it refuses more often, never less.
167
+ *
168
+ * With m = 0 there are no discordant pairs and π is not identified: the result
169
+ * is the degenerate [0, 0] with p = 1. That is NOT evidence of equivalence —
170
+ * callers must treat a zero-width interval as "cannot decide", not as "no
171
+ * difference". Inputs are paired 0/1 (or boolean) arrays, control first.
172
+ * Throws on unequal lengths.
173
+ */
174
+ function pairedRiskDifferenceExact(control, treatment, confidence = .95) {
175
+ if (control.length !== treatment.length) throw new Error(`pairedRiskDifferenceExact: unequal sample sizes (${control.length} vs ${treatment.length})`);
176
+ if (confidence <= 0 || confidence >= 1) throw new Error(`pairedRiskDifferenceExact: confidence must be in (0,1), got ${confidence}`);
177
+ const n = control.length;
178
+ if (n === 0) return {
179
+ n: 0,
180
+ b: 0,
181
+ c: 0,
182
+ nDiscordant: 0,
183
+ riskDifference: 0,
184
+ lower: 0,
185
+ upper: 0,
186
+ confidence,
187
+ pValue: 1
188
+ };
189
+ let b = 0;
190
+ let c = 0;
191
+ for (let i = 0; i < n; i++) {
192
+ const ctrl = control[i] ? 1 : 0;
193
+ const treat = treatment[i] ? 1 : 0;
194
+ if (treat === 1 && ctrl === 0) b++;
195
+ else if (treat === 0 && ctrl === 1) c++;
196
+ }
197
+ const m = b + c;
198
+ const riskDifference = (b - c) / n;
199
+ const pValue = binomialSignTwoSided(b, c);
200
+ if (m === 0) return {
201
+ n,
202
+ b,
203
+ c,
204
+ nDiscordant: 0,
205
+ riskDifference: 0,
206
+ lower: 0,
207
+ upper: 0,
208
+ confidence,
209
+ pValue
210
+ };
211
+ const alpha = 1 - confidence;
212
+ const piLow = b === 0 ? 0 : betaQuantile(alpha / 2, b, m - b + 1);
213
+ const piHigh = b === m ? 1 : betaQuantile(1 - alpha / 2, b + 1, m - b);
214
+ const scale = m / n;
215
+ return {
216
+ n,
217
+ b,
218
+ c,
219
+ nDiscordant: m,
220
+ riskDifference,
221
+ lower: Math.max(-1, (2 * piLow - 1) * scale),
222
+ upper: Math.min(1, (2 * piHigh - 1) * scale),
223
+ confidence,
224
+ pValue
225
+ };
226
+ }
227
+ /** Inverse regularized incomplete beta by bisection on
228
+ * {@link regularizedIncompleteBeta}, which is monotone increasing in x. 80
229
+ * halvings of [0,1] resolve to ~8e-25, far past the continued fraction's own
230
+ * 3e-15 tolerance, so the quantile is as exact as the CDF it inverts. */
231
+ function betaQuantile(p, a, b) {
232
+ if (p <= 0) return 0;
233
+ if (p >= 1) return 1;
234
+ let lo = 0;
235
+ let hi = 1;
236
+ for (let i = 0; i < 80; i++) {
237
+ const mid = (lo + hi) / 2;
238
+ if (regularizedIncompleteBeta(mid, a, b) < p) lo = mid;
239
+ else hi = mid;
240
+ }
241
+ return (lo + hi) / 2;
242
+ }
243
+ /**
244
+ * Constrained MLE of q = P(treatment loses) under the hypothesis RD = `delta`.
245
+ *
246
+ * Profiling the two concordant cells out of the multinomial leaves
247
+ * `L(q) = b·log(q+delta) + c·log(q) + e·log(1 − 2q − delta)` with `e = n − b − c`,
248
+ * whose stationary point is the positive root of
249
+ * `2n·q² − [(b + c) − delta·(b + 3c + 2e)]·q − c·delta·(1 − delta) = 0`.
250
+ * At `delta = 0` this returns `(b + c) / 2n`, the familiar null.
251
+ */
252
+ function constrainedLossRate(b, c, n, delta) {
253
+ const e = n - b - c;
254
+ const quadratic = 2 * n;
255
+ const linear = -(b + c - delta * (b + 3 * c + 2 * e));
256
+ const constant = -c * delta * (1 - delta);
257
+ const discriminant = linear * linear - 4 * quadratic * constant;
258
+ const root = discriminant > 0 ? Math.sqrt(discriminant) : 0;
259
+ const q = (-linear + root) / (2 * quadratic);
260
+ return Math.min(Math.max(q, Math.max(0, -delta)), Math.max(0, (1 - delta) / 2));
261
+ }
262
+ /** Tango's score statistic for H0: RD = `delta`. `Var(b − c) = n·(2q + delta −
263
+ * delta²)` under that hypothesis, evaluated at the constrained MLE of q. */
264
+ function tangoScore(b, c, n, delta) {
265
+ const numerator = b - c - n * delta;
266
+ const variance = n * (2 * constrainedLossRate(b, c, n, delta) + delta * (1 - delta));
267
+ if (!(variance > 0)) {
268
+ if (numerator === 0) return 0;
269
+ return numerator > 0 ? Number.POSITIVE_INFINITY : Number.NEGATIVE_INFINITY;
270
+ }
271
+ return numerator / Math.sqrt(variance);
272
+ }
273
+ /**
274
+ * Paired risk difference with TANGO'S (1998) SCORE INTERVAL — the estimator a
275
+ * promotion gate may decide on **at a nonzero margin**.
276
+ *
277
+ * {@link pairedRiskDifferenceExact} conditions on the observed discordant count
278
+ * `m = b + c`, builds a Clopper-Pearson interval for the win share among those
279
+ * `m` pairs, and multiplies by the observed `m/n`. That is exact for testing
280
+ * RD = 0 — it is dual to McNemar — but it is NOT a confidence interval for the
281
+ * population risk difference at a nonzero margin, because the sampling
282
+ * variability of `m/n` itself is discarded. The gap is not academic: with the
283
+ * production caller's `pairedDeltaThreshold: -0.05`, a process whose true risk
284
+ * difference sits exactly on that margin clears a nominal-95 % `lower > margin`
285
+ * check 24.75 % of the time at n = 40 and 43.95 % at n = 76 (2000 replicates
286
+ * each) when the conditional interval decides.
287
+ *
288
+ * Tango's interval inverts the score test of RD = delta, which estimates the
289
+ * nuisance loss rate under each hypothesised delta instead of fixing it at the
290
+ * observed value, so `m` contributes its own uncertainty. It is the method
291
+ * `ratesci::scorepairci` uses for paired risk-difference noninferiority, and it
292
+ * is not conditional, so it stays valid as the margin moves away from zero.
293
+ *
294
+ * The bounds are found by bisecting `tangoScore(delta) = ±z` — the score is
295
+ * monotone decreasing in delta, so each crossing is unique. Inputs are paired
296
+ * 0/1 (or boolean) arrays, control first. Throws on unequal lengths.
297
+ */
298
+ function pairedRiskDifferenceScore(control, treatment, confidence = .95) {
299
+ if (control.length !== treatment.length) throw new Error(`pairedRiskDifferenceScore: unequal sample sizes (${control.length} vs ${treatment.length})`);
300
+ if (confidence <= 0 || confidence >= 1) throw new Error(`pairedRiskDifferenceScore: confidence must be in (0,1), got ${confidence}`);
301
+ const n = control.length;
302
+ if (n === 0) return {
303
+ n: 0,
304
+ b: 0,
305
+ c: 0,
306
+ nDiscordant: 0,
307
+ riskDifference: 0,
308
+ lower: -1,
309
+ upper: 1,
310
+ confidence
311
+ };
312
+ let b = 0;
313
+ let c = 0;
314
+ for (let i = 0; i < n; i++) {
315
+ const ctrl = control[i] ? 1 : 0;
316
+ const treat = treatment[i] ? 1 : 0;
317
+ if (treat === 1 && ctrl === 0) b++;
318
+ else if (treat === 0 && ctrl === 1) c++;
319
+ }
320
+ const riskDifference = (b - c) / n;
321
+ const z = zQuantile(1 - (1 - confidence) / 2);
322
+ let lo = -1;
323
+ let hi = riskDifference;
324
+ for (let i = 0; i < 200; i++) {
325
+ const mid = (lo + hi) / 2;
326
+ if (tangoScore(b, c, n, mid) > z) lo = mid;
327
+ else hi = mid;
328
+ }
329
+ const lower = (lo + hi) / 2;
330
+ let ulo = riskDifference;
331
+ let uhi = 1;
332
+ for (let i = 0; i < 200; i++) {
333
+ const mid = (ulo + uhi) / 2;
334
+ if (tangoScore(b, c, n, mid) > -z) ulo = mid;
335
+ else uhi = mid;
336
+ }
337
+ const upper = (ulo + uhi) / 2;
338
+ return {
339
+ n,
340
+ b,
341
+ c,
342
+ nDiscordant: b + c,
343
+ riskDifference,
344
+ lower: Math.max(-1, lower),
345
+ upper: Math.min(1, upper),
346
+ confidence
347
+ };
348
+ }
349
+ /**
350
+ * The common positive level `s` such that EVERY value across both paired arms is
351
+ * exactly 0 or `s` — i.e. the outcome is pass/fail, whatever encoding it arrived
352
+ * in. Returns null when the outcomes are not two-point, when the two arms use
353
+ * different levels, or when no positive value was observed at all (all-zero
354
+ * arms: the level is not identified, and there is nothing to decide anyway).
355
+ *
356
+ * This is the scale-aware successor to {@link isBinaryOutcomeVector}, which only
357
+ * recognises literal {0, 1}. Judges in this codebase emit dimensions on 0-100 as
358
+ * well as [0,1] (see `detectScale` in `campaign/gates/statistical-heldout.ts`),
359
+ * so a pass/fail dimension routinely arrives as {0, 100} and a {0,1}-only test
360
+ * silently sends it down the median path that cannot see it. Any positive level
361
+ * is accepted, not just 1 and 100: for a two-point {0, s} outcome the mean paired
362
+ * delta is exactly s·(b − c)/n, so the binary estimators apply after dividing by
363
+ * s and rescaling the result back into the caller's native units.
364
+ *
365
+ * Non-finite values ⇒ null: an unusable outcome must not be classified as a
366
+ * clean pass/fail shape.
367
+ */
368
+ function pairedBinaryScale(before, after) {
369
+ let level = null;
370
+ for (const arm of [before, after]) for (let i = 0; i < arm.length; i++) {
371
+ const v = arm[i];
372
+ if (!Number.isFinite(v)) return null;
373
+ if (v === 0) continue;
374
+ if (v < 0) return null;
375
+ if (level === null) level = v;
376
+ else if (v !== level) return null;
377
+ }
378
+ return level;
379
+ }
380
+ /**
381
+ * Unbiased pass@k for code generation (Chen et al. 2021, "Evaluating Large
382
+ * Language Models Trained on Code"). Given `n` independent samples for one
383
+ * problem of which `c` pass, the probability that at least one of a random k of
384
+ * them passes is 1 − C(n−c, k) / C(n, k). Estimating pass@k as "did any of the
385
+ * first k pass" is biased high at small n; this is the variance-reduced estimator
386
+ * averaged implicitly over all k-subsets. Average the per-problem values across
387
+ * the suite for the corpus pass@k. Computed in the numerically stable product
388
+ * form. Requires 1 ≤ k ≤ n and 0 ≤ c ≤ n.
389
+ */
390
+ function passAtK(n, c, k) {
391
+ if (!Number.isInteger(n) || !Number.isInteger(c) || !Number.isInteger(k)) throw new Error(`passAtK: n, c, k must be integers (got n=${n}, c=${c}, k=${k})`);
392
+ if (k < 1 || k > n || c < 0 || c > n) throw new Error(`passAtK: require 1 ≤ k ≤ n and 0 ≤ c ≤ n (got n=${n}, c=${c}, k=${k})`);
393
+ if (n - c < k) return 1;
394
+ let prob = 1;
395
+ for (let i = n - c + 1; i <= n; i++) prob *= 1 - k / i;
396
+ return 1 - prob;
397
+ }
398
+ //#endregion
399
+ //#region src/math/normal.ts
400
+ /**
401
+ * Standard normal cumulative distribution using Abramowitz and Stegun 7.1.26.
402
+ *
403
+ * The approximation is evaluated as erf(x / sqrt(2)). Computing the negative
404
+ * tail from the complementary term avoids cancellation when x is far below 0.
405
+ * The maximum absolute CDF error is approximately 7.5e-8.
406
+ */
407
+ function normalCdf(x) {
408
+ if (x === 0) return .5;
409
+ const a1 = .254829592;
410
+ const a2 = -.284496736;
411
+ const a3 = 1.421413741;
412
+ const a4 = -1.453152027;
413
+ const a5 = 1.061405429;
414
+ const p = .3275911;
415
+ const scaled = Math.abs(x) / Math.SQRT2;
416
+ const t = 1 / (1 + p * scaled);
417
+ const complement = ((((a5 * t + a4) * t + a3) * t + a2) * t + a1) * t * Math.exp(-scaled * scaled);
418
+ return x < 0 ? complement / 2 : 1 - complement / 2;
419
+ }
420
+ //#endregion
421
+ //#region src/statistics/rank-tests.ts
422
+ /** Maximum dynamic-programming cells used by an exact two-sample rank test. */
423
+ const MANN_WHITNEY_EXACT_MAX_STATES = 8192;
424
+ /** Maximum inner-loop transitions used by an exact two-sample rank test. */
425
+ const MANN_WHITNEY_EXACT_MAX_WORK = 25e4;
426
+ /** Non-zero differences up to which the signed-rank null is enumerated exactly. */
427
+ const WILCOXON_EXACT_MAX_N = 20;
428
+ /** Resamples used when a rank test falls back to Monte Carlo permutation. */
429
+ const DEFAULT_PERMUTATIONS = 1e5;
430
+ /**
431
+ * Mann-Whitney U — two independent samples, no distributional assumption.
432
+ *
433
+ * Exact conditional (permutation) p by default when the dynamic program fits
434
+ * {@link MANN_WHITNEY_EXACT_MAX_STATES} cells and
435
+ * {@link MANN_WHITNEY_EXACT_MAX_WORK} transitions, seeded Monte Carlo
436
+ * permutation above those limits. This keeps imbalanced designs such as 1+24
437
+ * exact without admitting expensive balanced designs merely because they have
438
+ * the same total size. Throws on non-finite input and on `method:
439
+ * 'asymptotic'` where an exact answer is available. Empty input yields `p = 1,
440
+ * pFloor = 1` — no design, no attainable evidence.
441
+ */
442
+ function mannWhitneyU(a, b, opts = {}) {
443
+ assertFiniteSample("mannWhitneyU", "a", a);
444
+ assertFiniteSample("mannWhitneyU", "b", b);
445
+ const n1 = a.length;
446
+ const n2 = b.length;
447
+ if (n1 === 0 || n2 === 0) return {
448
+ u: 0,
449
+ uA: 0,
450
+ p: 1,
451
+ method: "exact",
452
+ pFloor: 1
453
+ };
454
+ const total = n1 + n2;
455
+ const combined = [...a.map((v) => ({
456
+ v,
457
+ fromA: true
458
+ })), ...b.map((v) => ({
459
+ v,
460
+ fromA: false
461
+ }))].sort((x, y) => x.v - y.v);
462
+ const { midranks, tieTerm } = midranksWithTieTerm(combined.map((entry) => entry.v));
463
+ let rankSumA = 0;
464
+ for (let k = 0; k < total; k++) if (combined[k].fromA) rankSumA += midranks[k];
465
+ const uA = rankSumA - n1 * (n1 + 1) / 2;
466
+ const u = Math.min(uA, n1 * n2 - uA);
467
+ const doubled = midranks.map((rank) => Math.round(rank * 2));
468
+ const doubledDeviation = Math.abs(2 * uA - n1 * n2);
469
+ const selectedN = Math.min(n1, n2);
470
+ const otherN = total - selectedN;
471
+ const exactCost = exactTwoSampleCost(doubled, selectedN);
472
+ const designFloor = exactTwoSampleFloor(doubled, selectedN);
473
+ const method = selectRankTestMethod("mannWhitneyU", opts.method ?? "auto", `n1=${n1}, n2=${n2}`, exactCost.states <= 8192 && exactCost.work <= 25e4, designFloor, `${MANN_WHITNEY_EXACT_MAX_STATES.toLocaleString("en-US")} states and ${MANN_WHITNEY_EXACT_MAX_WORK.toLocaleString("en-US")} transitions`);
474
+ if (method === "exact") {
475
+ const { p, pFloor } = exactTwoSampleP(doubled, selectedN, otherN, doubledDeviation);
476
+ return {
477
+ u,
478
+ uA,
479
+ p,
480
+ method,
481
+ pFloor
482
+ };
483
+ }
484
+ if (method === "asymptotic") return {
485
+ u,
486
+ uA,
487
+ p: asymptoticTwoSidedP(doubledDeviation / 2, twoSampleSigma(n1, n2, total, tieTerm)),
488
+ method,
489
+ pFloor: designFloor
490
+ };
491
+ const permutations = resolvePermutations("mannWhitneyU", opts.permutations);
492
+ const rng = opts.seed === void 0 ? makeRng(symmetricTwoSampleSeed(a, b)) : makeRng(opts.seed);
493
+ let atLeastAsExtreme = 0;
494
+ const pool = [...doubled];
495
+ for (let iteration = 0; iteration < permutations; iteration++) {
496
+ let doubledRankSum = 0;
497
+ for (let k = 0; k < selectedN; k++) {
498
+ const pick = k + Math.floor(rng() * (total - k));
499
+ const swapped = pool[pick];
500
+ pool[pick] = pool[k];
501
+ pool[k] = swapped;
502
+ doubledRankSum += swapped;
503
+ }
504
+ if (Math.abs(doubledRankSum - selectedN * (selectedN + 1) - selectedN * otherN) >= doubledDeviation) atLeastAsExtreme++;
505
+ }
506
+ const pFloor = Math.max(1 / (permutations + 1), designFloor);
507
+ return {
508
+ u,
509
+ uA,
510
+ p: Math.max((1 + atLeastAsExtreme) / (permutations + 1), pFloor),
511
+ method,
512
+ pFloor
513
+ };
514
+ }
515
+ /**
516
+ * Wilcoxon signed-rank — paired, no distributional assumption on the deltas.
517
+ *
518
+ * Exact conditional (sign-flip) p by default at `n ≤
519
+ * {@link WILCOXON_EXACT_MAX_N}` non-zero differences, seeded Monte Carlo
520
+ * permutation above it. Throws on non-finite input and on `method:
521
+ * 'asymptotic'` where an exact answer is available.
522
+ *
523
+ * `n` is the count of NON-ZERO differences: exact ties are dropped before
524
+ * ranking, so a run of tied pairs shrinks the design and raises `pFloor`.
525
+ * All-tied input yields `p = 1, pFloor = 1` — no attainable evidence, which
526
+ * `pFloor` states rather than leaving `p = 1` to be read as a measured null.
527
+ */
528
+ function wilcoxonSignedRank(before, after, opts = {}) {
529
+ if (before.length !== after.length) throw new ValidationError(`wilcoxonSignedRank: unequal sample sizes (${before.length} vs ${after.length})`);
530
+ assertFiniteSample("wilcoxonSignedRank", "before", before);
531
+ assertFiniteSample("wilcoxonSignedRank", "after", after);
532
+ const diffs = before.map((b, i) => after[i] - b).filter((d) => d !== 0);
533
+ const n = diffs.length;
534
+ if (n === 0) return {
535
+ w: 0,
536
+ p: 1,
537
+ method: "exact",
538
+ pFloor: 1,
539
+ nNonZero: 0
540
+ };
541
+ const order = diffs.map((d, i) => ({
542
+ abs: Math.abs(d),
543
+ i
544
+ })).sort((x, y) => x.abs - y.abs);
545
+ const { midranks, tieTerm } = midranksWithTieTerm(order.map((entry) => entry.abs));
546
+ const ranks = new Array(n);
547
+ for (let k = 0; k < n; k++) ranks[order[k].i] = midranks[k];
548
+ let wPlus = 0;
549
+ for (let k = 0; k < n; k++) if (diffs[k] > 0) wPlus += ranks[k];
550
+ const doubled = midranks.map((rank) => Math.round(rank * 2));
551
+ const doubledDeviation = Math.abs(2 * wPlus - n * (n + 1) / 2);
552
+ const designFloor = Math.min(1, 2 ** (1 - n));
553
+ const method = selectRankTestMethod("wilcoxonSignedRank", opts.method ?? "auto", `n=${n} non-zero differences`, n <= 20, designFloor, `20 non-zero differences`);
554
+ if (method === "exact") {
555
+ const { p, pFloor } = exactSignedRankP(doubled, doubledDeviation);
556
+ return {
557
+ w: wPlus,
558
+ p,
559
+ method,
560
+ pFloor,
561
+ nNonZero: n
562
+ };
563
+ }
564
+ if (method === "asymptotic") {
565
+ const variance = n * (n + 1) * (2 * n + 1) / 24 - tieTerm / 48;
566
+ return {
567
+ w: wPlus,
568
+ p: asymptoticTwoSidedP(doubledDeviation / 2, Math.sqrt(variance)),
569
+ method,
570
+ pFloor: designFloor,
571
+ nNonZero: n
572
+ };
573
+ }
574
+ const permutations = resolvePermutations("wilcoxonSignedRank", opts.permutations);
575
+ const rng = makeRng(opts.seed, before, after);
576
+ const doubledCentre = n * (n + 1) / 2;
577
+ let atLeastAsExtreme = 0;
578
+ for (let iteration = 0; iteration < permutations; iteration++) {
579
+ let doubledWPlus = 0;
580
+ for (let k = 0; k < n; k++) if (rng() < .5) doubledWPlus += doubled[k];
581
+ if (Math.abs(doubledWPlus - doubledCentre) >= doubledDeviation) atLeastAsExtreme++;
582
+ }
583
+ return {
584
+ w: wPlus,
585
+ p: (1 + atLeastAsExtreme) / (permutations + 1),
586
+ method,
587
+ pFloor: Math.max(1 / (permutations + 1), designFloor),
588
+ nNonZero: n
589
+ };
590
+ }
591
+ /**
592
+ * Average ranks over an ASCENDING-sorted array, plus `Σ(t³ − t)` over tie
593
+ * groups of size `t` — the correction term both asymptotic rank-test variances
594
+ * need.
595
+ */
596
+ function midranksWithTieTerm(sorted) {
597
+ const midranks = new Array(sorted.length);
598
+ let tieTerm = 0;
599
+ let i = 0;
600
+ while (i < sorted.length) {
601
+ let j = i;
602
+ while (j < sorted.length && sorted[j] === sorted[i]) j++;
603
+ const average = (i + 1 + j) / 2;
604
+ for (let k = i; k < j; k++) midranks[k] = average;
605
+ const groupSize = j - i;
606
+ if (groupSize > 1) tieTerm += groupSize ** 3 - groupSize;
607
+ i = j;
608
+ }
609
+ return {
610
+ midranks,
611
+ tieTerm
612
+ };
613
+ }
614
+ function selectRankTestMethod(fn, request, design, exactFeasible, designFloor, threshold) {
615
+ if (request === "auto") return exactFeasible ? "exact" : "permutation";
616
+ if (request === "exact") {
617
+ if (exactFeasible) return "exact";
618
+ throw new ValidationError(`${fn}: method 'exact' is out of range at ${design} — enumeration is bounded by ${threshold}. Use 'auto' for the seeded Monte Carlo permutation, which converges to the same answer.`);
619
+ }
620
+ if (exactFeasible) throw new ValidationError(`${fn}: method 'asymptotic' is refused at ${design} — the exact p-grid at this design starts at ${formatProbability(designFloor)}, so an asymptotic p below it describes no attainable outcome. Use method 'exact' (the default) or add repetitions past ${threshold}.`);
621
+ return "asymptotic";
622
+ }
623
+ function resolvePermutations(fn, permutations) {
624
+ if (permutations === void 0) return DEFAULT_PERMUTATIONS;
625
+ if (!Number.isInteger(permutations) || permutations < 1) throw new ValidationError(`${fn}: permutations must be a positive integer, got ${permutations}`);
626
+ return permutations;
627
+ }
628
+ /** Two-sided normal-approximation tail with the continuity correction. */
629
+ function asymptoticTwoSidedP(deviation, sigma) {
630
+ if (!(sigma > 0)) return 1;
631
+ return Math.min(1, 2 * (1 - normalCdf(Math.max(0, deviation - .5) / sigma)));
632
+ }
633
+ /** SD of U under the permutation null, corrected for the realised ties. The
634
+ * tie term reduces (N+1) and reaches it exactly when every value is tied, so
635
+ * the variance floors at 0 rather than going negative. */
636
+ function twoSampleSigma(n1, n2, total, tieTerm) {
637
+ if (total < 2) return 0;
638
+ const variance = n1 * n2 / 12 * (total + 1 - tieTerm / (total * (total - 1)));
639
+ return Math.sqrt(Math.max(0, variance));
640
+ }
641
+ function logChoose(n, k) {
642
+ return lnGamma(n + 1) - lnGamma(k + 1) - lnGamma(n - k + 1);
643
+ }
644
+ function formatProbability(value) {
645
+ return value >= 1e-4 || value === 0 ? value.toFixed(4) : value.toExponential(3);
646
+ }
647
+ /**
648
+ * Exact DP allocation and loop count for this observed rank vector.
649
+ *
650
+ * The smaller arm is sufficient because selecting its complement produces the
651
+ * same two-sided U deviation while using fewer rows in the state table.
652
+ */
653
+ function exactTwoSampleCost(doubledRanks, selectedN) {
654
+ const maxSum = doubledRanks.reduce((sum, rank) => sum + rank, 0);
655
+ let work = 0;
656
+ for (let placed = 0; placed < doubledRanks.length; placed++) work += Math.min(selectedN, placed + 1) * (maxSum - doubledRanks[placed] + 1);
657
+ return {
658
+ states: (selectedN + 1) * (maxSum + 1),
659
+ work
660
+ };
661
+ }
662
+ /**
663
+ * Smallest attainable two-sided p under the observed ties.
664
+ *
665
+ * Only subsets with the minimum or maximum rank sum can attain the largest
666
+ * deviation. Their multiplicity is the number of ways to choose within the
667
+ * tie group at each boundary, so this calculation is exact without allocating
668
+ * the full null distribution.
669
+ */
670
+ function exactTwoSampleFloor(doubledRanks, selectedN) {
671
+ const total = doubledRanks.length;
672
+ const otherN = total - selectedN;
673
+ const minimumSum = doubledRanks.slice(0, selectedN).reduce((sum, rank) => sum + rank, 0);
674
+ const maximumSum = doubledRanks.slice(total - selectedN).reduce((sum, rank) => sum + rank, 0);
675
+ if (minimumSum === maximumSum) return 1;
676
+ const centre = selectedN * (selectedN + 1) + selectedN * otherN;
677
+ const minimumDeviation = Math.abs(minimumSum - centre);
678
+ const maximumDeviation = Math.abs(maximumSum - centre);
679
+ const totalLogWays = logChoose(total, selectedN);
680
+ const minimumMass = Math.exp(logExtremeSubsetWays(doubledRanks, selectedN, "minimum") - totalLogWays);
681
+ const maximumMass = Math.exp(logExtremeSubsetWays(doubledRanks, selectedN, "maximum") - totalLogWays);
682
+ if (minimumDeviation > maximumDeviation) return minimumMass;
683
+ if (maximumDeviation > minimumDeviation) return maximumMass;
684
+ return Math.min(1, minimumMass + maximumMass);
685
+ }
686
+ function logExtremeSubsetWays(sortedRanks, selectedN, side) {
687
+ const boundaryIndex = side === "minimum" ? selectedN - 1 : sortedRanks.length - selectedN;
688
+ const boundary = sortedRanks[boundaryIndex];
689
+ let first = boundaryIndex;
690
+ let afterLast = boundaryIndex + 1;
691
+ while (first > 0 && sortedRanks[first - 1] === boundary) first--;
692
+ while (afterLast < sortedRanks.length && sortedRanks[afterLast] === boundary) afterLast++;
693
+ return logChoose(afterLast - first, selectedN - (side === "minimum" ? first : sortedRanks.length - afterLast));
694
+ }
695
+ /**
696
+ * Exact conditional two-sided p for the two-sample rank test.
697
+ *
698
+ * Convolves the observed doubled midranks into the null distribution of group
699
+ * a's rank sum over every `C(n₁+n₂, n₁)` split — identical to enumerating the
700
+ * splits, but `O(N·n₁·ΣR)` instead of `O(C(N,n₁)·n₁n₂)`. Conditioning on the
701
+ * realised multiset makes the tie handling exact rather than a correction.
702
+ *
703
+ * The null is symmetric about `n₁n₂/2` (negating every value maps `U → n₁n₂ −
704
+ * U` and permutes the split set onto itself), so the two-sided p is the mass
705
+ * at least as far from the centre as the observation.
706
+ */
707
+ function exactTwoSampleP(doubledRanks, n1, n2, doubledDeviation) {
708
+ const maxSum = doubledRanks.reduce((sum, rank) => sum + rank, 0);
709
+ const width = maxSum + 1;
710
+ const ways = Array.from({ length: n1 + 1 }, () => new Float64Array(width));
711
+ ways[0][0] = 1;
712
+ let placed = 0;
713
+ for (const rank of doubledRanks) {
714
+ for (let k = Math.min(n1, placed + 1); k >= 1; k--) {
715
+ const from = ways[k - 1];
716
+ const into = ways[k];
717
+ for (let sum = maxSum - rank; sum >= 0; sum--) {
718
+ const count = from[sum];
719
+ if (count !== 0) into[sum + rank] += count;
720
+ }
721
+ }
722
+ placed++;
723
+ }
724
+ const shift = n1 * (n1 + 1) + n1 * n2;
725
+ const chosen = ways[n1];
726
+ let totalWays = 0;
727
+ let extremeWays = 0;
728
+ let tailWays = 0;
729
+ let maxDeviation = -1;
730
+ for (let sum = 0; sum < width; sum++) {
731
+ const count = chosen[sum];
732
+ if (count === 0) continue;
733
+ totalWays += count;
734
+ const deviation = Math.abs(sum - shift);
735
+ if (deviation >= doubledDeviation) tailWays += count;
736
+ if (deviation > maxDeviation) {
737
+ maxDeviation = deviation;
738
+ extremeWays = count;
739
+ } else if (deviation === maxDeviation) extremeWays += count;
740
+ }
741
+ return {
742
+ p: tailWays / totalWays,
743
+ pFloor: extremeWays / totalWays
744
+ };
745
+ }
746
+ /**
747
+ * Exact conditional two-sided p for the paired signed-rank test.
748
+ *
749
+ * Convolves the observed doubled absolute midranks over all `2ⁿ` sign
750
+ * assignments in `O(n·ΣR)`. Probabilities rather than counts keep `2ⁿ` off the
751
+ * arithmetic. The null is symmetric about `n(n+1)/4`.
752
+ */
753
+ function exactSignedRankP(doubledRanks, doubledDeviation) {
754
+ const maxSum = doubledRanks.reduce((sum, rank) => sum + rank, 0);
755
+ const width = maxSum + 1;
756
+ let mass = new Float64Array(width);
757
+ mass[0] = 1;
758
+ for (const rank of doubledRanks) {
759
+ const next = new Float64Array(width);
760
+ for (let sum = 0; sum < width; sum++) {
761
+ const probability = mass[sum];
762
+ if (probability === 0) continue;
763
+ next[sum] += probability * .5;
764
+ next[sum + rank] += probability * .5;
765
+ }
766
+ mass = next;
767
+ }
768
+ const centre = maxSum / 2;
769
+ let tail = 0;
770
+ let extreme = 0;
771
+ let maxDeviation = -1;
772
+ for (let sum = 0; sum < width; sum++) {
773
+ const probability = mass[sum];
774
+ if (probability === 0) continue;
775
+ const deviation = Math.abs(sum - centre);
776
+ if (deviation >= doubledDeviation) tail += probability;
777
+ if (deviation > maxDeviation) {
778
+ maxDeviation = deviation;
779
+ extreme = probability;
780
+ } else if (deviation === maxDeviation) extreme += probability;
781
+ }
782
+ return {
783
+ p: Math.min(1, tail),
784
+ pFloor: Math.min(1, extreme)
785
+ };
786
+ }
787
+ //#endregion
788
+ //#region src/paired-arms.ts
789
+ /**
790
+ * Matched-pair arm comparison — "did the treatment arm beat the baseline arm
791
+ * on the SAME work items?"
792
+ *
793
+ * An arm A/B over run records is only trustworthy when it is PAIRED: the same
794
+ * task/scenario/seed evaluated under both arms, compared item-by-item, so
795
+ * inter-item difficulty variance cancels instead of masquerading as an arm
796
+ * effect. This module owns the two error-prone steps every consumer otherwise
797
+ * hand-rolls:
798
+ *
799
+ * 1. Pairing — matching rows across arms by `pairKey` (and by `repKey`
800
+ * within multi-rep items), with leftovers REPORTED rather than silently
801
+ * dropped (a silently unbalanced pairing biases every paired statistic
802
+ * downstream). Pairing never keys on outcome content: matching reps by
803
+ * their outcomes deflates discordant-pair counts and makes McNemar
804
+ * anti-conservative, so reps pair only by (`pairKey`, `repKey`) identity.
805
+ * 2. Composition — feeding the matched pairs to the correct paired
806
+ * estimators that already live in `statistics`: `mcnemar` +
807
+ * `pairedRiskDifference` for pass/fail, `pairedBootstrap` +
808
+ * `wilcoxonSignedRank` for continuous metrics. No statistic is
809
+ * re-implemented here.
810
+ *
811
+ * The row shape is deliberately structural — callers project a `RunRecord`
812
+ * (or any record) into `{ pairKey, arm, pass?, metrics? }`. Arm names are
813
+ * caller-supplied parameters; the module ships no domain literal.
814
+ */
815
+ /**
816
+ * Match rows across two arms into (baseline, treatment) pairs by `pairKey`.
817
+ *
818
+ * A `pairKey` with at most one row per arm pairs directly, no `repKey`
819
+ * needed. A `pairKey` with multiple reps in either arm requires `repKey` on
820
+ * every one of its rows, and reps pair only on exact (`pairKey`, `repKey`)
821
+ * match — pairing is keyed purely on row identity, never on outcome content
822
+ * (outcome-keyed matching deflates discordant counts and biases McNemar), and
823
+ * is therefore independent of input order. Reps whose `repKey` has no
824
+ * counterpart in the other arm, and items present in only one arm, land in
825
+ * the unpaired lists — reported, never truncated.
826
+ *
827
+ * Fail-loud: throws when either named arm has zero rows (an unknown arm
828
+ * name would otherwise read as "everything unpaired"), when the two arm
829
+ * names are equal, when a multi-rep `pairKey` has a row without `repKey`, or
830
+ * when a (`pairKey`, arm) group repeats a `repKey` (the match would be
831
+ * ambiguous).
832
+ */
833
+ function pairArms(rows, opts) {
834
+ const { baselineArm, treatmentArm } = opts;
835
+ if (baselineArm === treatmentArm) throw new ValidationError(`pairArms: baselineArm and treatmentArm are both '${baselineArm}' — an arm cannot be compared to itself`);
836
+ const byArm = /* @__PURE__ */ new Map();
837
+ const armsSeen = /* @__PURE__ */ new Set();
838
+ for (const row of rows) {
839
+ armsSeen.add(row.arm);
840
+ if (row.arm !== baselineArm && row.arm !== treatmentArm) continue;
841
+ const byKey = byArm.get(row.arm) ?? /* @__PURE__ */ new Map();
842
+ const group = byKey.get(row.pairKey) ?? [];
843
+ group.push(row);
844
+ byKey.set(row.pairKey, group);
845
+ byArm.set(row.arm, byKey);
846
+ }
847
+ for (const arm of [baselineArm, treatmentArm]) if (!byArm.has(arm)) throw new ValidationError(`pairArms: no rows for arm '${arm}' (arms present: ${[...armsSeen].sort().join(", ") || "<none>"})`);
848
+ const baselineByKey = byArm.get(baselineArm);
849
+ const treatmentByKey = byArm.get(treatmentArm);
850
+ const allKeys = [.../* @__PURE__ */ new Set([...baselineByKey.keys(), ...treatmentByKey.keys()])].sort();
851
+ const pairs = [];
852
+ const unpairedBaseline = [];
853
+ const unpairedTreatment = [];
854
+ for (const pairKey of allKeys) {
855
+ const b = baselineByKey.get(pairKey) ?? [];
856
+ const t = treatmentByKey.get(pairKey) ?? [];
857
+ if (b.length <= 1 && t.length <= 1) {
858
+ if (b.length === 1 && t.length === 1) {
859
+ const baseline = b[0];
860
+ const treatment = t[0];
861
+ if (baseline.repKey !== void 0 || treatment.repKey !== void 0) {
862
+ if (baseline.repKey === void 0 || treatment.repKey === void 0 || baseline.repKey !== treatment.repKey) {
863
+ unpairedBaseline.push(baseline);
864
+ unpairedTreatment.push(treatment);
865
+ continue;
866
+ }
867
+ }
868
+ pairs.push({
869
+ pairKey,
870
+ repIndex: 0,
871
+ baseline,
872
+ treatment
873
+ });
874
+ } else {
875
+ unpairedBaseline.push(...b);
876
+ unpairedTreatment.push(...t);
877
+ }
878
+ continue;
879
+ }
880
+ const bByRep = indexByRepKey(b, pairKey, baselineArm);
881
+ const tByRep = indexByRepKey(t, pairKey, treatmentArm);
882
+ const repKeys = [.../* @__PURE__ */ new Set([...bByRep.keys(), ...tByRep.keys()])].sort();
883
+ let repIndex = 0;
884
+ for (const repKey of repKeys) {
885
+ const baseline = bByRep.get(repKey);
886
+ const treatment = tByRep.get(repKey);
887
+ if (baseline !== void 0 && treatment !== void 0) pairs.push({
888
+ pairKey,
889
+ repIndex: repIndex++,
890
+ baseline,
891
+ treatment
892
+ });
893
+ else if (baseline !== void 0) unpairedBaseline.push(baseline);
894
+ else if (treatment !== void 0) unpairedTreatment.push(treatment);
895
+ }
896
+ }
897
+ return {
898
+ pairs,
899
+ unpairedBaseline,
900
+ unpairedTreatment
901
+ };
902
+ }
903
+ /** Index a multi-rep (pairKey, arm) group by `repKey`, enforcing that every
904
+ * row carries one and that no repKey repeats within the group. */
905
+ function indexByRepKey(group, pairKey, arm) {
906
+ const byRep = /* @__PURE__ */ new Map();
907
+ for (const row of group) {
908
+ if (row.repKey === void 0) throw new ValidationError(`pairArms: pairKey '${pairKey}' has multiple reps in an arm, but a row in arm '${arm}' is missing repKey — multi-rep items require an explicit repKey on every row so reps pair by identity (pairing reps by outcome or by index would bias the paired statistics)`);
909
+ if (byRep.has(row.repKey)) throw new ValidationError(`pairArms: duplicate repKey '${row.repKey}' for pairKey '${pairKey}' in arm '${arm}' — (pairKey, repKey) must uniquely identify a rep within an arm`);
910
+ byRep.set(row.repKey, row);
911
+ }
912
+ return byRep;
913
+ }
914
+ /**
915
+ * Full matched-pair arm comparison: pair via {@link pairArms}, then compose
916
+ * the paired estimators from `statistics` over the matched pairs.
917
+ *
918
+ * Correctness uses only the pairs where both sides carry `pass` (`mcnemar.n`
919
+ * is that subset's size); each metric uses only the pairs where both sides
920
+ * carry a finite value for it, with the remainder counted in `nMissing`.
921
+ * Deltas are treatment − baseline throughout.
922
+ *
923
+ * Fail-loud: inherits {@link pairArms}'s unknown-arm throw, and throws on a
924
+ * non-finite metric value — silently treating corrupt telemetry as "metric
925
+ * absent" would misreport it as missing coverage.
926
+ */
927
+ function comparePairedArms(rows, opts) {
928
+ const { pairs, unpairedBaseline, unpairedTreatment } = pairArms(rows, opts);
929
+ let correctness = null;
930
+ const baselinePass = [];
931
+ const treatmentPass = [];
932
+ for (const pair of pairs) {
933
+ if (pair.baseline.pass === void 0 || pair.treatment.pass === void 0) continue;
934
+ baselinePass.push(pair.baseline.pass ? 1 : 0);
935
+ treatmentPass.push(pair.treatment.pass ? 1 : 0);
936
+ }
937
+ if (baselinePass.length > 0) {
938
+ const mc = mcnemar(baselinePass, treatmentPass);
939
+ correctness = {
940
+ b10: mc.b,
941
+ b01: mc.c,
942
+ mcnemar: mc,
943
+ riskDifference: pairedRiskDifference(baselinePass, treatmentPass)
944
+ };
945
+ }
946
+ const metricDeltas = (opts.metricNames ?? [...new Set(pairs.flatMap((p) => [...Object.keys(p.baseline.metrics ?? {}), ...Object.keys(p.treatment.metrics ?? {})]))].sort()).map((name) => {
947
+ const before = [];
948
+ const after = [];
949
+ let nMissing = 0;
950
+ for (const pair of pairs) {
951
+ const b = metricValue(pair.baseline, name);
952
+ const t = metricValue(pair.treatment, name);
953
+ if (b === void 0 || t === void 0) {
954
+ nMissing++;
955
+ continue;
956
+ }
957
+ before.push(b);
958
+ after.push(t);
959
+ }
960
+ const bootstrapCi = before.length === 0 ? null : pairedBootstrap(before, after, opts.bootstrap);
961
+ return {
962
+ name,
963
+ n: before.length,
964
+ nMissing,
965
+ medianDelta: bootstrapCi?.median ?? null,
966
+ meanDelta: bootstrapCi?.mean ?? null,
967
+ bootstrapCi,
968
+ wilcoxon: before.length === 0 ? null : wilcoxonSignedRank(before, after)
969
+ };
970
+ });
971
+ return {
972
+ nPairs: pairs.length,
973
+ nUnpairedBaseline: unpairedBaseline.length,
974
+ nUnpairedTreatment: unpairedTreatment.length,
975
+ correctness,
976
+ metricDeltas
977
+ };
978
+ }
979
+ /**
980
+ * Pair two RunRecord arms by the identity of the evaluated work:
981
+ * `(experimentId, scenarioId, seed)`.
982
+ *
983
+ * Falling back to array order, candidate id, or experiment id can compare
984
+ * different tasks and fabricate lift. Duplicate identities throw.
985
+ */
986
+ function pairRunRecords(baselineRuns, treatmentRuns) {
987
+ const baselineRows = runRecordArmRows(baselineRuns, "baseline");
988
+ const treatmentRows = runRecordArmRows(treatmentRuns, "treatment");
989
+ validateRunRecordArmRows(baselineRows, "baseline");
990
+ validateRunRecordArmRows(treatmentRows, "treatment");
991
+ if (baselineRows.length === 0 || treatmentRows.length === 0) return {
992
+ pairs: [],
993
+ unpairedBaseline: baselineRows.map((row) => row.run),
994
+ unpairedTreatment: treatmentRows.map((row) => row.run)
995
+ };
996
+ const result = pairArms([...baselineRows, ...treatmentRows], {
997
+ baselineArm: "baseline",
998
+ treatmentArm: "treatment"
999
+ });
1000
+ return {
1001
+ pairs: result.pairs.map((pair) => {
1002
+ const baseline = pair.baseline;
1003
+ const treatment = pair.treatment;
1004
+ return {
1005
+ pairKey: pair.pairKey,
1006
+ repKey: baseline.repKey,
1007
+ baseline: baseline.run,
1008
+ treatment: treatment.run
1009
+ };
1010
+ }),
1011
+ unpairedBaseline: result.unpairedBaseline.map((row) => row.run),
1012
+ unpairedTreatment: result.unpairedTreatment.map((row) => row.run)
1013
+ };
1014
+ }
1015
+ function runRecordArmRows(runs, arm) {
1016
+ return runs.map((run) => {
1017
+ const scenarioId = run.scenarioId.trim();
1018
+ if (!scenarioId) throw new ValidationError(`pairRunRecords: run '${run.runId}' is missing scenarioId; paired comparisons require explicit scenario identity`);
1019
+ return {
1020
+ pairKey: JSON.stringify([run.experimentId, scenarioId]),
1021
+ repKey: String(run.seed),
1022
+ arm,
1023
+ run
1024
+ };
1025
+ });
1026
+ }
1027
+ function validateRunRecordArmRows(rows, arm) {
1028
+ const byPairKey = /* @__PURE__ */ new Map();
1029
+ for (const row of rows) {
1030
+ const group = byPairKey.get(row.pairKey) ?? [];
1031
+ group.push(row);
1032
+ byPairKey.set(row.pairKey, group);
1033
+ }
1034
+ for (const [pairKey, group] of byPairKey) if (group.length > 1) indexByRepKey(group, pairKey, arm);
1035
+ }
1036
+ function metricValue(row, name) {
1037
+ const v = row.metrics?.[name];
1038
+ if (v === void 0) return void 0;
1039
+ if (!Number.isFinite(v)) throw new ValidationError(`comparePairedArms: non-finite value for metric '${name}' on pairKey '${row.pairKey}' (arm '${row.arm}'): ${v}`);
1040
+ return v;
1041
+ }
1042
+ //#endregion
1043
+ export { passAtK as _, MANN_WHITNEY_EXACT_MAX_STATES as a, mannWhitneyU as c, isBinaryOutcomeVector as d, mcnemar as f, pairedRiskDifferenceScore as g, pairedRiskDifferenceExact as h, DEFAULT_PERMUTATIONS as i, wilcoxonSignedRank as l, pairedRiskDifference as m, pairArms as n, MANN_WHITNEY_EXACT_MAX_WORK as o, pairedBinaryScale as p, pairRunRecords as r, WILCOXON_EXACT_MAX_N as s, comparePairedArms as t, normalCdf as u, wilson as v };
1044
+
1045
+ //# sourceMappingURL=paired-arms-D-XRF_fy.js.map