@tangle-network/agent-eval 0.144.11 → 0.144.13

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (378) hide show
  1. package/CHANGELOG.md +11 -0
  2. package/dist/agent-profile-B9_GGsG8.d.ts +84 -0
  3. package/dist/agent-profile-B9_GGsG8.d.ts.map +1 -0
  4. package/dist/{agent-profile-cell-BOP-iA9Q.d.ts → agent-profile-cell-BkcRDikH.d.ts} +2 -2
  5. package/dist/{agent-profile-cell-BOP-iA9Q.d.ts.map → agent-profile-cell-BkcRDikH.d.ts.map} +1 -1
  6. package/dist/analyst/index.d.ts +134 -16
  7. package/dist/analyst/index.d.ts.map +1 -1
  8. package/dist/analyst/index.js +364 -10
  9. package/dist/analyst/index.js.map +1 -1
  10. package/dist/backend-integrity-CsVin_Wb.d.ts +280 -0
  11. package/dist/backend-integrity-CsVin_Wb.d.ts.map +1 -0
  12. package/dist/{baseline-C-GocmIW.js → baseline-BhPRQBVn.js} +3 -2
  13. package/dist/{baseline-C-GocmIW.js.map → baseline-BhPRQBVn.js.map} +1 -1
  14. package/dist/{benchmark-CWeqGl7x.js → benchmark-BhT16ep9.js} +2 -2
  15. package/dist/{benchmark-CWeqGl7x.js.map → benchmark-BhT16ep9.js.map} +1 -1
  16. package/dist/{agent-profile-DPi7IZg7.d.ts → benchmark-Ceoan7vk.d.ts} +4 -91
  17. package/dist/benchmark-Ceoan7vk.d.ts.map +1 -0
  18. package/dist/{benchmark-command-BteMFN62.js → benchmark-command-CY6cbhOB.js} +36 -79
  19. package/dist/benchmark-command-CY6cbhOB.js.map +1 -0
  20. package/dist/benchmarks/index.d.ts +244 -2
  21. package/dist/benchmarks/index.d.ts.map +1 -0
  22. package/dist/benchmarks/index.js +733 -1
  23. package/dist/benchmarks/index.js.map +1 -0
  24. package/dist/builder-eval/index.d.ts +23 -2
  25. package/dist/builder-eval/index.d.ts.map +1 -1
  26. package/dist/builder-eval/index.js +227 -3
  27. package/dist/builder-eval/index.js.map +1 -1
  28. package/dist/campaign/index.d.ts +10 -8
  29. package/dist/campaign/index.js +9 -6
  30. package/dist/{campaign-C2TTzQII.js → campaign-BYjBAypg.js} +21 -634
  31. package/dist/campaign-BYjBAypg.js.map +1 -0
  32. package/dist/{canonical-D011XM8r.js → canonical-D-XsTQ6_.js} +2 -2
  33. package/dist/{canonical-D011XM8r.js.map → canonical-D-XsTQ6_.js.map} +1 -1
  34. package/dist/{index-CQTZ-4XN.d.ts → capture-fetch-DDvpjVRU.d.ts} +3 -3
  35. package/dist/capture-fetch-DDvpjVRU.d.ts.map +1 -0
  36. package/dist/{default-registry-BmktKy8r.js → chat-client-Bp4Ebuuc.js} +1276 -1276
  37. package/dist/chat-client-Bp4Ebuuc.js.map +1 -0
  38. package/dist/cli.js +2 -2
  39. package/dist/{client-C9gzZE59.d.ts → client-KUbGulm_.d.ts} +4 -4
  40. package/dist/{client-C9gzZE59.d.ts.map → client-KUbGulm_.d.ts.map} +1 -1
  41. package/dist/contract/index.d.ts +13 -390
  42. package/dist/contract/index.d.ts.map +1 -1
  43. package/dist/contract/index.js +18 -542
  44. package/dist/contract/index.js.map +1 -1
  45. package/dist/{cost-ledger-DMFxsLKr.js → cost-ledger-BSe92yAV.js} +3 -3
  46. package/dist/{cost-ledger-DMFxsLKr.js.map → cost-ledger-BSe92yAV.js.map} +1 -1
  47. package/dist/{cost-ledger-Bv_e8XHY.d.ts → cost-ledger-DbQdN3nO.d.ts} +2 -2
  48. package/dist/{cost-ledger-Bv_e8XHY.d.ts.map → cost-ledger-DbQdN3nO.d.ts.map} +1 -1
  49. package/dist/{counterfactual-CxmxAONP.d.ts → counterfactual--bpysZF0.d.ts} +2 -14
  50. package/dist/{counterfactual-CxmxAONP.d.ts.map → counterfactual--bpysZF0.d.ts.map} +1 -1
  51. package/dist/{counterfactual-CWPTrMH7.js → counterfactual-lDfCx0Uz.js} +3 -31
  52. package/dist/{counterfactual-CWPTrMH7.js.map → counterfactual-lDfCx0Uz.js.map} +1 -1
  53. package/dist/{dataset-C8xaLXdY.d.ts → dataset-CJjKqQfA.d.ts} +2 -8
  54. package/dist/dataset-CJjKqQfA.d.ts.map +1 -0
  55. package/dist/{default-registry-Bf8Woqmq.d.ts → default-registry-Di6HP6pG.d.ts} +7 -12
  56. package/dist/default-registry-Di6HP6pG.d.ts.map +1 -0
  57. package/dist/{analyze-runs-C30yljDJ.js → define-agent-eval-iqjT--ZZ.js} +542 -130
  58. package/dist/define-agent-eval-iqjT--ZZ.js.map +1 -0
  59. package/dist/define-agent-eval-sH24zBfM.d.ts +388 -0
  60. package/dist/define-agent-eval-sH24zBfM.d.ts.map +1 -0
  61. package/dist/descriptive-B5MwKfbf.js +144 -0
  62. package/dist/descriptive-B5MwKfbf.js.map +1 -0
  63. package/dist/{dspy-rlm-engine-DbTk4JdR.js → dspy-rlm-engine-B_qhSc21.js} +4 -4
  64. package/dist/{dspy-rlm-engine-DbTk4JdR.js.map → dspy-rlm-engine-B_qhSc21.js.map} +1 -1
  65. package/dist/effect-sizes-DiH8MGOH.js +82 -0
  66. package/dist/effect-sizes-DiH8MGOH.js.map +1 -0
  67. package/dist/{engine-3hL-XqwJ.d.ts → engine-otFpE2gF.d.ts} +10 -38
  68. package/dist/engine-otFpE2gF.d.ts.map +1 -0
  69. package/dist/{errors-CKPfb2aH.d.ts → errors-DEE6u6ot.d.ts} +2 -14
  70. package/dist/{errors-CKPfb2aH.d.ts.map → errors-DEE6u6ot.d.ts.map} +1 -1
  71. package/dist/{errors-D-LKuDhb.js → errors-Dngq5h35.js} +2 -8
  72. package/dist/{errors-D-LKuDhb.js.map → errors-Dngq5h35.js.map} +1 -1
  73. package/dist/{eval-campaign-B_7wcnav.js → eval-campaign-UB-usSQ2.js} +6 -6
  74. package/dist/{eval-campaign-B_7wcnav.js.map → eval-campaign-UB-usSQ2.js.map} +1 -1
  75. package/dist/{exact-types-DSFFpLLI.d.ts → exact-types-BH1twmAJ.d.ts} +2 -2
  76. package/dist/{exact-types-DSFFpLLI.d.ts.map → exact-types-BH1twmAJ.d.ts.map} +1 -1
  77. package/dist/experiment/index.d.ts +9 -6
  78. package/dist/experiment/index.d.ts.map +1 -1
  79. package/dist/experiment/index.js +11 -7
  80. package/dist/experiment/index.js.map +1 -1
  81. package/dist/experiment-tracker-C29gXM4B.js +269 -0
  82. package/dist/experiment-tracker-C29gXM4B.js.map +1 -0
  83. package/dist/{experiment-tracker-IMntXr6J.d.ts → experiment-tracker-DWHZBAYL.d.ts} +77 -121
  84. package/dist/experiment-tracker-DWHZBAYL.d.ts.map +1 -0
  85. package/dist/{external-optimizer-contracts-CZuJNcT5.d.ts → external-optimizer-contracts-lixrOZdX.d.ts} +3 -3
  86. package/dist/{external-optimizer-contracts-CZuJNcT5.d.ts.map → external-optimizer-contracts-lixrOZdX.d.ts.map} +1 -1
  87. package/dist/external-optimizer-process-BTiNB-RH.js +301 -0
  88. package/dist/external-optimizer-process-BTiNB-RH.js.map +1 -0
  89. package/dist/{single-run-lock-DFWHEB09.js → external-optimizer-subprocess-DrJ9hR8u.js} +144 -438
  90. package/dist/external-optimizer-subprocess-DrJ9hR8u.js.map +1 -0
  91. package/dist/extract-usage-BrQ8mCLX.js +155 -0
  92. package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
  93. package/dist/{failure-cluster-CqcvCcdR.d.ts → failure-cluster-BLURuWG4.d.ts} +2 -3
  94. package/dist/failure-cluster-BLURuWG4.d.ts.map +1 -0
  95. package/dist/{feedback-trajectory-CacHpxsp.d.ts → feedback-trajectory-juOozjAc.d.ts} +5 -36
  96. package/dist/feedback-trajectory-juOozjAc.d.ts.map +1 -0
  97. package/dist/fuzz.d.ts +2 -2
  98. package/dist/fuzz.js +2 -2
  99. package/dist/hosted/index.d.ts +3 -3
  100. package/dist/{index-CvSN3IG1.d.ts → index-B8Ui1mr1.d.ts} +2 -2
  101. package/dist/{index-CvSN3IG1.d.ts.map → index-B8Ui1mr1.d.ts.map} +1 -1
  102. package/dist/{skill-usage-CO9OLRBx.d.ts → index-BWDrSVfw.d.ts} +11 -136
  103. package/dist/index-BWDrSVfw.d.ts.map +1 -0
  104. package/dist/index-Ba3YrbAL.d.ts +1 -0
  105. package/dist/{index-BnEuDAK2.d.ts → index-COQYtuRF.d.ts} +3 -3
  106. package/dist/{index-BnEuDAK2.d.ts.map → index-COQYtuRF.d.ts.map} +1 -1
  107. package/dist/{index-C3ssXVLv.d.ts → index-CvjYbU0D.d.ts} +2 -2
  108. package/dist/{index-C3ssXVLv.d.ts.map → index-CvjYbU0D.d.ts.map} +1 -1
  109. package/dist/{index-DPPGNJ_R.d.ts → index-DSmEylT9.d.ts} +17 -115
  110. package/dist/index-DSmEylT9.d.ts.map +1 -0
  111. package/dist/index.d.ts +2397 -5308
  112. package/dist/index.d.ts.map +1 -1
  113. package/dist/index.js +5914 -10496
  114. package/dist/index.js.map +1 -1
  115. package/dist/{insight-report-CgX_s0Ez.d.ts → insight-report-CRi-Ufrj.d.ts} +4 -4
  116. package/dist/{insight-report-CgX_s0Ez.d.ts.map → insight-report-CRi-Ufrj.d.ts.map} +1 -1
  117. package/dist/{integrity-MLzHOfV9.js → integrity-Cy9WHAtb.js} +2 -2
  118. package/dist/{integrity-MLzHOfV9.js.map → integrity-Cy9WHAtb.js.map} +1 -1
  119. package/dist/{integrity-DY6tIbl0.js → integrity-DysDBWDu.js} +2 -2
  120. package/dist/{integrity-DY6tIbl0.js.map → integrity-DysDBWDu.js.map} +1 -1
  121. package/dist/{integrity-BuqEKu-x.d.ts → integrity-OrcI9Nau.d.ts} +3 -3
  122. package/dist/{integrity-BuqEKu-x.d.ts.map → integrity-OrcI9Nau.d.ts.map} +1 -1
  123. package/dist/internal-BDHPCnjk.js +230 -0
  124. package/dist/internal-BDHPCnjk.js.map +1 -0
  125. package/dist/judge-calibration-C5CbMYce.d.ts +117 -0
  126. package/dist/judge-calibration-C5CbMYce.d.ts.map +1 -0
  127. package/dist/judge-calibration-DZkWrm5H.js +317 -0
  128. package/dist/judge-calibration-DZkWrm5H.js.map +1 -0
  129. package/dist/{kind-factory-B8-r8-y8.js → kind-factory-CPmSd58s.js} +208 -208
  130. package/dist/{kind-factory-B8-r8-y8.js.map → kind-factory-CPmSd58s.js.map} +1 -1
  131. package/dist/ledger-core/index.d.ts +1 -1
  132. package/dist/ledger-core/index.js +2 -2
  133. package/dist/{ledger-core-DXZIqu17.js → ledger-core-BmZt19oQ.js} +110 -110
  134. package/dist/{ledger-core-DXZIqu17.js.map → ledger-core-BmZt19oQ.js.map} +1 -1
  135. package/dist/{llm-client-DzvMUsS_.js → llm-client-d0-2TT1g.js} +4 -64
  136. package/dist/{llm-client-DzvMUsS_.js.map → llm-client-d0-2TT1g.js.map} +1 -1
  137. package/dist/{skillopt-optimization-method-CQdVeM8k.js → llm-judge-dZ8P6nGI.js} +3349 -5401
  138. package/dist/llm-judge-dZ8P6nGI.js.map +1 -0
  139. package/dist/matrix/index.d.ts +2 -2
  140. package/dist/meta-eval/index.d.ts +3 -3
  141. package/dist/meta-eval/index.js +3 -3
  142. package/dist/{metrics-C9YY1OcL.js → metrics-Cl0L1KUy.js} +2 -108
  143. package/dist/{metrics-C9YY1OcL.js.map → metrics-Cl0L1KUy.js.map} +1 -1
  144. package/dist/{mint-B2O60ACG.js → mint-Dj9Ww_3I.js} +4 -4
  145. package/dist/{mint-B2O60ACG.js.map → mint-Dj9Ww_3I.js.map} +1 -1
  146. package/dist/{multi-layer-verifier-DnAqwl0h.d.ts → multi-layer-verifier-DIguZc8Z.d.ts} +3 -3
  147. package/dist/{multi-layer-verifier-DnAqwl0h.d.ts.map → multi-layer-verifier-DIguZc8Z.d.ts.map} +1 -1
  148. package/dist/multiplicity-DIWHvysC.d.ts +43 -0
  149. package/dist/multiplicity-DIWHvysC.d.ts.map +1 -0
  150. package/dist/multishot/index.d.ts +3 -3
  151. package/dist/multishot/index.js +1 -1
  152. package/dist/openapi.json +1 -1
  153. package/dist/{opencode-sqlite-8r6WUfHc.js → opencode-sqlite-DJWAXLms.js} +2 -2
  154. package/dist/{opencode-sqlite-8r6WUfHc.js.map → opencode-sqlite-DJWAXLms.js.map} +1 -1
  155. package/dist/package-version-D7lQHt_-.js +34 -0
  156. package/dist/package-version-D7lQHt_-.js.map +1 -0
  157. package/dist/paired-arms-D-XRF_fy.js +1045 -0
  158. package/dist/paired-arms-D-XRF_fy.js.map +1 -0
  159. package/dist/paired-promotion-decision-CGzg0cI_.d.ts +251 -0
  160. package/dist/paired-promotion-decision-CGzg0cI_.d.ts.map +1 -0
  161. package/dist/paired-tests-BHIhYVdu.js +213 -0
  162. package/dist/paired-tests-BHIhYVdu.js.map +1 -0
  163. package/dist/pareto-BqNW3LJR.d.ts +117 -0
  164. package/dist/pareto-BqNW3LJR.d.ts.map +1 -0
  165. package/dist/pipelines/index.d.ts +3 -64
  166. package/dist/pipelines/index.d.ts.map +1 -1
  167. package/dist/pipelines/index.js +4 -284
  168. package/dist/pipelines/index.js.map +1 -1
  169. package/dist/power-and-mde-CHIrXJll.js +195 -0
  170. package/dist/power-and-mde-CHIrXJll.js.map +1 -0
  171. package/dist/{promotion-policy-CrLrmys8.js → power-preflight-DEw-uC7q.js} +4 -184
  172. package/dist/power-preflight-DEw-uC7q.js.map +1 -0
  173. package/dist/pre-registration-CZwSFQS4.d.ts +577 -0
  174. package/dist/pre-registration-CZwSFQS4.d.ts.map +1 -0
  175. package/dist/{prime-protocol-BfSalTfR.js → prime-protocol-6tZTVsWm.js} +72 -21
  176. package/dist/prime-protocol-6tZTVsWm.js.map +1 -0
  177. package/dist/produced-state-DU79a81m.js +586 -0
  178. package/dist/produced-state-DU79a81m.js.map +1 -0
  179. package/dist/profile-cell.d.ts +1 -1
  180. package/dist/profile-cell.js +1 -1
  181. package/dist/promotion-policy-D0nPhkSy.d.ts +134 -0
  182. package/dist/promotion-policy-D0nPhkSy.d.ts.map +1 -0
  183. package/dist/promotion-policy-xzA40Evo.js +186 -0
  184. package/dist/promotion-policy-xzA40Evo.js.map +1 -0
  185. package/dist/{skillopt-optimization-method-USDKhxSA.d.ts → provenance-CCdxgLDT.d.ts} +57 -478
  186. package/dist/provenance-CCdxgLDT.d.ts.map +1 -0
  187. package/dist/registry-oJeeI4-a.d.ts +178 -0
  188. package/dist/registry-oJeeI4-a.d.ts.map +1 -0
  189. package/dist/{release-report-DA2BCu5a.d.ts → release-confidence-BFRE5WSp.d.ts} +180 -112
  190. package/dist/release-confidence-BFRE5WSp.d.ts.map +1 -0
  191. package/dist/{release-report-BUYmoKo2.js → release-confidence-CxDuiAev.js} +125 -217
  192. package/dist/release-confidence-CxDuiAev.js.map +1 -0
  193. package/dist/reporting.d.ts +6 -5
  194. package/dist/reporting.js +7 -5
  195. package/dist/{researcher-BchpD55R.d.ts → researcher-DaN4GST-.d.ts} +7 -40
  196. package/dist/{researcher-BchpD55R.d.ts.map → researcher-DaN4GST-.d.ts.map} +1 -1
  197. package/dist/{reward-hacking-BDToousL.js → reward-hacking-DNgjilrV.js} +3 -3
  198. package/dist/reward-hacking-DNgjilrV.js.map +1 -0
  199. package/dist/{reward-hacking-BI0OMAlo.d.ts → reward-hacking-DSSTuI9r.d.ts} +5 -5
  200. package/dist/{reward-hacking-BI0OMAlo.d.ts.map → reward-hacking-DSSTuI9r.d.ts.map} +1 -1
  201. package/dist/rl.d.ts +7 -7
  202. package/dist/rl.js +11 -10
  203. package/dist/rl.js.map +1 -1
  204. package/dist/rollout/index.d.ts +2 -2
  205. package/dist/rollout/index.js +4 -4
  206. package/dist/{rollout-DRrksrcV.js → rollout-BWtw0I_6.js} +3 -3
  207. package/dist/{rollout-DRrksrcV.js.map → rollout-BWtw0I_6.js.map} +1 -1
  208. package/dist/{rubric-predictive-validity-DvkjPYCe.d.ts → rubric-predictive-validity-BgxtKe4G.d.ts} +2 -2
  209. package/dist/{rubric-predictive-validity-DvkjPYCe.d.ts.map → rubric-predictive-validity-BgxtKe4G.d.ts.map} +1 -1
  210. package/dist/{rubric-predictive-validity-BRR632r1.js → rubric-predictive-validity-Cwwyd7ah.js} +2 -2
  211. package/dist/{rubric-predictive-validity-BRR632r1.js.map → rubric-predictive-validity-Cwwyd7ah.js.map} +1 -1
  212. package/dist/{run-record-BmSPWXJR.js → run-record-BvHPVS-i.js} +3 -3
  213. package/dist/{run-record-BmSPWXJR.js.map → run-record-BvHPVS-i.js.map} +1 -1
  214. package/dist/{run-record-CF4Dwpxr.d.ts → run-record-CKiihE6f.d.ts} +4 -4
  215. package/dist/{run-record-CF4Dwpxr.d.ts.map → run-record-CKiihE6f.d.ts.map} +1 -1
  216. package/dist/{proposal-findings-2GIUo1et.js → run-score-lDzV0X8j.js} +31 -31
  217. package/dist/run-score-lDzV0X8j.js.map +1 -0
  218. package/dist/sandbox-harness-BlSOu4LX.d.ts +70 -0
  219. package/dist/sandbox-harness-BlSOu4LX.d.ts.map +1 -0
  220. package/dist/{schema-Cef2cFmb2.d.ts → schema-Cef2cFmb.d.ts} +1 -1
  221. package/dist/schema-Cef2cFmb.d.ts.map +1 -0
  222. package/dist/semantic-concept-judge-BiJxScqe.js +406 -0
  223. package/dist/semantic-concept-judge-BiJxScqe.js.map +1 -0
  224. package/dist/{sequential-D-BLJBKU.js → sequential-C458DXNf.js} +4 -3
  225. package/dist/{sequential-D-BLJBKU.js.map → sequential-C458DXNf.js.map} +1 -1
  226. package/dist/sequential-eprocess-CbUt2htw.js +83 -0
  227. package/dist/sequential-eprocess-CbUt2htw.js.map +1 -0
  228. package/dist/series-convergence-D1cL1f-4.d.ts +129 -0
  229. package/dist/series-convergence-D1cL1f-4.d.ts.map +1 -0
  230. package/dist/{server-iu0ede49.js → server-ulsOdrTI.js} +5 -21
  231. package/dist/server-ulsOdrTI.js.map +1 -0
  232. package/dist/skillopt-optimization-method-C5cotF4E.d.ts +427 -0
  233. package/dist/skillopt-optimization-method-C5cotF4E.d.ts.map +1 -0
  234. package/dist/skillopt-optimization-method-jjdnc3YK.js +1999 -0
  235. package/dist/skillopt-optimization-method-jjdnc3YK.js.map +1 -0
  236. package/dist/{statistical-heldout-TQ-4CYiN.d.ts → statistical-heldout-DoFF5KJX.d.ts} +5 -278
  237. package/dist/statistical-heldout-DoFF5KJX.d.ts.map +1 -0
  238. package/dist/{store-otlp-Dw8PPIlL.js → store-otlp-CsptLYpN.js} +3 -3
  239. package/dist/{store-otlp-Dw8PPIlL.js.map → store-otlp-CsptLYpN.js.map} +1 -1
  240. package/dist/store-tool-spans-Cq9mFd-q.js +667 -0
  241. package/dist/store-tool-spans-Cq9mFd-q.js.map +1 -0
  242. package/dist/store-tool-spans-DbDLOBOb.d.ts +342 -0
  243. package/dist/store-tool-spans-DbDLOBOb.d.ts.map +1 -0
  244. package/dist/student-t-CvBq2mve.js +38 -0
  245. package/dist/student-t-CvBq2mve.js.map +1 -0
  246. package/dist/{summary-report-Lf-5I7xh.js → summary-report-Blysd6Z2.js} +6 -3
  247. package/dist/{summary-report-Lf-5I7xh.js.map → summary-report-Blysd6Z2.js.map} +1 -1
  248. package/dist/{summary-report-BNR7DWTj.d.ts → summary-report-CaL-Hnxt.d.ts} +5 -5
  249. package/dist/{summary-report-BNR7DWTj.d.ts.map → summary-report-CaL-Hnxt.d.ts.map} +1 -1
  250. package/dist/supervisor-run/index.d.ts +391 -3
  251. package/dist/supervisor-run/index.d.ts.map +1 -0
  252. package/dist/supervisor-run/index.js +1689 -2
  253. package/dist/{supervisor-run-D_sokXcO.js.map → supervisor-run/index.js.map} +1 -1
  254. package/dist/{extract-usage-CdZdoj1s.js → task-failure-attributes-CpQ4y5RD.js} +5 -157
  255. package/dist/task-failure-attributes-CpQ4y5RD.js.map +1 -0
  256. package/dist/{tool-groups-CZPGGlHf.d.ts → tool-groups-Cteb03Ps.d.ts} +3 -3
  257. package/dist/tool-groups-Cteb03Ps.d.ts.map +1 -0
  258. package/dist/tool-waste-BDdBZG1F.js +803 -0
  259. package/dist/tool-waste-BDdBZG1F.js.map +1 -0
  260. package/dist/tool-waste-DjRDEsuI.d.ts +128 -0
  261. package/dist/tool-waste-DjRDEsuI.d.ts.map +1 -0
  262. package/dist/trace-repair/index.d.ts +14 -5
  263. package/dist/trace-repair/index.d.ts.map +1 -1
  264. package/dist/trace-repair/index.js +35 -7
  265. package/dist/trace-repair/index.js.map +1 -1
  266. package/dist/traces.d.ts +406 -7
  267. package/dist/traces.d.ts.map +1 -0
  268. package/dist/traces.js +1011 -10
  269. package/dist/traces.js.map +1 -0
  270. package/dist/trajectory-replay/index.d.ts +16 -3
  271. package/dist/trajectory-replay/index.d.ts.map +1 -1
  272. package/dist/trajectory-replay/index.js +52 -5
  273. package/dist/trajectory-replay/index.js.map +1 -1
  274. package/dist/types-BEPZc6eo.d.ts +93 -0
  275. package/dist/types-BEPZc6eo.d.ts.map +1 -0
  276. package/dist/{usage-receipt-t7vAzCRQ.js → types-BI4fT3HN.js} +67 -67
  277. package/dist/types-BI4fT3HN.js.map +1 -0
  278. package/dist/{types-BnjdJ70P.d.ts → types-BZ59Ahr8.d.ts} +5 -5
  279. package/dist/{types-BnjdJ70P.d.ts.map → types-BZ59Ahr8.d.ts.map} +1 -1
  280. package/dist/{types-D216SgwM.d.ts → types-Cx3YUh2r.d.ts} +4 -241
  281. package/dist/types-Cx3YUh2r.d.ts.map +1 -0
  282. package/dist/{types-BvDKaULh.d.ts → types-DLQx4mKU.d.ts} +4 -4
  283. package/dist/{types-BvDKaULh.d.ts.map → types-DLQx4mKU.d.ts.map} +1 -1
  284. package/dist/{types-D4mog56g.d.ts → types-yLK8gXE9.d.ts} +2 -2
  285. package/dist/{types-D4mog56g.d.ts.map → types-yLK8gXE9.d.ts.map} +1 -1
  286. package/dist/verdict-BndeTAh_.js +61 -0
  287. package/dist/verdict-BndeTAh_.js.map +1 -0
  288. package/dist/verdict-E4eRNf7-.d.ts +392 -0
  289. package/dist/verdict-E4eRNf7-.d.ts.map +1 -0
  290. package/dist/{verdict-cache-BCcOh0kF.js → verdict-cache-mZf5FEiY.js} +3 -55
  291. package/dist/{verdict-cache-BCcOh0kF.js.map → verdict-cache-mZf5FEiY.js.map} +1 -1
  292. package/dist/wire/index.d.ts +3 -3
  293. package/dist/wire/index.d.ts.map +1 -1
  294. package/dist/wire/index.js +1 -1
  295. package/docs/charter.md +3 -3
  296. package/docs/control-runtime.md +3 -42
  297. package/docs/experiment.md +0 -1
  298. package/docs/feature-guide.md +2 -2
  299. package/docs/trace-repair-grader.md +1 -0
  300. package/docs/trajectory-replay.md +1 -0
  301. package/docs/verdicts.md +43 -0
  302. package/docs/verification-strategies.md +3 -2
  303. package/package.json +6 -11
  304. package/dist/agent-profile-DPi7IZg7.d.ts.map +0 -1
  305. package/dist/analyze-runs-C30yljDJ.js.map +0 -1
  306. package/dist/baseline-CavEbRyH.d.ts +0 -136
  307. package/dist/baseline-CavEbRyH.d.ts.map +0 -1
  308. package/dist/benchmark-command-BteMFN62.js.map +0 -1
  309. package/dist/benchmarks-Dzs8CKb1.js +0 -755
  310. package/dist/benchmarks-Dzs8CKb1.js.map +0 -1
  311. package/dist/campaign-C2TTzQII.js.map +0 -1
  312. package/dist/completion-verifier-DJA5BhPb.d.ts +0 -414
  313. package/dist/completion-verifier-DJA5BhPb.d.ts.map +0 -1
  314. package/dist/control.d.ts +0 -3
  315. package/dist/control.js +0 -2
  316. package/dist/dataset-C8xaLXdY.d.ts.map +0 -1
  317. package/dist/default-registry-Bf8Woqmq.d.ts.map +0 -1
  318. package/dist/default-registry-BmktKy8r.js.map +0 -1
  319. package/dist/engine-3hL-XqwJ.d.ts.map +0 -1
  320. package/dist/experiment-tracker-CnRICnMl.js +0 -500
  321. package/dist/experiment-tracker-CnRICnMl.js.map +0 -1
  322. package/dist/experiment-tracker-IMntXr6J.d.ts.map +0 -1
  323. package/dist/extract-usage-CdZdoj1s.js.map +0 -1
  324. package/dist/failure-cluster-CqcvCcdR.d.ts.map +0 -1
  325. package/dist/feedback-trajectory-CacHpxsp.d.ts.map +0 -1
  326. package/dist/index-BZ3-y4YL.d.ts +0 -391
  327. package/dist/index-BZ3-y4YL.d.ts.map +0 -1
  328. package/dist/index-CQTZ-4XN.d.ts.map +0 -1
  329. package/dist/index-DPPGNJ_R.d.ts.map +0 -1
  330. package/dist/index-YE4KdKbO2.d.ts +0 -335
  331. package/dist/index-YE4KdKbO2.d.ts.map +0 -1
  332. package/dist/paired-arms-iZ08VFMN.js +0 -260
  333. package/dist/paired-arms-iZ08VFMN.js.map +0 -1
  334. package/dist/paired-promotion-decision-B6zJ3gYM.d.ts +0 -114
  335. package/dist/paired-promotion-decision-B6zJ3gYM.d.ts.map +0 -1
  336. package/dist/prime-protocol-BfSalTfR.js.map +0 -1
  337. package/dist/promotion-policy-Ckjhzg_4.d.ts +0 -289
  338. package/dist/promotion-policy-Ckjhzg_4.d.ts.map +0 -1
  339. package/dist/promotion-policy-CrLrmys8.js.map +0 -1
  340. package/dist/proposal-findings-2GIUo1et.js.map +0 -1
  341. package/dist/propose-review-control-dSNPjFUH.js +0 -1458
  342. package/dist/propose-review-control-dSNPjFUH.js.map +0 -1
  343. package/dist/release-report-BUYmoKo2.js.map +0 -1
  344. package/dist/release-report-DA2BCu5a.d.ts.map +0 -1
  345. package/dist/replay-CohS93nE.js +0 -1859
  346. package/dist/replay-CohS93nE.js.map +0 -1
  347. package/dist/replay-DbhZ4Ked.d.ts +0 -834
  348. package/dist/replay-DbhZ4Ked.d.ts.map +0 -1
  349. package/dist/reward-hacking-BDToousL.js.map +0 -1
  350. package/dist/run-evidence-Dhi3C81V.d.ts +0 -225
  351. package/dist/run-evidence-Dhi3C81V.d.ts.map +0 -1
  352. package/dist/schema-Cef2cFmb2.d.ts.map +0 -1
  353. package/dist/semantic-concept-judge-D1z-KepS.js +0 -767
  354. package/dist/semantic-concept-judge-D1z-KepS.js.map +0 -1
  355. package/dist/series-convergence-ofsqPWhs.d.ts +0 -35
  356. package/dist/series-convergence-ofsqPWhs.d.ts.map +0 -1
  357. package/dist/server-iu0ede49.js.map +0 -1
  358. package/dist/single-run-lock-DFWHEB09.js.map +0 -1
  359. package/dist/skill-usage-CO9OLRBx.d.ts.map +0 -1
  360. package/dist/skillopt-optimization-method-CQdVeM8k.js.map +0 -1
  361. package/dist/skillopt-optimization-method-USDKhxSA.d.ts.map +0 -1
  362. package/dist/statistical-heldout-TQ-4CYiN.d.ts.map +0 -1
  363. package/dist/statistics-ByxzSiOM.js +0 -2212
  364. package/dist/statistics-ByxzSiOM.js.map +0 -1
  365. package/dist/statistics-D6Uebe_4.d.ts +0 -968
  366. package/dist/statistics-D6Uebe_4.d.ts.map +0 -1
  367. package/dist/supervisor-run-D_sokXcO.js +0 -1690
  368. package/dist/test-graded-scenario-D1TaI2va.d.ts +0 -141
  369. package/dist/test-graded-scenario-D1TaI2va.d.ts.map +0 -1
  370. package/dist/test-graded-scenario-JHcKQNpq.js +0 -318
  371. package/dist/test-graded-scenario-JHcKQNpq.js.map +0 -1
  372. package/dist/tool-groups-CZPGGlHf.d.ts.map +0 -1
  373. package/dist/tool-use-metrics-DEGMKycK.js +0 -370
  374. package/dist/tool-use-metrics-DEGMKycK.js.map +0 -1
  375. package/dist/types-D216SgwM.d.ts.map +0 -1
  376. package/dist/usage-receipt-t7vAzCRQ.js.map +0 -1
  377. package/dist/verdict-DExhxfgR.d.ts +0 -201
  378. package/dist/verdict-DExhxfgR.d.ts.map +0 -1
@@ -1,2212 +0,0 @@
1
- import { c as ValidationError } from "./errors-D-LKuDhb.js";
2
- //#region src/judge-calibration.ts
3
- /**
4
- * Judge calibration — measure judge quality against human gold + bias.
5
- *
6
- * Workflow:
7
- * 1. Build a golden set: {itemId, humanScore}[].
8
- * 2. Run candidate judges; each produces {itemId, score}.
9
- * 3. `calibrateJudge(golden, candidate)` reports κ + Pearson + MAE.
10
- * 4. `calibrateJudgeContinuous(golden, candidate)` adds quadratic-weighted
11
- * κ over the un-rounded [0,1] scores plus ICC(2,1), Pearson, Spearman,
12
- * and bootstrap CIs — use this for fine-grained judges where rounding
13
- * to int discards information (e.g. 0.78 vs 0.81 both round to 1 and
14
- * look "perfectly agreed" to integer κ).
15
- * 5. Run bias probes (positional, verbosity, self-preference) to
16
- * detect systematic score inflation.
17
- * 6. For N≥2 judges on the same items, `continuousAgreement(scores)`
18
- * reports ICC(2,1) + κ_w + Pearson + Spearman with bootstrap CIs.
19
- *
20
- * Returns actionable diagnostics, not a single number. Consumers then
21
- * decide whether to trust the judge, retrain it, or add a tie-breaker.
22
- */
23
- /**
24
- * Measure judge quality against human gold labels: computes Cohen's κ, Pearson correlation, and MAE over matched item ids.
25
- */
26
- function calibrateJudge(golden, candidate) {
27
- const map = /* @__PURE__ */ new Map();
28
- for (const g of golden) map.set(g.itemId, {
29
- h: g.humanScore,
30
- j: NaN
31
- });
32
- for (const c of candidate) {
33
- const entry = map.get(c.itemId);
34
- if (entry) entry.j = c.score;
35
- }
36
- const common = [...map.values()].filter((v) => Number.isFinite(v.j));
37
- const n = common.length;
38
- if (n < 2) return {
39
- n,
40
- pearson: NaN,
41
- kappa: NaN,
42
- mae: NaN,
43
- worstItems: []
44
- };
45
- const humans = common.map((c) => c.h);
46
- const judges = common.map((c) => c.j);
47
- return {
48
- n,
49
- pearson: pearsonR(humans, judges),
50
- kappa: weightedKappa(humans.map(Math.round), judges.map(Math.round)),
51
- mae: common.map((c) => Math.abs(c.j - c.h)).reduce((a, b) => a + b, 0) / n,
52
- worstItems: [...map.entries()].filter(([, v]) => Number.isFinite(v.j)).map(([itemId, v]) => ({
53
- itemId,
54
- judge: v.j,
55
- human: v.h,
56
- delta: Math.abs(v.j - v.h)
57
- })).sort((a, b) => b.delta - a.delta).slice(0, 5)
58
- };
59
- }
60
- /**
61
- * Feed the same items to the judge twice with A/B swapped and pass all
62
- * results here. Items that don't appear in both positions are ignored.
63
- */
64
- function positionalBias(scores) {
65
- const pairs = /* @__PURE__ */ new Map();
66
- for (const s of scores) {
67
- const slot = pairs.get(s.itemId) ?? {};
68
- if (s.positionOfAInput === "first") slot.first = s.score;
69
- else if (s.positionOfAInput === "second") slot.second = s.score;
70
- pairs.set(s.itemId, slot);
71
- }
72
- const deltas = [];
73
- for (const { first, second } of pairs.values()) if (first !== void 0 && second !== void 0) deltas.push(first - second);
74
- if (deltas.length === 0) return {
75
- avgDelta: 0,
76
- n: 0
77
- };
78
- return {
79
- avgDelta: deltas.reduce((a, b) => a + b, 0) / deltas.length,
80
- n: deltas.length
81
- };
82
- }
83
- function verbosityBias(samples) {
84
- const n = samples.length;
85
- if (n < 3) return {
86
- pearson: NaN,
87
- n
88
- };
89
- return {
90
- pearson: pearsonR(samples.map((s) => s.outputLen), samples.map((s) => s.score)),
91
- n
92
- };
93
- }
94
- /**
95
- * Pass the same scenarios scored with judge-model X grading outputs from
96
- * model X (in-family) and model Y (out-of-family). Non-zero delta
97
- * indicates self-preference.
98
- */
99
- function selfPreference(samples) {
100
- const inF = samples.filter((s) => s.inFamily).map((s) => s.score);
101
- const outF = samples.filter((s) => !s.inFamily).map((s) => s.score);
102
- if (inF.length === 0 || outF.length === 0) return {
103
- inFamilyMean: 0,
104
- outOfFamilyMean: 0,
105
- deltaMean: 0,
106
- n: 0
107
- };
108
- const inMean = inF.reduce((a, b) => a + b, 0) / inF.length;
109
- const outMean = outF.reduce((a, b) => a + b, 0) / outF.length;
110
- return {
111
- inFamilyMean: inMean,
112
- outOfFamilyMean: outMean,
113
- deltaMean: inMean - outMean,
114
- n: samples.length
115
- };
116
- }
117
- /** Quadratic weighted Cohen's κ over bounded integer scores. */
118
- function weightedKappa(a, b) {
119
- if (a.length !== b.length || a.length === 0) return NaN;
120
- const min = Math.min(...a, ...b);
121
- const K = Math.max(...a, ...b) - min + 1;
122
- if (K < 2) return 1;
123
- const observed = Array.from({ length: K }, () => new Array(K).fill(0));
124
- const rowMarg = new Array(K).fill(0);
125
- const colMarg = new Array(K).fill(0);
126
- for (let i = 0; i < a.length; i++) {
127
- const ai = a[i] - min;
128
- const bi = b[i] - min;
129
- const row = observed[ai];
130
- row[bi] = (row[bi] ?? 0) + 1;
131
- rowMarg[ai]++;
132
- colMarg[bi]++;
133
- }
134
- let num = 0;
135
- let den = 0;
136
- for (let i = 0; i < K; i++) for (let j = 0; j < K; j++) {
137
- const w = (i - j) ** 2 / (K - 1) ** 2;
138
- const expected = rowMarg[i] * colMarg[j] / a.length;
139
- num += w * observed[i][j];
140
- den += w * expected;
141
- }
142
- if (den === 0) return 1;
143
- return 1 - num / den;
144
- }
145
- /**
146
- * Inter-rater agreement on continuous (typically [0,1]) scores.
147
- *
148
- * `scores` has shape [n_items][n_raters]. Rows with any non-finite entry
149
- * are dropped. Returns NaN metrics if fewer than 2 raters or 2 complete
150
- * items remain.
151
- */
152
- function continuousAgreement(scores, opts = {}) {
153
- const bootstrap = opts.bootstrap ?? 1e3;
154
- const weights = opts.weights ?? "quadratic";
155
- const seed = opts.seed ?? 12648430;
156
- const ciLevel = opts.ciLevel ?? .95;
157
- const matrix = scores.filter((row) => row.length >= 2 && row.every((v) => Number.isFinite(v)));
158
- const raters = matrix[0]?.length ?? 0;
159
- const clean = matrix.filter((row) => row.length === raters);
160
- const nClean = clean.length;
161
- if (nClean < 2 || raters < 2) return {
162
- weightedKappa: NaN,
163
- icc: NaN,
164
- pearson: NaN,
165
- spearman: NaN,
166
- ci: {
167
- icc: [NaN, NaN],
168
- weightedKappa: [NaN, NaN]
169
- },
170
- n: nClean,
171
- raters
172
- };
173
- const kappa = continuousWeightedKappa(clean, weights);
174
- const icc = icc21(clean);
175
- const pearson = avgPairwise(clean, pearsonR);
176
- const spearman = avgPairwise(clean, spearmanR);
177
- const ciIcc = [NaN, NaN];
178
- const ciKappa = [NaN, NaN];
179
- if (bootstrap > 0) {
180
- const rng = mulberry32$1(seed);
181
- const iccs = [];
182
- const kappas = [];
183
- for (let b = 0; b < bootstrap; b++) {
184
- const sample = new Array(nClean);
185
- for (let i = 0; i < nClean; i++) sample[i] = clean[Math.floor(rng() * nClean)];
186
- const iccB = icc21(sample);
187
- const kB = continuousWeightedKappa(sample, weights);
188
- if (Number.isFinite(iccB)) iccs.push(iccB);
189
- if (Number.isFinite(kB)) kappas.push(kB);
190
- }
191
- const [lo, hi] = percentileBounds(ciLevel);
192
- if (iccs.length > 0) {
193
- iccs.sort((a, b) => a - b);
194
- ciIcc[0] = quantile(iccs, lo);
195
- ciIcc[1] = quantile(iccs, hi);
196
- }
197
- if (kappas.length > 0) {
198
- kappas.sort((a, b) => a - b);
199
- ciKappa[0] = quantile(kappas, lo);
200
- ciKappa[1] = quantile(kappas, hi);
201
- }
202
- }
203
- return {
204
- weightedKappa: kappa,
205
- icc,
206
- pearson,
207
- spearman,
208
- ci: {
209
- icc: ciIcc,
210
- weightedKappa: ciKappa
211
- },
212
- n: nClean,
213
- raters
214
- };
215
- }
216
- /**
217
- * Extends `calibrateJudge` with continuous-value agreement metrics while
218
- * retaining its base calibration summary.
219
- */
220
- function calibrateJudgeContinuous(golden, candidate, opts = {}) {
221
- const base = calibrateJudge(golden, candidate);
222
- const map = /* @__PURE__ */ new Map();
223
- for (const g of golden) map.set(g.itemId, {
224
- h: g.humanScore,
225
- j: NaN
226
- });
227
- for (const c of candidate) {
228
- const entry = map.get(c.itemId);
229
- if (entry) entry.j = c.score;
230
- }
231
- const rows = [];
232
- for (const v of map.values()) if (Number.isFinite(v.j)) rows.push([v.h, v.j]);
233
- const agreement = continuousAgreement(rows, opts);
234
- return {
235
- ...base,
236
- weightedKappaContinuous: agreement.weightedKappa,
237
- icc: agreement.icc,
238
- spearman: agreement.spearman,
239
- ci: agreement.ci
240
- };
241
- }
242
- /**
243
- * Quadratic-weighted κ on continuous scores. With weights w(x,y) = (x-y)^2
244
- * (or |x-y| for linear) the formula collapses to:
245
- *
246
- * κ_w = 1 − E_obs[w] / E_exp[w]
247
- *
248
- * where E_obs averages w over paired (a_i, b_i) and E_exp averages w over
249
- * the independent product distribution (sum_{i,j} w(a_i, b_j) / n^2).
250
- * The normalisation by (max-min)^2 in the integer version cancels in the
251
- * ratio, so we don't need it here. Generalises to N raters by averaging κ_w
252
- * over all rater pairs (mean pairwise weighted agreement).
253
- */
254
- function continuousWeightedKappa(rows, scheme) {
255
- if (rows.length === 0) return NaN;
256
- const raters = rows[0].length;
257
- if (raters < 2) return NaN;
258
- const wFn = scheme === "linear" ? (x, y) => Math.abs(x - y) : (x, y) => (x - y) ** 2;
259
- let sum = 0;
260
- let pairs = 0;
261
- for (let r1 = 0; r1 < raters; r1++) for (let r2 = r1 + 1; r2 < raters; r2++) {
262
- const a = rows.map((row) => row[r1]);
263
- const b = rows.map((row) => row[r2]);
264
- const n = a.length;
265
- let obs = 0;
266
- for (let i = 0; i < n; i++) obs += wFn(a[i], b[i]);
267
- obs /= n;
268
- let exp = 0;
269
- for (let i = 0; i < n; i++) for (let j = 0; j < n; j++) exp += wFn(a[i], b[j]);
270
- exp /= n * n;
271
- if (exp === 0) sum += obs === 0 ? 1 : 0;
272
- else sum += 1 - obs / exp;
273
- pairs++;
274
- }
275
- return pairs === 0 ? NaN : sum / pairs;
276
- }
277
- /**
278
- * ICC(2,1) — two-way random effects, absolute agreement, single rater.
279
- *
280
- * ICC(2,1) = (MSR − MSE) / (MSR + (k−1)·MSE + k·(MSC − MSE)/n)
281
- *
282
- * where MSR = between-rows MS, MSC = between-columns MS, MSE = residual MS,
283
- * n = rows (items), k = columns (raters).
284
- */
285
- function icc21(rows) {
286
- const n = rows.length;
287
- if (n < 2) return NaN;
288
- const k = rows[0].length;
289
- if (k < 2) return NaN;
290
- const rowMeans = rows.map((row) => row.reduce((s, v) => s + v, 0) / k);
291
- const colMeans = new Array(k).fill(0);
292
- for (let j = 0; j < k; j++) {
293
- let s = 0;
294
- for (let i = 0; i < n; i++) s += rows[i][j];
295
- colMeans[j] = s / n;
296
- }
297
- let grand = 0;
298
- for (let i = 0; i < n; i++) grand += rowMeans[i];
299
- grand /= n;
300
- let ssR = 0;
301
- for (let i = 0; i < n; i++) ssR += (rowMeans[i] - grand) ** 2;
302
- ssR *= k;
303
- let ssC = 0;
304
- for (let j = 0; j < k; j++) ssC += (colMeans[j] - grand) ** 2;
305
- ssC *= n;
306
- let ssT = 0;
307
- for (let i = 0; i < n; i++) for (let j = 0; j < k; j++) ssT += (rows[i][j] - grand) ** 2;
308
- const ssE = ssT - ssR - ssC;
309
- const dfR = n - 1;
310
- const dfC = k - 1;
311
- const dfE = (n - 1) * (k - 1);
312
- const msR = ssR / dfR;
313
- const msC = ssC / dfC;
314
- const msE = dfE > 0 ? ssE / dfE : 0;
315
- const denom = msR + (k - 1) * msE + k * (msC - msE) / n;
316
- if (denom === 0) return msR === 0 && msE === 0 ? 1 : 0;
317
- return (msR - msE) / denom;
318
- }
319
- /** Average pairwise statistic over all rater pairs. */
320
- function avgPairwise(rows, fn) {
321
- const k = rows[0]?.length ?? 0;
322
- if (k < 2) return NaN;
323
- let sum = 0;
324
- let pairs = 0;
325
- for (let i = 0; i < k; i++) for (let j = i + 1; j < k; j++) {
326
- const r = fn(rows.map((row) => row[i]), rows.map((row) => row[j]));
327
- if (Number.isFinite(r)) {
328
- sum += r;
329
- pairs++;
330
- }
331
- }
332
- return pairs === 0 ? NaN : sum / pairs;
333
- }
334
- /** Seeded PRNG — Mulberry32. Deterministic across platforms. */
335
- function mulberry32$1(seed) {
336
- let a = seed >>> 0;
337
- return () => {
338
- a = a + 1831565813 >>> 0;
339
- let t = a;
340
- t = Math.imul(t ^ t >>> 15, t | 1);
341
- t ^= t + Math.imul(t ^ t >>> 7, t | 61);
342
- return ((t ^ t >>> 14) >>> 0) / 4294967296;
343
- };
344
- }
345
- function percentileBounds(ciLevel) {
346
- const tail = (1 - ciLevel) / 2;
347
- return [tail, 1 - tail];
348
- }
349
- /** Linear-interpolated quantile of a pre-sorted ascending array. */
350
- function quantile(sorted, q) {
351
- if (sorted.length === 0) return NaN;
352
- if (sorted.length === 1) return sorted[0];
353
- const pos = q * (sorted.length - 1);
354
- const lo = Math.floor(pos);
355
- const hi = Math.ceil(pos);
356
- if (lo === hi) return sorted[lo];
357
- const frac = pos - lo;
358
- return sorted[lo] * (1 - frac) + sorted[hi] * frac;
359
- }
360
- //#endregion
361
- //#region src/math/normal.ts
362
- /**
363
- * Standard normal cumulative distribution using Abramowitz and Stegun 7.1.26.
364
- *
365
- * The approximation is evaluated as erf(x / sqrt(2)). Computing the negative
366
- * tail from the complementary term avoids cancellation when x is far below 0.
367
- * The maximum absolute CDF error is approximately 7.5e-8.
368
- */
369
- function normalCdf(x) {
370
- if (x === 0) return .5;
371
- const a1 = .254829592;
372
- const a2 = -.284496736;
373
- const a3 = 1.421413741;
374
- const a4 = -1.453152027;
375
- const a5 = 1.061405429;
376
- const p = .3275911;
377
- const scaled = Math.abs(x) / Math.SQRT2;
378
- const t = 1 / (1 + p * scaled);
379
- const complement = ((((a5 * t + a4) * t + a3) * t + a2) * t + a1) * t * Math.exp(-scaled * scaled);
380
- return x < 0 ? complement / 2 : 1 - complement / 2;
381
- }
382
- //#endregion
383
- //#region src/math/special-functions.ts
384
- /** Lanczos approximation to ln Gamma(z). */
385
- function lnGamma(z) {
386
- const g = 7;
387
- const coefficients = [
388
- .9999999999998099,
389
- 676.5203681218851,
390
- -1259.1392167224028,
391
- 771.3234287776531,
392
- -176.6150291621406,
393
- 12.507343278686905,
394
- -.13857109526572012,
395
- 9984369578019572e-21,
396
- 1.5056327351493116e-7
397
- ];
398
- if (z < .5) return Math.log(Math.PI / Math.sin(Math.PI * z)) - lnGamma(1 - z);
399
- z -= 1;
400
- let x = coefficients[0];
401
- for (let i = 1; i < 9; i++) x += coefficients[i] / (z + i);
402
- const t = z + g + .5;
403
- return .5 * Math.log(2 * Math.PI) + (z + .5) * Math.log(t) - t + Math.log(x);
404
- }
405
- /**
406
- * Regularized incomplete beta function I_x(a, b).
407
- *
408
- * The Lentz continued fraction converges only for `x < (a+1)/(a+b+2)`; outside
409
- * that domain it must be reached through the symmetry `I_x(a,b) = 1 −
410
- * I_{1−x}(b,a)`. `studentTCdf` drives `x → 1` as `|t| → 0`, so the mirrored
411
- * branch is the one every near-null t-statistic takes.
412
- */
413
- function regularizedIncompleteBeta(x, a, b) {
414
- if (x <= 0) return 0;
415
- if (x >= 1) return 1;
416
- const logBeta = lnGamma(a) + lnGamma(b) - lnGamma(a + b);
417
- const front = Math.exp(Math.log(x) * a + Math.log(1 - x) * b - logBeta);
418
- if (x < (a + 1) / (a + b + 2)) return front * betaContinuedFraction(x, a, b) / a;
419
- return 1 - front * betaContinuedFraction(1 - x, b, a) / b;
420
- }
421
- /** Modified Lentz evaluation of the beta continued fraction at `x`. */
422
- function betaContinuedFraction(x, a, b) {
423
- const maxIterations = 300;
424
- const epsilon = 3e-15;
425
- let c = 1;
426
- let d = 1 - (a + b) * x / (a + 1);
427
- if (Math.abs(d) < 1e-30) d = 1e-30;
428
- d = 1 / d;
429
- let fraction = d;
430
- for (let m = 1; m <= maxIterations; m++) {
431
- const m2 = 2 * m;
432
- let numerator = m * (b - m) * x / ((a + m2 - 1) * (a + m2));
433
- d = 1 + numerator * d;
434
- if (Math.abs(d) < 1e-30) d = 1e-30;
435
- c = 1 + numerator / c;
436
- if (Math.abs(c) < 1e-30) c = 1e-30;
437
- d = 1 / d;
438
- fraction *= d * c;
439
- numerator = -((a + m) * (a + b + m) * x) / ((a + m2) * (a + m2 + 1));
440
- d = 1 + numerator * d;
441
- if (Math.abs(d) < 1e-30) d = 1e-30;
442
- c = 1 + numerator / c;
443
- if (Math.abs(c) < 1e-30) c = 1e-30;
444
- d = 1 / d;
445
- const delta = d * c;
446
- fraction *= delta;
447
- if (Math.abs(delta - 1) < epsilon) break;
448
- }
449
- return fraction;
450
- }
451
- //#endregion
452
- //#region src/math/student-t.ts
453
- /**
454
- * Student-t CDF via the regularized incomplete beta function.
455
- */
456
- function studentTCdf(t, degreesOfFreedom) {
457
- if (degreesOfFreedom <= 0) return .5;
458
- const beta = regularizedIncompleteBeta(degreesOfFreedom / (degreesOfFreedom + t * t), degreesOfFreedom / 2, .5);
459
- return t >= 0 ? 1 - .5 * beta : .5 * beta;
460
- }
461
- /**
462
- * Inverse Student-t CDF, solved against {@link studentTCdf}.
463
- *
464
- * The CDF is monotone, so bracket expansion followed by bisection is stable
465
- * across fractional degrees of freedom and does not need a separate
466
- * approximation with a different error profile.
467
- */
468
- function studentTQuantile(probability, degreesOfFreedom) {
469
- if (!Number.isFinite(probability) || probability < 0 || probability > 1) throw new RangeError(`studentTQuantile: probability must be in [0,1], got ${probability}`);
470
- if (!Number.isFinite(degreesOfFreedom) || degreesOfFreedom <= 0) throw new RangeError(`studentTQuantile: degreesOfFreedom must be positive and finite, got ${degreesOfFreedom}`);
471
- if (probability === 0) return Number.NEGATIVE_INFINITY;
472
- if (probability === 1) return Number.POSITIVE_INFINITY;
473
- if (probability === .5) return 0;
474
- if (probability < .5) return -studentTQuantile(1 - probability, degreesOfFreedom);
475
- let low = 0;
476
- let high = 1;
477
- while (studentTCdf(high, degreesOfFreedom) < probability) high *= 2;
478
- for (let iteration = 0; iteration < 64; iteration++) {
479
- const middle = (low + high) / 2;
480
- if (studentTCdf(middle, degreesOfFreedom) < probability) low = middle;
481
- else high = middle;
482
- }
483
- return (low + high) / 2;
484
- }
485
- //#endregion
486
- //#region src/statistics.ts
487
- /** Identity: dimensions already follow "higher = better" by prompt convention
488
- * (inverted dims like hallucination are scored 10 = best at the source). */
489
- const normalizeScores = (scores) => scores;
490
- /** Weighted mean — falls back to uniform weights when omitted */
491
- function weightedMean(scores) {
492
- if (scores.length === 0) return 0;
493
- let totalWeight = 0;
494
- let weightedSum = 0;
495
- for (const { score, weight } of scores) {
496
- const w = weight ?? 1;
497
- weightedSum += score * w;
498
- totalWeight += w;
499
- }
500
- return totalWeight > 0 ? weightedSum / totalWeight : 0;
501
- }
502
- /**
503
- * Percentile bootstrap confidence interval on the mean of `scores`.
504
- *
505
- * Descriptive spread. It is not a significance test, and at small n its bounds
506
- * are anti-conservative in the same way {@link pairedBootstrap}'s are — see
507
- * {@link BOOTSTRAP_GATE_MIN_N}. With no `seed` the resampling is seeded from
508
- * the scores themselves, so the interval is reproducible either way.
509
- */
510
- function confidenceInterval(scores, confidence = .95, opts = {}) {
511
- if (scores.length === 0) return {
512
- mean: 0,
513
- lower: 0,
514
- upper: 0
515
- };
516
- if (scores.length === 1) return {
517
- mean: scores[0],
518
- lower: scores[0],
519
- upper: scores[0]
520
- };
521
- const n = scores.length;
522
- const mean = scores.reduce((a, b) => a + b, 0) / n;
523
- const B = opts.resamples ?? 1e3;
524
- const rng = makeRng(opts.seed, scores);
525
- const bootstrapMeans = [];
526
- for (let i = 0; i < B; i++) {
527
- let sum = 0;
528
- for (let j = 0; j < n; j++) sum += scores[Math.floor(rng() * n)];
529
- bootstrapMeans.push(sum / n);
530
- }
531
- bootstrapMeans.sort((a, b) => a - b);
532
- const alpha = 1 - confidence;
533
- const lowerIdx = Math.floor(alpha / 2 * B);
534
- const upperIdx = Math.floor((1 - alpha / 2) * B) - 1;
535
- return {
536
- mean,
537
- lower: bootstrapMeans[lowerIdx],
538
- upper: bootstrapMeans[Math.min(upperIdx, B - 1)]
539
- };
540
- }
541
- /**
542
- * Inter-rater reliability — Krippendorff's α under the squared-difference
543
- * metric, pooled across dimensions.
544
- *
545
- * Each inner array is one judge's scores. Items are matched by position
546
- * WITHIN a dimension: the k-th score a judge supplies carrying dimension
547
- * `d` is item k of `d`, and the ratings compared against each other are
548
- * the ones different judges gave to the same item. Every judge that scores
549
- * a dimension at all must supply the same number of scores for it —
550
- * ragged input cannot be aligned into items and throws rather than
551
- * comparing mismatched items.
552
- *
553
- * α = 1 − D_observed / D_expected: D_observed averages the squared
554
- * difference over within-item judge pairs, D_expected over every pair of
555
- * ratings irrespective of item. α = 1 is perfect agreement, 0 is chance,
556
- * negative is systematic disagreement.
557
- */
558
- function interRaterReliability(judgeScores) {
559
- if (judgeScores.length < 2) return 1;
560
- const perDimension = /* @__PURE__ */ new Map();
561
- for (let judgeIndex = 0; judgeIndex < judgeScores.length; judgeIndex++) for (const s of judgeScores[judgeIndex]) {
562
- let byJudge = perDimension.get(s.dimension);
563
- if (byJudge === void 0) {
564
- byJudge = Array.from({ length: judgeScores.length }, () => []);
565
- perDimension.set(s.dimension, byJudge);
566
- }
567
- byJudge[judgeIndex].push(s.score);
568
- }
569
- const allValues = [];
570
- const pairDiffs = [];
571
- for (const [dimension, byJudge] of perDimension) {
572
- const scoring = byJudge.filter((scores) => scores.length > 0);
573
- if (scoring.length < 2) continue;
574
- const itemCount = scoring[0].length;
575
- if (scoring.some((scores) => scores.length !== itemCount)) throw new ValidationError(`interRaterReliability: dimension '${dimension}' has judges supplying ${scoring.map((scores) => scores.length).join("/")} scores — items cannot be aligned`);
576
- for (let item = 0; item < itemCount; item++) {
577
- const ratings = scoring.map((scores) => scores[item]);
578
- for (const v of ratings) allValues.push(v);
579
- for (let i = 0; i < ratings.length; i++) for (let j = i + 1; j < ratings.length; j++) pairDiffs.push((ratings[i] - ratings[j]) ** 2);
580
- }
581
- }
582
- if (pairDiffs.length === 0 || allValues.length < 2) return 1;
583
- const observedDisagreement = pairDiffs.reduce((a, b) => a + b, 0) / pairDiffs.length;
584
- let expectedDisagreement = 0;
585
- let expectedCount = 0;
586
- for (let i = 0; i < allValues.length; i++) for (let j = i + 1; j < allValues.length; j++) {
587
- expectedDisagreement += (allValues[i] - allValues[j]) ** 2;
588
- expectedCount++;
589
- }
590
- expectedDisagreement = expectedCount > 0 ? expectedDisagreement / expectedCount : 0;
591
- if (expectedDisagreement === 0) return 1;
592
- return 1 - observedDisagreement / expectedDisagreement;
593
- }
594
- /** Maximum dynamic-programming cells used by an exact two-sample rank test. */
595
- const MANN_WHITNEY_EXACT_MAX_STATES = 8192;
596
- /** Maximum inner-loop transitions used by an exact two-sample rank test. */
597
- const MANN_WHITNEY_EXACT_MAX_WORK = 25e4;
598
- /** Non-zero differences up to which the signed-rank null is enumerated exactly. */
599
- const WILCOXON_EXACT_MAX_N = 20;
600
- /** Resamples used when a rank test falls back to Monte Carlo permutation. */
601
- const DEFAULT_PERMUTATIONS = 1e5;
602
- /**
603
- * Mann-Whitney U — two independent samples, no distributional assumption.
604
- *
605
- * Exact conditional (permutation) p by default when the dynamic program fits
606
- * {@link MANN_WHITNEY_EXACT_MAX_STATES} cells and
607
- * {@link MANN_WHITNEY_EXACT_MAX_WORK} transitions, seeded Monte Carlo
608
- * permutation above those limits. This keeps imbalanced designs such as 1+24
609
- * exact without admitting expensive balanced designs merely because they have
610
- * the same total size. Throws on non-finite input and on `method:
611
- * 'asymptotic'` where an exact answer is available. Empty input yields `p = 1,
612
- * pFloor = 1` — no design, no attainable evidence.
613
- */
614
- function mannWhitneyU(a, b, opts = {}) {
615
- assertFiniteSample("mannWhitneyU", "a", a);
616
- assertFiniteSample("mannWhitneyU", "b", b);
617
- const n1 = a.length;
618
- const n2 = b.length;
619
- if (n1 === 0 || n2 === 0) return {
620
- u: 0,
621
- uA: 0,
622
- p: 1,
623
- method: "exact",
624
- pFloor: 1
625
- };
626
- const total = n1 + n2;
627
- const combined = [...a.map((v) => ({
628
- v,
629
- fromA: true
630
- })), ...b.map((v) => ({
631
- v,
632
- fromA: false
633
- }))].sort((x, y) => x.v - y.v);
634
- const { midranks, tieTerm } = midranksWithTieTerm(combined.map((entry) => entry.v));
635
- let rankSumA = 0;
636
- for (let k = 0; k < total; k++) if (combined[k].fromA) rankSumA += midranks[k];
637
- const uA = rankSumA - n1 * (n1 + 1) / 2;
638
- const u = Math.min(uA, n1 * n2 - uA);
639
- const doubled = midranks.map((rank) => Math.round(rank * 2));
640
- const doubledDeviation = Math.abs(2 * uA - n1 * n2);
641
- const selectedN = Math.min(n1, n2);
642
- const otherN = total - selectedN;
643
- const exactCost = exactTwoSampleCost(doubled, selectedN);
644
- const designFloor = exactTwoSampleFloor(doubled, selectedN);
645
- const method = selectRankTestMethod("mannWhitneyU", opts.method ?? "auto", `n1=${n1}, n2=${n2}`, exactCost.states <= 8192 && exactCost.work <= 25e4, designFloor, `${MANN_WHITNEY_EXACT_MAX_STATES.toLocaleString("en-US")} states and ${MANN_WHITNEY_EXACT_MAX_WORK.toLocaleString("en-US")} transitions`);
646
- if (method === "exact") {
647
- const { p, pFloor } = exactTwoSampleP(doubled, selectedN, otherN, doubledDeviation);
648
- return {
649
- u,
650
- uA,
651
- p,
652
- method,
653
- pFloor
654
- };
655
- }
656
- if (method === "asymptotic") return {
657
- u,
658
- uA,
659
- p: asymptoticTwoSidedP(doubledDeviation / 2, twoSampleSigma(n1, n2, total, tieTerm)),
660
- method,
661
- pFloor: designFloor
662
- };
663
- const permutations = resolvePermutations("mannWhitneyU", opts.permutations);
664
- const rng = opts.seed === void 0 ? makeRng(symmetricTwoSampleSeed(a, b)) : makeRng(opts.seed);
665
- let atLeastAsExtreme = 0;
666
- const pool = [...doubled];
667
- for (let iteration = 0; iteration < permutations; iteration++) {
668
- let doubledRankSum = 0;
669
- for (let k = 0; k < selectedN; k++) {
670
- const pick = k + Math.floor(rng() * (total - k));
671
- const swapped = pool[pick];
672
- pool[pick] = pool[k];
673
- pool[k] = swapped;
674
- doubledRankSum += swapped;
675
- }
676
- if (Math.abs(doubledRankSum - selectedN * (selectedN + 1) - selectedN * otherN) >= doubledDeviation) atLeastAsExtreme++;
677
- }
678
- const pFloor = Math.max(1 / (permutations + 1), designFloor);
679
- return {
680
- u,
681
- uA,
682
- p: Math.max((1 + atLeastAsExtreme) / (permutations + 1), pFloor),
683
- method,
684
- pFloor
685
- };
686
- }
687
- /** Partial credit: returns 0-1 ratio of current toward target */
688
- function partialCredit(current, target) {
689
- if (target <= 0) return 1;
690
- return Math.min(1, Math.max(0, current / target));
691
- }
692
- /**
693
- * Paired t-test — before/after measurements on the SAME items.
694
- * Pairing removes inter-item variance, giving tighter significance than
695
- * an unpaired test when comparing prompt v1 vs prompt v2 on identical
696
- * scenarios.
697
- *
698
- * Returns `t = p = null` where the statistic is undefined: fewer than two
699
- * pairs, or a non-zero constant delta whose observed variance is zero. A
700
- * constant shift carries no information about the variance it would have to
701
- * be compared against, so the honest answer is "undefined", not `p = 0` —
702
- * three observations cannot buy absolute certainty. This is the same contract
703
- * {@link pairedCohensDz} states for the same condition. An all-zero delta is
704
- * different: it is a measured null, and returns `t = 0, p = 1`.
705
- */
706
- function pairedTTest(before, after) {
707
- if (before.length !== after.length) throw new ValidationError(`pairedTTest: unequal sample sizes (${before.length} vs ${after.length})`);
708
- assertFiniteSample("pairedTTest", "before", before);
709
- assertFiniteSample("pairedTTest", "after", after);
710
- const n = before.length;
711
- if (n < 2) return {
712
- t: null,
713
- df: 0,
714
- p: null
715
- };
716
- const diffs = before.map((b, i) => after[i] - b);
717
- const mean = diffs.reduce((a, b) => a + b, 0) / n;
718
- const variance = diffs.reduce((acc, d) => acc + (d - mean) ** 2, 0) / (n - 1);
719
- const se = Math.sqrt(variance / n);
720
- if (se === 0) return mean === 0 ? {
721
- t: 0,
722
- df: n - 1,
723
- p: 1
724
- } : {
725
- t: null,
726
- df: n - 1,
727
- p: null
728
- };
729
- const t = mean / se;
730
- const df = n - 1;
731
- return {
732
- t,
733
- df,
734
- p: 2 * (1 - studentTCdf(Math.abs(t), df))
735
- };
736
- }
737
- /**
738
- * Wilcoxon signed-rank — paired, no distributional assumption on the deltas.
739
- *
740
- * Exact conditional (sign-flip) p by default at `n ≤
741
- * {@link WILCOXON_EXACT_MAX_N}` non-zero differences, seeded Monte Carlo
742
- * permutation above it. Throws on non-finite input and on `method:
743
- * 'asymptotic'` where an exact answer is available.
744
- *
745
- * `n` is the count of NON-ZERO differences: exact ties are dropped before
746
- * ranking, so a run of tied pairs shrinks the design and raises `pFloor`.
747
- * All-tied input yields `p = 1, pFloor = 1` — no attainable evidence, which
748
- * `pFloor` states rather than leaving `p = 1` to be read as a measured null.
749
- */
750
- function wilcoxonSignedRank(before, after, opts = {}) {
751
- if (before.length !== after.length) throw new ValidationError(`wilcoxonSignedRank: unequal sample sizes (${before.length} vs ${after.length})`);
752
- assertFiniteSample("wilcoxonSignedRank", "before", before);
753
- assertFiniteSample("wilcoxonSignedRank", "after", after);
754
- const diffs = before.map((b, i) => after[i] - b).filter((d) => d !== 0);
755
- const n = diffs.length;
756
- if (n === 0) return {
757
- w: 0,
758
- p: 1,
759
- method: "exact",
760
- pFloor: 1,
761
- nNonZero: 0
762
- };
763
- const order = diffs.map((d, i) => ({
764
- abs: Math.abs(d),
765
- i
766
- })).sort((x, y) => x.abs - y.abs);
767
- const { midranks, tieTerm } = midranksWithTieTerm(order.map((entry) => entry.abs));
768
- const ranks = new Array(n);
769
- for (let k = 0; k < n; k++) ranks[order[k].i] = midranks[k];
770
- let wPlus = 0;
771
- for (let k = 0; k < n; k++) if (diffs[k] > 0) wPlus += ranks[k];
772
- const doubled = midranks.map((rank) => Math.round(rank * 2));
773
- const doubledDeviation = Math.abs(2 * wPlus - n * (n + 1) / 2);
774
- const designFloor = Math.min(1, 2 ** (1 - n));
775
- const method = selectRankTestMethod("wilcoxonSignedRank", opts.method ?? "auto", `n=${n} non-zero differences`, n <= 20, designFloor, `20 non-zero differences`);
776
- if (method === "exact") {
777
- const { p, pFloor } = exactSignedRankP(doubled, doubledDeviation);
778
- return {
779
- w: wPlus,
780
- p,
781
- method,
782
- pFloor,
783
- nNonZero: n
784
- };
785
- }
786
- if (method === "asymptotic") {
787
- const variance = n * (n + 1) * (2 * n + 1) / 24 - tieTerm / 48;
788
- return {
789
- w: wPlus,
790
- p: asymptoticTwoSidedP(doubledDeviation / 2, Math.sqrt(variance)),
791
- method,
792
- pFloor: designFloor,
793
- nNonZero: n
794
- };
795
- }
796
- const permutations = resolvePermutations("wilcoxonSignedRank", opts.permutations);
797
- const rng = makeRng(opts.seed, before, after);
798
- const doubledCentre = n * (n + 1) / 2;
799
- let atLeastAsExtreme = 0;
800
- for (let iteration = 0; iteration < permutations; iteration++) {
801
- let doubledWPlus = 0;
802
- for (let k = 0; k < n; k++) if (rng() < .5) doubledWPlus += doubled[k];
803
- if (Math.abs(doubledWPlus - doubledCentre) >= doubledDeviation) atLeastAsExtreme++;
804
- }
805
- return {
806
- w: wPlus,
807
- p: (1 + atLeastAsExtreme) / (permutations + 1),
808
- method,
809
- pFloor: Math.max(1 / (permutations + 1), designFloor),
810
- nNonZero: n
811
- };
812
- }
813
- /**
814
- * Cohen's d — standardized effect size for two independent groups.
815
- * Positive d means group b has higher mean than group a.
816
- * Rule of thumb: |d| < 0.2 negligible, 0.2–0.5 small, 0.5–0.8 medium, > 0.8 large.
817
- *
818
- * Returns null where the standardized effect is undefined: fewer than two
819
- * observations in either group, or a zero pooled standard deviation with
820
- * unequal means. Null is NOT "no effect" — zero within-group spread across a
821
- * real mean gap is an unbounded effect, the opposite of negligible. Equal
822
- * means with zero spread is a genuine 0. Same contract as
823
- * {@link pairedCohensDz}.
824
- */
825
- function cohensD(a, b) {
826
- if (a.length < 2 || b.length < 2) return null;
827
- const meanA = a.reduce((x, y) => x + y, 0) / a.length;
828
- const meanB = b.reduce((x, y) => x + y, 0) / b.length;
829
- const varA = a.reduce((acc, x) => acc + (x - meanA) ** 2, 0) / (a.length - 1);
830
- const varB = b.reduce((acc, x) => acc + (x - meanB) ** 2, 0) / (b.length - 1);
831
- const pooled = Math.sqrt(((a.length - 1) * varA + (b.length - 1) * varB) / (a.length + b.length - 2));
832
- if (pooled === 0) return meanB === meanA ? 0 : null;
833
- return (meanB - meanA) / pooled;
834
- }
835
- /**
836
- * Cohen's dz for paired observations: mean(after - before) divided by the
837
- * sample standard deviation of those within-pair deltas.
838
- *
839
- * Returns null when fewer than two pairs exist or a non-zero constant delta
840
- * has zero observed variance. In that case the standardized effect is
841
- * undefined, not an arbitrarily large finite number.
842
- */
843
- function pairedCohensDz(before, after) {
844
- if (before.length !== after.length) throw new ValidationError(`pairedCohensDz: unequal sample sizes (${before.length} vs ${after.length})`);
845
- if (before.length < 2) return null;
846
- const deltas = before.map((value, index) => after[index] - value);
847
- if (deltas.some((value) => !Number.isFinite(value))) throw new ValidationError("pairedCohensDz: all paired values must be finite");
848
- const meanDelta = deltas.reduce((sum, value) => sum + value, 0) / deltas.length;
849
- const variance = deltas.reduce((sum, value) => sum + (value - meanDelta) ** 2, 0) / (deltas.length - 1);
850
- const standardDeviation = Math.sqrt(variance);
851
- const scale = Math.max(1, Math.abs(meanDelta), ...deltas.map(Math.abs));
852
- if (standardDeviation <= Number.EPSILON * scale) return meanDelta === 0 ? 0 : null;
853
- return meanDelta / standardDeviation;
854
- }
855
- /**
856
- * Cliff's delta — a non-parametric effect size for two independent samples.
857
- * `δ = (#(after > before) − #(after < before)) / (n_before · n_after)`,
858
- * ranging [-1, 1]. Positive ⇒ `after` tends to exceed `before` (improvement).
859
- *
860
- * Distribution-free counterpart to Cohen's d: no normality assumption, robust
861
- * to the bounded/skewed score distributions judges produce. Pairs with
862
- * `pairedBootstrap` / `wilcoxonSignedRank` for the non-parametric reporting
863
- * path. Returns 0 when either sample is empty.
864
- */
865
- function cliffsDelta(before, after) {
866
- const n = before.length * after.length;
867
- if (n === 0) return 0;
868
- let dominance = 0;
869
- for (const a of after) for (const b of before) if (a > b) dominance += 1;
870
- else if (a < b) dominance -= 1;
871
- return dominance / n;
872
- }
873
- /**
874
- * Map a Cliff's delta to a qualitative magnitude using the standard
875
- * Romano et al. thresholds (|δ|): <0.147 negligible, <0.33 small,
876
- * <0.474 medium, else large.
877
- */
878
- function interpretCliffs(delta) {
879
- const d = Math.abs(delta);
880
- if (d < .147) return "negligible";
881
- if (d < .33) return "small";
882
- if (d < .474) return "medium";
883
- return "large";
884
- }
885
- /**
886
- * Average-rank-with-ties transform (1-indexed). Tied values receive the mean
887
- * of the ranks they span, the standard correction for Spearman's ρ.
888
- */
889
- function ranks(xs) {
890
- const indexed = xs.map((v, i) => ({
891
- v,
892
- i
893
- })).sort((a, b) => a.v - b.v);
894
- const r = new Array(xs.length);
895
- let i = 0;
896
- while (i < indexed.length) {
897
- let j = i;
898
- while (j + 1 < indexed.length && indexed[j + 1].v === indexed[i].v) j++;
899
- const avg = (i + j) / 2 + 1;
900
- for (let k = i; k <= j; k++) r[indexed[k].i] = avg;
901
- i = j + 1;
902
- }
903
- return r;
904
- }
905
- /**
906
- * Pearson product-moment correlation coefficient r ∈ [-1, 1] between two
907
- * equal-length series. See the edge-case contract above: NaN for n < 2 or
908
- * unequal lengths, 1 when both series are constant, 0 when exactly one is.
909
- */
910
- function pearsonR(a, b) {
911
- if (a.length !== b.length || a.length < 2) return NaN;
912
- const n = a.length;
913
- const meanA = a.reduce((s, v) => s + v, 0) / n;
914
- const meanB = b.reduce((s, v) => s + v, 0) / n;
915
- let num = 0;
916
- let varA = 0;
917
- let varB = 0;
918
- for (let i = 0; i < n; i++) {
919
- const da = a[i] - meanA;
920
- const db = b[i] - meanB;
921
- num += da * db;
922
- varA += da * da;
923
- varB += db * db;
924
- }
925
- if (varA === 0 || varB === 0) return varA === 0 && varB === 0 ? 1 : 0;
926
- return num / Math.sqrt(varA * varB);
927
- }
928
- /**
929
- * Spearman's rank correlation ρ — Pearson over the average-rank-with-ties
930
- * transform of each series. Same edge-case contract as {@link pearsonR}.
931
- */
932
- function spearmanR(a, b) {
933
- if (a.length !== b.length || a.length < 2) return NaN;
934
- return pearsonR(ranks(a), ranks(b));
935
- }
936
- /**
937
- * Weighted composite over judge dimensions: `Σ(score_d · w_d) / Σ(w_d)` across
938
- * the weighted dimensions. The canonical replacement for the per-consumer
939
- * hand-rolled composite math (tax/legal/creative/gtm each ship a copy).
940
- *
941
- * Fail-loud: throws if a weighted dimension is missing from `dims`, if any
942
- * weight is negative, or if the weights sum to 0 — none of which can produce
943
- * a meaningful composite.
944
- */
945
- function weightedComposite(input) {
946
- const entries = Object.entries(input.weights);
947
- if (entries.length === 0) throw new Error("weightedComposite: `weights` is empty — nothing to combine");
948
- let weightedSum = 0;
949
- let weightTotal = 0;
950
- for (const [dim, weight] of entries) {
951
- if (weight < 0) throw new Error(`weightedComposite: weight for '${dim}' is negative (${weight})`);
952
- if (!(dim in input.dims)) throw new Error(`weightedComposite: weighted dimension '${dim}' is absent from \`dims\` — refusing to renormalise onto a different denominator`);
953
- weightedSum += input.dims[dim] * weight;
954
- weightTotal += weight;
955
- }
956
- if (weightTotal === 0) throw new Error("weightedComposite: weights sum to 0 — composite is undefined");
957
- const composite = weightedSum / weightTotal;
958
- return input.threshold === void 0 ? { composite } : {
959
- composite,
960
- pass: composite >= input.threshold
961
- };
962
- }
963
- /**
964
- * Corpus-wide inter-rater agreement across N items × M judges × D dimensions.
965
- *
966
- * For each dimension, builds the [n_items][n_judges] matrix of scores
967
- * (keeping only items every judge rated on that dimension), then runs
968
- * `continuousAgreement` to get ICC(2,1), κ_w, Pearson, Spearman, and
969
- * bootstrap CIs. Reports a pooled mean across dimensions as a single
970
- * "is this judge panel reliable on this corpus?" number.
971
- *
972
- * Fail-loud contract:
973
- * - Empty input throws.
974
- * - Fewer than 2 judges or fewer than 2 items per dimension throws.
975
- * - A judge present in some dimensions but with zero scored items on
976
- * another dimension throws (would silently shrink the matrix).
977
- * - Duplicate (itemId, judgeName, dimension) records throw.
978
- */
979
- function corpusInterRaterAgreement(records, opts = {}) {
980
- if (records.length === 0) throw new ValidationError("corpusInterRaterAgreement: no score records supplied");
981
- const judgesSeen = /* @__PURE__ */ new Set();
982
- const dimsSeen = /* @__PURE__ */ new Set();
983
- const grid = /* @__PURE__ */ new Map();
984
- for (const r of records) {
985
- if (!Number.isFinite(r.score)) throw new ValidationError(`corpusInterRaterAgreement: non-finite score for (item=${r.itemId}, judge=${r.judgeName}, dim=${r.dimension})`);
986
- judgesSeen.add(r.judgeName);
987
- dimsSeen.add(r.dimension);
988
- const byJudge = grid.get(r.dimension) ?? /* @__PURE__ */ new Map();
989
- const byItem = byJudge.get(r.judgeName) ?? /* @__PURE__ */ new Map();
990
- if (byItem.has(r.itemId)) throw new ValidationError(`corpusInterRaterAgreement: duplicate record for (item=${r.itemId}, judge=${r.judgeName}, dim=${r.dimension})`);
991
- byItem.set(r.itemId, r.score);
992
- byJudge.set(r.judgeName, byItem);
993
- grid.set(r.dimension, byJudge);
994
- }
995
- const targetDims = opts.dimensions ?? [...dimsSeen].sort();
996
- for (const d of targetDims) if (!dimsSeen.has(d)) throw new ValidationError(`corpusInterRaterAgreement: dimension '${d}' was requested but no records carry it`);
997
- const targetJudges = opts.judges ? [...opts.judges] : [...judgesSeen].sort();
998
- for (const j of targetJudges) if (!judgesSeen.has(j)) throw new ValidationError(`corpusInterRaterAgreement: judge '${j}' was requested but produced no records`);
999
- if (targetJudges.length < 2) throw new ValidationError(`corpusInterRaterAgreement: need ≥2 judges, got ${targetJudges.length}`);
1000
- const perDimension = [];
1001
- const iccs = [];
1002
- const kappas = [];
1003
- for (const dim of targetDims) {
1004
- const byJudge = grid.get(dim);
1005
- const judgeItemCounts = {};
1006
- for (const j of targetJudges) judgeItemCounts[j] = byJudge.get(j)?.size ?? 0;
1007
- const emptyJudges = targetJudges.filter((j) => judgeItemCounts[j] === 0);
1008
- if (emptyJudges.length > 0) throw new ValidationError(`corpusInterRaterAgreement: dimension '${dim}' has no scores from judge(s) ${emptyJudges.join(", ")} (counts: ${JSON.stringify(judgeItemCounts)})`);
1009
- let commonItems = null;
1010
- for (const j of targetJudges) {
1011
- const ids = new Set(byJudge.get(j).keys());
1012
- if (commonItems === null) commonItems = ids;
1013
- else commonItems = new Set([...commonItems].filter((x) => ids.has(x)));
1014
- }
1015
- const sortedItems = [...commonItems ?? /* @__PURE__ */ new Set()].sort();
1016
- if (sortedItems.length < 2) throw new ValidationError(`corpusInterRaterAgreement: dimension '${dim}' has ${sortedItems.length} item(s) rated by all ${targetJudges.length} judges (need ≥2)`);
1017
- const agreement = continuousAgreement(sortedItems.map((itemId) => targetJudges.map((j) => byJudge.get(j).get(itemId))), opts);
1018
- perDimension.push({
1019
- ...agreement,
1020
- dimension: dim,
1021
- itemIds: sortedItems,
1022
- judgeIds: [...targetJudges]
1023
- });
1024
- if (Number.isFinite(agreement.icc)) iccs.push(agreement.icc);
1025
- if (Number.isFinite(agreement.weightedKappa)) kappas.push(agreement.weightedKappa);
1026
- }
1027
- const mean = (xs) => xs.length === 0 ? NaN : xs.reduce((a, b) => a + b, 0) / xs.length;
1028
- return {
1029
- perDimension,
1030
- overallIcc: mean(iccs),
1031
- overallWeightedKappa: mean(kappas),
1032
- dimensions: targetDims,
1033
- judgeIds: targetJudges
1034
- };
1035
- }
1036
- /**
1037
- * Convenience adapter for `JudgeScore[]` data keyed externally by item.
1038
- *
1039
- * Use when you have per-item arrays of `JudgeScore[]` (e.g. one
1040
- * `ScenarioResult.judgeScores` per scenario) and want corpus-wide
1041
- * agreement without manually flattening. `itemId` must be unique per
1042
- * row of `itemsScores`.
1043
- */
1044
- function corpusInterRaterAgreementFromJudgeScores(itemsScores, opts = {}) {
1045
- const records = [];
1046
- const seen = /* @__PURE__ */ new Set();
1047
- for (const { itemId, scores } of itemsScores) {
1048
- if (seen.has(itemId)) throw new ValidationError(`corpusInterRaterAgreementFromJudgeScores: duplicate itemId '${itemId}'`);
1049
- seen.add(itemId);
1050
- for (const s of scores) records.push({
1051
- itemId,
1052
- judgeName: s.judgeName,
1053
- dimension: s.dimension,
1054
- score: s.score
1055
- });
1056
- }
1057
- return corpusInterRaterAgreement(records, opts);
1058
- }
1059
- /**
1060
- * Required N per arm for a two-sample comparison at target effect size,
1061
- * alpha, and power. Normal-approximation formula:
1062
- * n = 2 * ( (z_{1-α/2} + z_{1-β}) / d )^2
1063
- * where d is Cohen's d. Returns Infinity for effect ≤ 0.
1064
- */
1065
- function requiredSampleSize(opts) {
1066
- const effect = opts.effect;
1067
- if (!Number.isFinite(effect) || effect <= 0) return Infinity;
1068
- const alpha = opts.alpha ?? .05;
1069
- const power = opts.power ?? .8;
1070
- const n = 2 * ((zQuantile(opts.twoSided ?? true ? 1 - alpha / 2 : 1 - alpha) + zQuantile(power)) / effect) ** 2;
1071
- return Math.ceil(n);
1072
- }
1073
- /**
1074
- * Required number of paired observations for a target Cohen's dz.
1075
- * Unlike the independent-groups formula, this has no two-arm factor of two.
1076
- *
1077
- * Normal quantiles with no t correction, so treat the result as a LOWER bound:
1078
- * it returns 32 where the exact t-based answer is 34 at dz = 0.5, and 13 where
1079
- * it is 15 at dz = 0.8 — a 6–13 % shortfall precisely in the range a caller
1080
- * consults to decide whether 3–10 repetitions suffice.
1081
- */
1082
- function requiredPairedSampleSize(opts) {
1083
- const effect = opts.effect;
1084
- if (!Number.isFinite(effect) || effect <= 0) return Infinity;
1085
- const alpha = opts.alpha ?? .05;
1086
- const power = opts.power ?? .8;
1087
- const zAlpha = zQuantile(opts.twoSided ?? true ? 1 - alpha / 2 : 1 - alpha);
1088
- const zBeta = zQuantile(power);
1089
- return Math.ceil(((zAlpha + zBeta) / effect) ** 2);
1090
- }
1091
- /**
1092
- * Minimum detectable paired effect (standardised units) for a target paired
1093
- * sample size: d_min = (z_{1-α/2} + z_β) / sqrt(n_paired). Multiply by
1094
- * sd(deltas) for score units; treat as a lower bound — Wilcoxon and bootstrap
1095
- * have asymptotic relative efficiency below 1 vs the t-test on heavy tails.
1096
- */
1097
- function pairedMde(opts) {
1098
- if (!Number.isFinite(opts.nPaired) || opts.nPaired <= 0) return Infinity;
1099
- const alpha = opts.alpha ?? .05;
1100
- const power = opts.power ?? .8;
1101
- return (zQuantile(opts.twoSided ?? true ? 1 - alpha / 2 : 1 - alpha) + zQuantile(power)) / Math.sqrt(opts.nPaired);
1102
- }
1103
- /**
1104
- * Number of paired observations needed for a McNemar test to reach a target
1105
- * power — the pre-registration companion to {@link mcnemar}. Parametrised by the
1106
- * expected discordant-cell probabilities `p10` (P[treatment wins on a pair]) and
1107
- * `p01` (P[control wins]); concordant pairs carry no information, so the count
1108
- * is driven entirely by the discordant rate. Lachin's (1992) asymptotic normal
1109
- * approximation: with discordant rate `pDisc = p10 + p01` and marginal effect
1110
- * `δ = p10 − p01`,
1111
- * n = ( z_{1-α/2}·√pDisc + z_{1-β}·√(pDisc − δ²) )² / δ².
1112
- * Returns Infinity when there is no effect (p10 === p01). Asymptotic — at the
1113
- * tiny discordant counts where the exact {@link mcnemar} differs from the normal
1114
- * approximation, treat the result as a lower bound and prefer the discordant-pair
1115
- * floor.
1116
- */
1117
- function mcnemarRequiredN(opts) {
1118
- const { p10, p01 } = opts;
1119
- if (p10 < 0 || p01 < 0 || p10 + p01 > 1) throw new Error(`mcnemarRequiredN: require p10,p01 ≥ 0 and p10+p01 ≤ 1 (got ${p10}, ${p01})`);
1120
- const delta = p10 - p01;
1121
- if (delta === 0) return Infinity;
1122
- const alpha = opts.alpha ?? .05;
1123
- const power = opts.power ?? .8;
1124
- const twoSided = opts.twoSided ?? true;
1125
- const pDisc = p10 + p01;
1126
- const zAlpha = zQuantile(twoSided ? 1 - alpha / 2 : 1 - alpha);
1127
- const zBeta = zQuantile(power);
1128
- const n = (zAlpha * Math.sqrt(pDisc) + zBeta * Math.sqrt(Math.max(0, pDisc - delta * delta))) ** 2 / (delta * delta);
1129
- return Math.ceil(n);
1130
- }
1131
- /**
1132
- * Power of a McNemar test at a given number of paired observations, the inverse
1133
- * of {@link mcnemarRequiredN} (same Lachin asymptotic model, same parameters).
1134
- * Returns a value in [0, 1]; equals `alpha` when there is no effect.
1135
- */
1136
- function mcnemarPower(opts) {
1137
- const { p10, p01, nPairs } = opts;
1138
- if (p10 < 0 || p01 < 0 || p10 + p01 > 1) throw new Error(`mcnemarPower: require p10,p01 ≥ 0 and p10+p01 ≤ 1 (got ${p10}, ${p01})`);
1139
- const alpha = opts.alpha ?? .05;
1140
- const twoSided = opts.twoSided ?? true;
1141
- const delta = p10 - p01;
1142
- if (delta === 0 || nPairs <= 0) return alpha;
1143
- const pDisc = p10 + p01;
1144
- const zAlpha = zQuantile(twoSided ? 1 - alpha / 2 : 1 - alpha);
1145
- const denom = Math.sqrt(Math.max(1e-12, pDisc - delta * delta));
1146
- const zBeta = (Math.sqrt(nPairs) * Math.abs(delta) - zAlpha * Math.sqrt(pDisc)) / denom;
1147
- return Math.min(1, Math.max(0, normalCdf(zBeta)));
1148
- }
1149
- /**
1150
- * Bonferroni adjustment: multiply every p-value by the test count, clamp at 1.
1151
- *
1152
- * Rejects at `p_adjusted ≤ alpha` — the boundary is inclusive, matching
1153
- * {@link holm}, which uniformly dominates this correction and must therefore
1154
- * never reject less. Validates its inputs on the same terms.
1155
- */
1156
- function bonferroni(pValues, alpha = .05) {
1157
- assertAlpha("bonferroni", "alpha", alpha);
1158
- assertPValues("bonferroni", pValues);
1159
- const k = pValues.length;
1160
- const adjusted = pValues.map((p) => Math.min(1, p * k));
1161
- return {
1162
- adjusted,
1163
- significant: adjusted.map((p) => p <= alpha)
1164
- };
1165
- }
1166
- /**
1167
- * Holm step-down family-wise error adjustment.
1168
- *
1169
- * P-values are sorted from smallest to largest, multiplied by their remaining
1170
- * hypothesis count, and made monotonically non-decreasing before being mapped
1171
- * back to input order. This uniformly dominates plain Bonferroni while keeping
1172
- * strong family-wise error control under arbitrary dependence.
1173
- */
1174
- function holm(pValues, alpha = .05) {
1175
- assertAlpha("holm", "alpha", alpha);
1176
- assertPValues("holm", pValues);
1177
- const count = pValues.length;
1178
- if (count === 0) return {
1179
- adjusted: [],
1180
- significant: []
1181
- };
1182
- const ordered = pValues.map((pValue, index) => ({
1183
- pValue,
1184
- index
1185
- })).sort((a, b) => a.pValue - b.pValue || a.index - b.index);
1186
- const adjusted = new Array(count);
1187
- let previous = 0;
1188
- for (let rank = 0; rank < count; rank++) {
1189
- const entry = ordered[rank];
1190
- const stepAdjusted = Math.min(1, entry.pValue * (count - rank));
1191
- previous = Math.max(previous, stepAdjusted);
1192
- adjusted[entry.index] = previous;
1193
- }
1194
- return {
1195
- adjusted,
1196
- significant: adjusted.map((pValue) => pValue <= alpha)
1197
- };
1198
- }
1199
- /**
1200
- * Benjamini–Hochberg false discovery rate. Returns adjusted q-values and
1201
- * significance at the target FDR; handles ties and preserves q monotonicity.
1202
- *
1203
- * Rejects at `q ≤ fdr` — the BH rule is inclusive at the boundary, so an
1204
- * exactly-`fdr` q-value is a discovery.
1205
- */
1206
- function benjaminiHochberg(pValues, fdr = .05) {
1207
- assertAlpha("benjaminiHochberg", "fdr", fdr);
1208
- assertPValues("benjaminiHochberg", pValues);
1209
- const n = pValues.length;
1210
- if (n === 0) return {
1211
- qValues: [],
1212
- significant: []
1213
- };
1214
- const indexed = pValues.map((p, i) => ({
1215
- p,
1216
- i
1217
- })).sort((a, b) => a.p - b.p);
1218
- const q = new Array(n);
1219
- let minRight = 1;
1220
- for (let k = n - 1; k >= 0; k--) {
1221
- const rank = k + 1;
1222
- const entry = indexed[k];
1223
- const raw = n / rank * entry.p;
1224
- const bounded = Math.min(minRight, raw);
1225
- minRight = bounded;
1226
- q[entry.i] = Math.min(1, bounded);
1227
- }
1228
- return {
1229
- qValues: q,
1230
- significant: q.map((v) => v <= fdr)
1231
- };
1232
- }
1233
- function assertAlpha(fn, label, value) {
1234
- if (!Number.isFinite(value) || value <= 0 || value >= 1) throw new ValidationError(`${fn}: ${label} must be in (0,1), got ${value}`);
1235
- }
1236
- function assertPValues(fn, pValues) {
1237
- for (const [index, pValue] of pValues.entries()) if (!Number.isFinite(pValue) || pValue < 0 || pValue > 1) throw new ValidationError(`${fn}: pValues[${index}] must be in [0,1], got ${pValue}`);
1238
- }
1239
- /**
1240
- * Pairs below which a percentile bootstrap interval is descriptive spread only.
1241
- *
1242
- * `P(low > 0)` under a true null, against a nominal 2.5 %, measured over 4000
1243
- * seeded trials: 13.53 % at n = 3, 3.52 % at n = 10, 3.10 % at n = 20 on the
1244
- * median; 13.85 %, 4.90 %, 3.80 % on the mean. This is intrinsic to resampling
1245
- * three points, not an implementation error — scipy's BCa gives 16.0 % on the
1246
- * same n = 3 data — so no change to the estimator moves it. Below this floor
1247
- * the decision belongs to the exact sign test or exact signed-rank test.
1248
- */
1249
- const BOOTSTRAP_GATE_MIN_N = 20;
1250
- /**
1251
- * Paired bootstrap on (after − before) deltas. Returns a CI on the chosen
1252
- * statistic (median by default); pairs are resampled with replacement. Throws
1253
- * on unequal sample sizes.
1254
- *
1255
- * `low > threshold` carries the stated confidence ONLY at `n ≥
1256
- * {@link BOOTSTRAP_GATE_MIN_N}`, which `gateEligible` reports. Below it the
1257
- * check fires under a true null several times more often than nominal, so the
1258
- * interval is descriptive spread and a promotion must not turn on it.
1259
- */
1260
- function pairedBootstrap(before, after, opts = {}) {
1261
- if (before.length !== after.length) throw new Error(`pairedBootstrap: unequal sample sizes (${before.length} vs ${after.length})`);
1262
- const confidence = opts.confidence ?? .95;
1263
- const resamples = opts.resamples ?? 2e3;
1264
- const statistic = opts.statistic ?? "median";
1265
- if (confidence <= 0 || confidence >= 1) throw new Error(`pairedBootstrap: confidence must be in (0,1), got ${confidence}`);
1266
- const n = before.length;
1267
- const deltas = before.map((b, i) => after[i] - b);
1268
- const gateEligible = n >= 20;
1269
- if (n === 0) return {
1270
- n: 0,
1271
- median: 0,
1272
- mean: 0,
1273
- low: 0,
1274
- high: 0,
1275
- confidence,
1276
- resamples,
1277
- gateEligible
1278
- };
1279
- if (n === 1) {
1280
- const d = deltas[0];
1281
- return {
1282
- n: 1,
1283
- median: d,
1284
- mean: d,
1285
- low: d,
1286
- high: d,
1287
- confidence,
1288
- resamples,
1289
- gateEligible
1290
- };
1291
- }
1292
- const rng = makeRng(opts.seed, deltas);
1293
- const samples = new Array(resamples);
1294
- for (let b = 0; b < resamples; b++) if (statistic === "mean") {
1295
- let sum = 0;
1296
- for (let k = 0; k < n; k++) sum += deltas[Math.floor(rng() * n)];
1297
- samples[b] = sum / n;
1298
- } else {
1299
- const acc = new Array(n);
1300
- for (let k = 0; k < n; k++) acc[k] = deltas[Math.floor(rng() * n)];
1301
- samples[b] = medianInPlace(acc);
1302
- }
1303
- samples.sort((a, b) => a - b);
1304
- const alpha = 1 - confidence;
1305
- const lowIdx = Math.floor(alpha / 2 * resamples);
1306
- const highIdx = Math.min(resamples - 1, Math.ceil((1 - alpha / 2) * resamples) - 1);
1307
- return {
1308
- n,
1309
- median: medianInPlace([...deltas]),
1310
- mean: deltas.reduce((s, x) => s + x, 0) / n,
1311
- low: samples[lowIdx],
1312
- high: samples[Math.max(highIdx, lowIdx)],
1313
- confidence,
1314
- resamples,
1315
- gateEligible
1316
- };
1317
- }
1318
- /**
1319
- * Exact one-sided sign test over paired differences.
1320
- *
1321
- * Pass `after[i] - before[i]` for each matched item. `alternative = 'greater'`
1322
- * tests whether positive signs are more likely than negative signs and returns
1323
- * `P(Binomial(nNonTies, 0.5) >= positive)`. `alternative = 'less'` treats
1324
- * negative signs as successes instead. With a continuous difference
1325
- * distribution this is the usual directional median test. Exact zero
1326
- * differences are ties and do not enter the binomial denominator. All-tie and
1327
- * empty inputs return p = 1. Every input difference must be finite, and the
1328
- * direction must be chosen explicitly so a caller cannot select it after
1329
- * seeing the signs.
1330
- */
1331
- function pairedSignTest(differences, alternative) {
1332
- if (alternative !== "greater" && alternative !== "less") throw new ValidationError(`pairedSignTest: alternative must be 'greater' or 'less', got ${alternative}`);
1333
- let positive = 0;
1334
- let negative = 0;
1335
- let ties = 0;
1336
- for (let i = 0; i < differences.length; i++) {
1337
- const difference = differences[i];
1338
- if (!Number.isFinite(difference)) throw new ValidationError(`pairedSignTest: difference at index ${i} must be finite, got ${difference}`);
1339
- if (difference > 0) positive++;
1340
- else if (difference < 0) negative++;
1341
- else ties++;
1342
- }
1343
- const nNonTies = positive + negative;
1344
- const successes = alternative === "greater" ? positive : negative;
1345
- return {
1346
- n: differences.length,
1347
- positive,
1348
- negative,
1349
- ties,
1350
- nNonTies,
1351
- alternative,
1352
- pValue: binomialHalfUpperTail(successes, nNonTies)
1353
- };
1354
- }
1355
- /**
1356
- * Wilson score interval for a binomial proportion. Correct at small n and near
1357
- * 0/1, where the normal (Wald) approximation produces bounds outside [0, 1] and
1358
- * understates coverage. Use this for any pass-rate / hit-rate / realness-rate
1359
- * CI — the continuous `confidenceInterval` assumes the wrong distribution for a
1360
- * proportion. `n = 0 ⇒ {0, 0, 0}`.
1361
- */
1362
- function wilson(successes, n, confidence = .95) {
1363
- if (n <= 0) return {
1364
- estimate: 0,
1365
- lower: 0,
1366
- upper: 0
1367
- };
1368
- if (successes < 0 || successes > n) throw new Error(`wilson: successes (${successes}) must be in [0, ${n}]`);
1369
- const z = zQuantile(1 - (1 - confidence) / 2);
1370
- const p = successes / n;
1371
- const z2 = z * z;
1372
- const denom = 1 + z2 / n;
1373
- const center = (p + z2 / (2 * n)) / denom;
1374
- const half = z * Math.sqrt((p * (1 - p) + z2 / (4 * n)) / n) / denom;
1375
- return {
1376
- estimate: p,
1377
- lower: Math.max(0, center - half),
1378
- upper: Math.min(1, center + half)
1379
- };
1380
- }
1381
- /**
1382
- * Are these per-item outcomes binary (every value exactly 0 or 1)?
1383
- *
1384
- * The discriminator a promotion gate needs before choosing a paired statistic.
1385
- * On binary outcomes the paired delta vector lives in {-1, 0, +1} and is
1386
- * normally dominated by zeros (both arms solve, or both arms miss, most items),
1387
- * so its MEDIAN is pinned at exactly 0 no matter how large the real shift in
1388
- * success rate is — and a bootstrap CI on that median collapses to [0, 0].
1389
- * A gate keying on `ci.low > threshold` is then structurally unable to see
1390
- * either a gain or a regression. Detect this shape and switch to the
1391
- * paired-binary estimators ({@link mcnemar}, {@link pairedRiskDifference})
1392
- * instead of silently answering "no" forever.
1393
- *
1394
- * Empty input is NOT binary: there is no evidence of the outcome's shape, and
1395
- * defaulting an empty vector into the binary branch would pick a statistic on
1396
- * no data at all.
1397
- *
1398
- * NOT the right discriminator for a gate. It recognises the literal {0, 1}
1399
- * encoding and nothing else, so a pass/fail dimension emitted on 0-100 — which
1400
- * judges in this codebase do routinely — reads as non-binary, and a single
1401
- * partial-credit score in an otherwise pass/fail vector flips it to false while
1402
- * leaving the median just as blind. Gates want {@link pairedBinaryScale} (any
1403
- * two-point encoding). This predicate remains for callers that specifically
1404
- * mean "literally 0/1".
1405
- */
1406
- function isBinaryOutcomeVector(values) {
1407
- if (values.length === 0) return false;
1408
- for (let i = 0; i < values.length; i++) {
1409
- const v = values[i];
1410
- if (v !== 0 && v !== 1) return false;
1411
- }
1412
- return true;
1413
- }
1414
- /**
1415
- * McNemar's test for paired binary outcomes — the correct significance test for
1416
- * "does treatment change the success rate vs control on the SAME items". Only
1417
- * discordant pairs (one arm right, the other wrong) carry information; concordant
1418
- * pairs are uninformative, so a paired t-test / two-proportion z-test on the raw
1419
- * rates is wrong here. The p-value is exact: under H0 the b "treatment-wins" are
1420
- * Binomial(b + c, 0.5), so the two-sided p is the doubled binomial tail — correct
1421
- * at the small discordant counts typical of eval runs (no continuity-corrected
1422
- * chi-square approximation needed, though it is returned as `statistic` for
1423
- * reference). Inputs are paired 0/1 (or boolean) arrays, control first to match
1424
- * the module's (before, after) convention. Throws on unequal lengths.
1425
- */
1426
- function mcnemar(control, treatment) {
1427
- if (control.length !== treatment.length) throw new Error(`mcnemar: unequal sample sizes (${control.length} vs ${treatment.length})`);
1428
- const n = control.length;
1429
- let b = 0;
1430
- let c = 0;
1431
- for (let i = 0; i < n; i++) {
1432
- const ctrl = control[i] ? 1 : 0;
1433
- const treat = treatment[i] ? 1 : 0;
1434
- if (treat === 1 && ctrl === 0) b++;
1435
- else if (treat === 0 && ctrl === 1) c++;
1436
- }
1437
- const nDiscordant = b + c;
1438
- const statistic = nDiscordant === 0 ? 0 : (Math.abs(b - c) - 1) ** 2 / nDiscordant;
1439
- return {
1440
- n,
1441
- nDiscordant,
1442
- b,
1443
- c,
1444
- statistic,
1445
- pValue: binomialSignTwoSided(b, c)
1446
- };
1447
- }
1448
- /**
1449
- * Paired risk difference (the effect-size companion to {@link mcnemar}): the
1450
- * change in success rate p(treatment) − p(control) on matched items, which for
1451
- * paired binary data equals (b − c) / n. The CI uses the paired variance from
1452
- * the discordant counts, not the independent-samples formula (which overstates
1453
- * the interval by ignoring the pairing). Inputs are paired 0/1 (or boolean)
1454
- * arrays, control first. Throws on unequal lengths.
1455
- *
1456
- * REPORTING ONLY — do NOT decide a promotion on this interval. The CI is a Wald
1457
- * normal approximation, which badly UNDERCOVERS when only a handful of pairs are
1458
- * discordant: at n = 3 with b = 2, c = 0 it returns [0.133, 1.000], excluding 0,
1459
- * while McNemar's exact test on the same data gives p = 0.50. A gate keying on
1460
- * `lower > 0` would promote noise. Use {@link pairedRiskDifferenceExact}, whose
1461
- * interval is dual to the exact test by construction, for any decision.
1462
- */
1463
- function pairedRiskDifference(control, treatment, confidence = .95) {
1464
- if (control.length !== treatment.length) throw new Error(`pairedRiskDifference: unequal sample sizes (${control.length} vs ${treatment.length})`);
1465
- const n = control.length;
1466
- if (n === 0) return {
1467
- n: 0,
1468
- b: 0,
1469
- c: 0,
1470
- riskDifference: 0,
1471
- lower: 0,
1472
- upper: 0,
1473
- confidence
1474
- };
1475
- let b = 0;
1476
- let c = 0;
1477
- for (let i = 0; i < n; i++) {
1478
- const ctrl = control[i] ? 1 : 0;
1479
- const treat = treatment[i] ? 1 : 0;
1480
- if (treat === 1 && ctrl === 0) b++;
1481
- else if (treat === 0 && ctrl === 1) c++;
1482
- }
1483
- const rd = (b - c) / n;
1484
- const variance = (b + c - (b - c) ** 2 / n) / (n * n);
1485
- const half = zQuantile(1 - (1 - confidence) / 2) * Math.sqrt(Math.max(0, variance));
1486
- return {
1487
- n,
1488
- b,
1489
- c,
1490
- riskDifference: rd,
1491
- lower: Math.max(-1, rd - half),
1492
- upper: Math.min(1, rd + half),
1493
- confidence
1494
- };
1495
- }
1496
- /**
1497
- * Paired risk difference with the EXACT CONDITIONAL interval — the estimator a
1498
- * promotion gate may decide on.
1499
- *
1500
- * Conditional on the number of discordant pairs m = b + c, the treatment-win
1501
- * count b is Binomial(m, π) with π = P(treatment wins | discordant), and the
1502
- * risk difference is an exact reparameterisation: RD = (2π − 1)·m/n. So a
1503
- * Clopper-Pearson exact interval for π maps straight onto RD. This buys the
1504
- * property the Wald interval in {@link pairedRiskDifference} does not have:
1505
- *
1506
- * **`lower > 0` ⟺ McNemar's exact test rejects at α = 1 − confidence.**
1507
- *
1508
- * Clopper-Pearson excludes π = 0.5 exactly when the two-sided exact binomial
1509
- * test of π = 0.5 rejects, and that test IS {@link mcnemar}'s p-value — so the
1510
- * interval and the test can never disagree, and a gate keyed on `lower` cannot
1511
- * promote what the exact test refuses. The exact p is returned in the same
1512
- * object so the two are impossible to compute apart.
1513
- *
1514
- * The interval is conservative (exact intervals over-cover; conditioning on m
1515
- * discards the concordant pairs' information about m itself). That is the
1516
- * correct direction for a promotion gate: it refuses more often, never less.
1517
- *
1518
- * With m = 0 there are no discordant pairs and π is not identified: the result
1519
- * is the degenerate [0, 0] with p = 1. That is NOT evidence of equivalence —
1520
- * callers must treat a zero-width interval as "cannot decide", not as "no
1521
- * difference". Inputs are paired 0/1 (or boolean) arrays, control first.
1522
- * Throws on unequal lengths.
1523
- */
1524
- function pairedRiskDifferenceExact(control, treatment, confidence = .95) {
1525
- if (control.length !== treatment.length) throw new Error(`pairedRiskDifferenceExact: unequal sample sizes (${control.length} vs ${treatment.length})`);
1526
- if (confidence <= 0 || confidence >= 1) throw new Error(`pairedRiskDifferenceExact: confidence must be in (0,1), got ${confidence}`);
1527
- const n = control.length;
1528
- if (n === 0) return {
1529
- n: 0,
1530
- b: 0,
1531
- c: 0,
1532
- nDiscordant: 0,
1533
- riskDifference: 0,
1534
- lower: 0,
1535
- upper: 0,
1536
- confidence,
1537
- pValue: 1
1538
- };
1539
- let b = 0;
1540
- let c = 0;
1541
- for (let i = 0; i < n; i++) {
1542
- const ctrl = control[i] ? 1 : 0;
1543
- const treat = treatment[i] ? 1 : 0;
1544
- if (treat === 1 && ctrl === 0) b++;
1545
- else if (treat === 0 && ctrl === 1) c++;
1546
- }
1547
- const m = b + c;
1548
- const riskDifference = (b - c) / n;
1549
- const pValue = binomialSignTwoSided(b, c);
1550
- if (m === 0) return {
1551
- n,
1552
- b,
1553
- c,
1554
- nDiscordant: 0,
1555
- riskDifference: 0,
1556
- lower: 0,
1557
- upper: 0,
1558
- confidence,
1559
- pValue
1560
- };
1561
- const alpha = 1 - confidence;
1562
- const piLow = b === 0 ? 0 : betaQuantile(alpha / 2, b, m - b + 1);
1563
- const piHigh = b === m ? 1 : betaQuantile(1 - alpha / 2, b + 1, m - b);
1564
- const scale = m / n;
1565
- return {
1566
- n,
1567
- b,
1568
- c,
1569
- nDiscordant: m,
1570
- riskDifference,
1571
- lower: Math.max(-1, (2 * piLow - 1) * scale),
1572
- upper: Math.min(1, (2 * piHigh - 1) * scale),
1573
- confidence,
1574
- pValue
1575
- };
1576
- }
1577
- /** Inverse regularized incomplete beta by bisection on
1578
- * {@link regularizedIncompleteBeta}, which is monotone increasing in x. 80
1579
- * halvings of [0,1] resolve to ~8e-25, far past the continued fraction's own
1580
- * 3e-15 tolerance, so the quantile is as exact as the CDF it inverts. */
1581
- function betaQuantile(p, a, b) {
1582
- if (p <= 0) return 0;
1583
- if (p >= 1) return 1;
1584
- let lo = 0;
1585
- let hi = 1;
1586
- for (let i = 0; i < 80; i++) {
1587
- const mid = (lo + hi) / 2;
1588
- if (regularizedIncompleteBeta(mid, a, b) < p) lo = mid;
1589
- else hi = mid;
1590
- }
1591
- return (lo + hi) / 2;
1592
- }
1593
- /**
1594
- * Constrained MLE of q = P(treatment loses) under the hypothesis RD = `delta`.
1595
- *
1596
- * Profiling the two concordant cells out of the multinomial leaves
1597
- * `L(q) = b·log(q+delta) + c·log(q) + e·log(1 − 2q − delta)` with `e = n − b − c`,
1598
- * whose stationary point is the positive root of
1599
- * `2n·q² − [(b + c) − delta·(b + 3c + 2e)]·q − c·delta·(1 − delta) = 0`.
1600
- * At `delta = 0` this returns `(b + c) / 2n`, the familiar null.
1601
- */
1602
- function constrainedLossRate(b, c, n, delta) {
1603
- const e = n - b - c;
1604
- const quadratic = 2 * n;
1605
- const linear = -(b + c - delta * (b + 3 * c + 2 * e));
1606
- const constant = -c * delta * (1 - delta);
1607
- const discriminant = linear * linear - 4 * quadratic * constant;
1608
- const root = discriminant > 0 ? Math.sqrt(discriminant) : 0;
1609
- const q = (-linear + root) / (2 * quadratic);
1610
- return Math.min(Math.max(q, Math.max(0, -delta)), Math.max(0, (1 - delta) / 2));
1611
- }
1612
- /** Tango's score statistic for H0: RD = `delta`. `Var(b − c) = n·(2q + delta −
1613
- * delta²)` under that hypothesis, evaluated at the constrained MLE of q. */
1614
- function tangoScore(b, c, n, delta) {
1615
- const numerator = b - c - n * delta;
1616
- const variance = n * (2 * constrainedLossRate(b, c, n, delta) + delta * (1 - delta));
1617
- if (!(variance > 0)) {
1618
- if (numerator === 0) return 0;
1619
- return numerator > 0 ? Number.POSITIVE_INFINITY : Number.NEGATIVE_INFINITY;
1620
- }
1621
- return numerator / Math.sqrt(variance);
1622
- }
1623
- /**
1624
- * Paired risk difference with TANGO'S (1998) SCORE INTERVAL — the estimator a
1625
- * promotion gate may decide on **at a nonzero margin**.
1626
- *
1627
- * {@link pairedRiskDifferenceExact} conditions on the observed discordant count
1628
- * `m = b + c`, builds a Clopper-Pearson interval for the win share among those
1629
- * `m` pairs, and multiplies by the observed `m/n`. That is exact for testing
1630
- * RD = 0 — it is dual to McNemar — but it is NOT a confidence interval for the
1631
- * population risk difference at a nonzero margin, because the sampling
1632
- * variability of `m/n` itself is discarded. The gap is not academic: with the
1633
- * production caller's `pairedDeltaThreshold: -0.05`, a process whose true risk
1634
- * difference sits exactly on that margin clears a nominal-95 % `lower > margin`
1635
- * check 24.75 % of the time at n = 40 and 43.95 % at n = 76 (2000 replicates
1636
- * each) when the conditional interval decides.
1637
- *
1638
- * Tango's interval inverts the score test of RD = delta, which estimates the
1639
- * nuisance loss rate under each hypothesised delta instead of fixing it at the
1640
- * observed value, so `m` contributes its own uncertainty. It is the method
1641
- * `ratesci::scorepairci` uses for paired risk-difference noninferiority, and it
1642
- * is not conditional, so it stays valid as the margin moves away from zero.
1643
- *
1644
- * The bounds are found by bisecting `tangoScore(delta) = ±z` — the score is
1645
- * monotone decreasing in delta, so each crossing is unique. Inputs are paired
1646
- * 0/1 (or boolean) arrays, control first. Throws on unequal lengths.
1647
- */
1648
- function pairedRiskDifferenceScore(control, treatment, confidence = .95) {
1649
- if (control.length !== treatment.length) throw new Error(`pairedRiskDifferenceScore: unequal sample sizes (${control.length} vs ${treatment.length})`);
1650
- if (confidence <= 0 || confidence >= 1) throw new Error(`pairedRiskDifferenceScore: confidence must be in (0,1), got ${confidence}`);
1651
- const n = control.length;
1652
- if (n === 0) return {
1653
- n: 0,
1654
- b: 0,
1655
- c: 0,
1656
- nDiscordant: 0,
1657
- riskDifference: 0,
1658
- lower: -1,
1659
- upper: 1,
1660
- confidence
1661
- };
1662
- let b = 0;
1663
- let c = 0;
1664
- for (let i = 0; i < n; i++) {
1665
- const ctrl = control[i] ? 1 : 0;
1666
- const treat = treatment[i] ? 1 : 0;
1667
- if (treat === 1 && ctrl === 0) b++;
1668
- else if (treat === 0 && ctrl === 1) c++;
1669
- }
1670
- const riskDifference = (b - c) / n;
1671
- const z = zQuantile(1 - (1 - confidence) / 2);
1672
- let lo = -1;
1673
- let hi = riskDifference;
1674
- for (let i = 0; i < 200; i++) {
1675
- const mid = (lo + hi) / 2;
1676
- if (tangoScore(b, c, n, mid) > z) lo = mid;
1677
- else hi = mid;
1678
- }
1679
- const lower = (lo + hi) / 2;
1680
- let ulo = riskDifference;
1681
- let uhi = 1;
1682
- for (let i = 0; i < 200; i++) {
1683
- const mid = (ulo + uhi) / 2;
1684
- if (tangoScore(b, c, n, mid) > -z) ulo = mid;
1685
- else uhi = mid;
1686
- }
1687
- const upper = (ulo + uhi) / 2;
1688
- return {
1689
- n,
1690
- b,
1691
- c,
1692
- nDiscordant: b + c,
1693
- riskDifference,
1694
- lower: Math.max(-1, lower),
1695
- upper: Math.min(1, upper),
1696
- confidence
1697
- };
1698
- }
1699
- /**
1700
- * The common positive level `s` such that EVERY value across both paired arms is
1701
- * exactly 0 or `s` — i.e. the outcome is pass/fail, whatever encoding it arrived
1702
- * in. Returns null when the outcomes are not two-point, when the two arms use
1703
- * different levels, or when no positive value was observed at all (all-zero
1704
- * arms: the level is not identified, and there is nothing to decide anyway).
1705
- *
1706
- * This is the scale-aware successor to {@link isBinaryOutcomeVector}, which only
1707
- * recognises literal {0, 1}. Judges in this codebase emit dimensions on 0-100 as
1708
- * well as [0,1] (see `detectScale` in `campaign/gates/statistical-heldout.ts`),
1709
- * so a pass/fail dimension routinely arrives as {0, 100} and a {0,1}-only test
1710
- * silently sends it down the median path that cannot see it. Any positive level
1711
- * is accepted, not just 1 and 100: for a two-point {0, s} outcome the mean paired
1712
- * delta is exactly s·(b − c)/n, so the binary estimators apply after dividing by
1713
- * s and rescaling the result back into the caller's native units.
1714
- *
1715
- * Non-finite values ⇒ null: an unusable outcome must not be classified as a
1716
- * clean pass/fail shape.
1717
- */
1718
- function pairedBinaryScale(before, after) {
1719
- let level = null;
1720
- for (const arm of [before, after]) for (let i = 0; i < arm.length; i++) {
1721
- const v = arm[i];
1722
- if (!Number.isFinite(v)) return null;
1723
- if (v === 0) continue;
1724
- if (v < 0) return null;
1725
- if (level === null) level = v;
1726
- else if (v !== level) return null;
1727
- }
1728
- return level;
1729
- }
1730
- /** Fraction of paired observations whose delta is an exact tie (|after − before|
1731
- * < 1e-9). Throws on unequal sample sizes; 0 pairs ⇒ 0. */
1732
- function pairedDeltaTieFraction(before, after) {
1733
- if (before.length !== after.length) throw new Error(`pairedDeltaTieFraction: unequal sample sizes (${before.length} vs ${after.length})`);
1734
- const n = before.length;
1735
- if (n === 0) return 0;
1736
- let ties = 0;
1737
- for (let i = 0; i < n; i++) if (Math.abs(after[i] - before[i]) < 1e-9) ties++;
1738
- return ties / n;
1739
- }
1740
- /**
1741
- * The paired-delta statistic a DECISION is computed on, package-wide.
1742
- *
1743
- * The mean paired delta is the estimator that answers the question a promotion
1744
- * gate asks — "by how much did the candidate move the score" — in the caller's
1745
- * own units, and it equals the aggregate lift everyone quotes. The MEDIAN
1746
- * answers a different question and loses the answer to this one in every regime
1747
- * eval data actually lands in:
1748
- * - TWO-POINT (pass/fail) outcomes on any encoding: the delta vector lives in
1749
- * {−s, 0, +s} dominated by zeros, so the median and its whole bootstrap CI
1750
- * are pinned at exactly 0 however large the shift. (Decide these on
1751
- * {@link pairedRiskDifferenceExact} instead — same estimand, exact interval.)
1752
- * - TIE-DOMINATED outcomes: at half the pairs tied the sample median is 0 by
1753
- * construction, and `ci.low > threshold` then answers "no" forever at a
1754
- * non-negative threshold and "yes" forever at a negative one.
1755
- * - LOW-CARDINALITY outcomes, even well below half ties: judge dimensions on
1756
- * integer 0-100, and block scores like {⅔, 1} from averaging pass/fail
1757
- * leaves, put the median on a coarse lattice whose bootstrap percentiles
1758
- * land on atoms. Measured: 26 blocks of 3 pass/fail leaves carrying a real
1759
- * +12.8pp lift, only 23% of pairs tied, gives a median CI of [0, 0.333] —
1760
- * lower bound exactly 0, so a gate at threshold 0 refuses a real lift.
1761
- * That last case is why there is no tie-fraction threshold here: any cutoff on
1762
- * ties leaves the lattice case open on the other side of it.
1763
- *
1764
- * `heldoutSignificance` has defaulted to the mean since #316 for the same
1765
- * reason. The median remains available per call site for callers who
1766
- * specifically want outlier robustness and accept the blindness.
1767
- */
1768
- const DECISION_PAIRED_DELTA_STATISTIC = "mean";
1769
- /**
1770
- * Unbiased pass@k for code generation (Chen et al. 2021, "Evaluating Large
1771
- * Language Models Trained on Code"). Given `n` independent samples for one
1772
- * problem of which `c` pass, the probability that at least one of a random k of
1773
- * them passes is 1 − C(n−c, k) / C(n, k). Estimating pass@k as "did any of the
1774
- * first k pass" is biased high at small n; this is the variance-reduced estimator
1775
- * averaged implicitly over all k-subsets. Average the per-problem values across
1776
- * the suite for the corpus pass@k. Computed in the numerically stable product
1777
- * form. Requires 1 ≤ k ≤ n and 0 ≤ c ≤ n.
1778
- */
1779
- function passAtK(n, c, k) {
1780
- if (!Number.isInteger(n) || !Number.isInteger(c) || !Number.isInteger(k)) throw new Error(`passAtK: n, c, k must be integers (got n=${n}, c=${c}, k=${k})`);
1781
- if (k < 1 || k > n || c < 0 || c > n) throw new Error(`passAtK: require 1 ≤ k ≤ n and 0 ≤ c ≤ n (got n=${n}, c=${c}, k=${k})`);
1782
- if (n - c < k) return 1;
1783
- let prob = 1;
1784
- for (let i = n - c + 1; i <= n; i++) prob *= 1 - k / i;
1785
- return 1 - prob;
1786
- }
1787
- /**
1788
- * Two-sided exact p-value for b successes out of (b + c) Bernoulli(0.5) trials —
1789
- * the exact-binomial core of {@link mcnemar}. `min(1, 2·P(X ≤ min(b,c)))`. No
1790
- * discordant pairs ⇒ no evidence ⇒ p = 1. Summed in log space (lnGamma) so it
1791
- * stays exact at large discordant counts without overflow.
1792
- */
1793
- function binomialSignTwoSided(b, c) {
1794
- const nd = b + c;
1795
- if (nd === 0) return 1;
1796
- return Math.min(1, 2 * binomialHalfLowerTail(Math.min(b, c), nd));
1797
- }
1798
- /** P(X >= successes) for X ~ Binomial(n, 0.5). */
1799
- function binomialHalfUpperTail(successes, n) {
1800
- if (successes <= 0) return 1;
1801
- if (successes > n) return 0;
1802
- if (successes <= n / 2) return Math.max(0, 1 - binomialHalfLowerTail(successes - 1, n));
1803
- return binomialHalfLowerTail(n - successes, n);
1804
- }
1805
- /** P(X <= maxSuccesses) for X ~ Binomial(n, 0.5), accumulated in log space. */
1806
- function binomialHalfLowerTail(maxSuccesses, n) {
1807
- if (maxSuccesses < 0) return 0;
1808
- if (maxSuccesses >= n) return 1;
1809
- if (maxSuccesses === 0) return 2 ** -n;
1810
- const logHalfN = n * Math.log(.5);
1811
- let logTail = Number.NEGATIVE_INFINITY;
1812
- for (let i = 0; i <= maxSuccesses; i++) {
1813
- const logChoose = lnGamma(n + 1) - lnGamma(i + 1) - lnGamma(n - i + 1);
1814
- logTail = logAddExp(logTail, logChoose + logHalfN);
1815
- }
1816
- return Math.min(1, Math.exp(logTail));
1817
- }
1818
- function logAddExp(a, b) {
1819
- if (a === Number.NEGATIVE_INFINITY) return b;
1820
- if (b === Number.NEGATIVE_INFINITY) return a;
1821
- const max = Math.max(a, b);
1822
- return max + Math.log1p(Math.exp(Math.min(a, b) - max));
1823
- }
1824
- /**
1825
- * Betting test-martingale for bounded observations — the e-process core of
1826
- * anytime-valid sequential testing (Waudby-Smith & Ramdas, "Estimating means
1827
- * of bounded random variables by betting", JRSS-B 2024).
1828
- *
1829
- * Observations x_i ∈ [0,1]; H0: E[x] ≤ m₀ (`nullMean`, default 1/2). Wealth
1830
- *
1831
- * W_t = Π_{i≤t} (1 + λ_i (x_i − m₀)), W_0 = 1
1832
- *
1833
- * with the truncated GROW-style plug-in bet computed from PRIOR observations:
1834
- *
1835
- * λ_i = clamp((μ̂_{i−1} − m₀) / (σ̂²_{i−1} + (μ̂_{i−1} − m₀)²), 0, maxBet)
1836
- *
1837
- * where μ̂/σ̂² are the shrunk running estimates μ̂_t = (1/2 + Σx_i)/(t+1),
1838
- * σ̂²_t = (1/4 + Σ(x_i − μ̂_i)²)/(t+1).
1839
- *
1840
- * PREDICTABILITY INVARIANT (load-bearing): λ_i is a function of x_1..x_{i−1}
1841
- * ONLY — it may never see x_i. With λ_i ≥ 0 predictable, each factor has
1842
- * E[1 + λ_i(x_i − m₀) | past] ≤ 1 under H0, so W is a nonnegative
1843
- * supermartingale and Ville's inequality gives P(∃t: W_t ≥ 1/α) ≤ α — the
1844
- * type-I guarantee holds at ANY data-dependent stopping time. λ_1 is always 0
1845
- * (no prior evidence), so the first observation never moves wealth.
1846
- *
1847
- * `decided` latches at the first crossing W_t ≥ 1/α and never un-latches;
1848
- * wealth keeps updating after the crossing (the e-process remains valid), but
1849
- * the decision time is the first crossing.
1850
- */
1851
- function eProcess(opts = {}) {
1852
- const alpha = opts.alpha ?? .05;
1853
- const maxBet = opts.maxBet ?? .5;
1854
- const nullMean = opts.nullMean ?? .5;
1855
- if (!Number.isFinite(alpha) || alpha <= 0 || alpha >= 1) throw new ValidationError(`eProcess: alpha must be in (0,1), got ${alpha}`);
1856
- if (!Number.isFinite(nullMean) || nullMean <= 0 || nullMean >= 1) throw new ValidationError(`eProcess: nullMean must be in (0,1), got ${nullMean}`);
1857
- if (!Number.isFinite(maxBet) || maxBet <= 0 || maxBet >= 1 / nullMean) throw new ValidationError(`eProcess: maxBet must be in (0, 1/nullMean=${(1 / nullMean).toFixed(4)}) so wealth factors stay positive, got ${maxBet}`);
1858
- const threshold = 1 / alpha;
1859
- let wealth = 1;
1860
- let n = 0;
1861
- let decided = false;
1862
- let decidedAtN;
1863
- let sumX = 0;
1864
- let varSum = 0;
1865
- return {
1866
- update(x) {
1867
- if (typeof x !== "number" || !Number.isFinite(x) || x < 0 || x > 1) throw new ValidationError(`eProcess: observation must be a finite number in [0,1], got ${x}`);
1868
- const muPrev = (.5 + sumX) / (n + 1);
1869
- const varPrev = (.25 + varSum) / (n + 1);
1870
- const edge = muPrev - nullMean;
1871
- const lambda = Math.min(maxBet, Math.max(0, edge / (varPrev + edge * edge)));
1872
- wealth *= 1 + lambda * (x - nullMean);
1873
- n += 1;
1874
- sumX += x;
1875
- const muNow = (.5 + sumX) / (n + 1);
1876
- varSum += (x - muNow) ** 2;
1877
- if (!decided && wealth >= threshold) {
1878
- decided = true;
1879
- decidedAtN = n;
1880
- }
1881
- return {
1882
- wealth,
1883
- n,
1884
- decided
1885
- };
1886
- },
1887
- state() {
1888
- return {
1889
- wealth,
1890
- n,
1891
- decided,
1892
- alpha,
1893
- maxBet,
1894
- nullMean,
1895
- threshold,
1896
- decidedAtN
1897
- };
1898
- }
1899
- };
1900
- }
1901
- /** Every rank test refuses non-finite input. Beyond the arithmetic being
1902
- * undefined, the tie-grouping scan compares values with `===`, and
1903
- * `NaN === NaN` is false, so a NaN would leave the group boundary unable to
1904
- * advance and spin the loop forever. */
1905
- function assertFiniteSample(fn, label, xs) {
1906
- for (let i = 0; i < xs.length; i++) if (!Number.isFinite(xs[i])) throw new ValidationError(`${fn}: ${label}[${i}] must be finite, got ${xs[i]}`);
1907
- }
1908
- /**
1909
- * Average ranks over an ASCENDING-sorted array, plus `Σ(t³ − t)` over tie
1910
- * groups of size `t` — the correction term both asymptotic rank-test variances
1911
- * need.
1912
- */
1913
- function midranksWithTieTerm(sorted) {
1914
- const midranks = new Array(sorted.length);
1915
- let tieTerm = 0;
1916
- let i = 0;
1917
- while (i < sorted.length) {
1918
- let j = i;
1919
- while (j < sorted.length && sorted[j] === sorted[i]) j++;
1920
- const average = (i + 1 + j) / 2;
1921
- for (let k = i; k < j; k++) midranks[k] = average;
1922
- const groupSize = j - i;
1923
- if (groupSize > 1) tieTerm += groupSize ** 3 - groupSize;
1924
- i = j;
1925
- }
1926
- return {
1927
- midranks,
1928
- tieTerm
1929
- };
1930
- }
1931
- function selectRankTestMethod(fn, request, design, exactFeasible, designFloor, threshold) {
1932
- if (request === "auto") return exactFeasible ? "exact" : "permutation";
1933
- if (request === "exact") {
1934
- if (exactFeasible) return "exact";
1935
- throw new ValidationError(`${fn}: method 'exact' is out of range at ${design} — enumeration is bounded by ${threshold}. Use 'auto' for the seeded Monte Carlo permutation, which converges to the same answer.`);
1936
- }
1937
- if (exactFeasible) throw new ValidationError(`${fn}: method 'asymptotic' is refused at ${design} — the exact p-grid at this design starts at ${formatProbability(designFloor)}, so an asymptotic p below it describes no attainable outcome. Use method 'exact' (the default) or add repetitions past ${threshold}.`);
1938
- return "asymptotic";
1939
- }
1940
- function resolvePermutations(fn, permutations) {
1941
- if (permutations === void 0) return DEFAULT_PERMUTATIONS;
1942
- if (!Number.isInteger(permutations) || permutations < 1) throw new ValidationError(`${fn}: permutations must be a positive integer, got ${permutations}`);
1943
- return permutations;
1944
- }
1945
- /** Two-sided normal-approximation tail with the continuity correction. */
1946
- function asymptoticTwoSidedP(deviation, sigma) {
1947
- if (!(sigma > 0)) return 1;
1948
- return Math.min(1, 2 * (1 - normalCdf(Math.max(0, deviation - .5) / sigma)));
1949
- }
1950
- /** SD of U under the permutation null, corrected for the realised ties. The
1951
- * tie term reduces (N+1) and reaches it exactly when every value is tied, so
1952
- * the variance floors at 0 rather than going negative. */
1953
- function twoSampleSigma(n1, n2, total, tieTerm) {
1954
- if (total < 2) return 0;
1955
- const variance = n1 * n2 / 12 * (total + 1 - tieTerm / (total * (total - 1)));
1956
- return Math.sqrt(Math.max(0, variance));
1957
- }
1958
- function logChoose(n, k) {
1959
- return lnGamma(n + 1) - lnGamma(k + 1) - lnGamma(n - k + 1);
1960
- }
1961
- function formatProbability(value) {
1962
- return value >= 1e-4 || value === 0 ? value.toFixed(4) : value.toExponential(3);
1963
- }
1964
- /**
1965
- * Exact DP allocation and loop count for this observed rank vector.
1966
- *
1967
- * The smaller arm is sufficient because selecting its complement produces the
1968
- * same two-sided U deviation while using fewer rows in the state table.
1969
- */
1970
- function exactTwoSampleCost(doubledRanks, selectedN) {
1971
- const maxSum = doubledRanks.reduce((sum, rank) => sum + rank, 0);
1972
- let work = 0;
1973
- for (let placed = 0; placed < doubledRanks.length; placed++) work += Math.min(selectedN, placed + 1) * (maxSum - doubledRanks[placed] + 1);
1974
- return {
1975
- states: (selectedN + 1) * (maxSum + 1),
1976
- work
1977
- };
1978
- }
1979
- /**
1980
- * Smallest attainable two-sided p under the observed ties.
1981
- *
1982
- * Only subsets with the minimum or maximum rank sum can attain the largest
1983
- * deviation. Their multiplicity is the number of ways to choose within the
1984
- * tie group at each boundary, so this calculation is exact without allocating
1985
- * the full null distribution.
1986
- */
1987
- function exactTwoSampleFloor(doubledRanks, selectedN) {
1988
- const total = doubledRanks.length;
1989
- const otherN = total - selectedN;
1990
- const minimumSum = doubledRanks.slice(0, selectedN).reduce((sum, rank) => sum + rank, 0);
1991
- const maximumSum = doubledRanks.slice(total - selectedN).reduce((sum, rank) => sum + rank, 0);
1992
- if (minimumSum === maximumSum) return 1;
1993
- const centre = selectedN * (selectedN + 1) + selectedN * otherN;
1994
- const minimumDeviation = Math.abs(minimumSum - centre);
1995
- const maximumDeviation = Math.abs(maximumSum - centre);
1996
- const totalLogWays = logChoose(total, selectedN);
1997
- const minimumMass = Math.exp(logExtremeSubsetWays(doubledRanks, selectedN, "minimum") - totalLogWays);
1998
- const maximumMass = Math.exp(logExtremeSubsetWays(doubledRanks, selectedN, "maximum") - totalLogWays);
1999
- if (minimumDeviation > maximumDeviation) return minimumMass;
2000
- if (maximumDeviation > minimumDeviation) return maximumMass;
2001
- return Math.min(1, minimumMass + maximumMass);
2002
- }
2003
- function logExtremeSubsetWays(sortedRanks, selectedN, side) {
2004
- const boundaryIndex = side === "minimum" ? selectedN - 1 : sortedRanks.length - selectedN;
2005
- const boundary = sortedRanks[boundaryIndex];
2006
- let first = boundaryIndex;
2007
- let afterLast = boundaryIndex + 1;
2008
- while (first > 0 && sortedRanks[first - 1] === boundary) first--;
2009
- while (afterLast < sortedRanks.length && sortedRanks[afterLast] === boundary) afterLast++;
2010
- return logChoose(afterLast - first, selectedN - (side === "minimum" ? first : sortedRanks.length - afterLast));
2011
- }
2012
- /**
2013
- * Exact conditional two-sided p for the two-sample rank test.
2014
- *
2015
- * Convolves the observed doubled midranks into the null distribution of group
2016
- * a's rank sum over every `C(n₁+n₂, n₁)` split — identical to enumerating the
2017
- * splits, but `O(N·n₁·ΣR)` instead of `O(C(N,n₁)·n₁n₂)`. Conditioning on the
2018
- * realised multiset makes the tie handling exact rather than a correction.
2019
- *
2020
- * The null is symmetric about `n₁n₂/2` (negating every value maps `U → n₁n₂ −
2021
- * U` and permutes the split set onto itself), so the two-sided p is the mass
2022
- * at least as far from the centre as the observation.
2023
- */
2024
- function exactTwoSampleP(doubledRanks, n1, n2, doubledDeviation) {
2025
- const maxSum = doubledRanks.reduce((sum, rank) => sum + rank, 0);
2026
- const width = maxSum + 1;
2027
- const ways = Array.from({ length: n1 + 1 }, () => new Float64Array(width));
2028
- ways[0][0] = 1;
2029
- let placed = 0;
2030
- for (const rank of doubledRanks) {
2031
- for (let k = Math.min(n1, placed + 1); k >= 1; k--) {
2032
- const from = ways[k - 1];
2033
- const into = ways[k];
2034
- for (let sum = maxSum - rank; sum >= 0; sum--) {
2035
- const count = from[sum];
2036
- if (count !== 0) into[sum + rank] += count;
2037
- }
2038
- }
2039
- placed++;
2040
- }
2041
- const shift = n1 * (n1 + 1) + n1 * n2;
2042
- const chosen = ways[n1];
2043
- let totalWays = 0;
2044
- let extremeWays = 0;
2045
- let tailWays = 0;
2046
- let maxDeviation = -1;
2047
- for (let sum = 0; sum < width; sum++) {
2048
- const count = chosen[sum];
2049
- if (count === 0) continue;
2050
- totalWays += count;
2051
- const deviation = Math.abs(sum - shift);
2052
- if (deviation >= doubledDeviation) tailWays += count;
2053
- if (deviation > maxDeviation) {
2054
- maxDeviation = deviation;
2055
- extremeWays = count;
2056
- } else if (deviation === maxDeviation) extremeWays += count;
2057
- }
2058
- return {
2059
- p: tailWays / totalWays,
2060
- pFloor: extremeWays / totalWays
2061
- };
2062
- }
2063
- /**
2064
- * Exact conditional two-sided p for the paired signed-rank test.
2065
- *
2066
- * Convolves the observed doubled absolute midranks over all `2ⁿ` sign
2067
- * assignments in `O(n·ΣR)`. Probabilities rather than counts keep `2ⁿ` off the
2068
- * arithmetic. The null is symmetric about `n(n+1)/4`.
2069
- */
2070
- function exactSignedRankP(doubledRanks, doubledDeviation) {
2071
- const maxSum = doubledRanks.reduce((sum, rank) => sum + rank, 0);
2072
- const width = maxSum + 1;
2073
- let mass = new Float64Array(width);
2074
- mass[0] = 1;
2075
- for (const rank of doubledRanks) {
2076
- const next = new Float64Array(width);
2077
- for (let sum = 0; sum < width; sum++) {
2078
- const probability = mass[sum];
2079
- if (probability === 0) continue;
2080
- next[sum] += probability * .5;
2081
- next[sum + rank] += probability * .5;
2082
- }
2083
- mass = next;
2084
- }
2085
- const centre = maxSum / 2;
2086
- let tail = 0;
2087
- let extreme = 0;
2088
- let maxDeviation = -1;
2089
- for (let sum = 0; sum < width; sum++) {
2090
- const probability = mass[sum];
2091
- if (probability === 0) continue;
2092
- const deviation = Math.abs(sum - centre);
2093
- if (deviation >= doubledDeviation) tail += probability;
2094
- if (deviation > maxDeviation) {
2095
- maxDeviation = deviation;
2096
- extreme = probability;
2097
- } else if (deviation === maxDeviation) extreme += probability;
2098
- }
2099
- return {
2100
- p: Math.min(1, tail),
2101
- pFloor: Math.min(1, extreme)
2102
- };
2103
- }
2104
- /** Standard-normal inverse CDF (Acklam approximation). */
2105
- function zQuantile(p) {
2106
- if (p <= 0 || p >= 1) {
2107
- if (p === 0) return -Infinity;
2108
- if (p === 1) return Infinity;
2109
- return NaN;
2110
- }
2111
- const a = [
2112
- -39.69683028665376,
2113
- 220.9460984245205,
2114
- -275.9285104469687,
2115
- 138.357751867269,
2116
- -30.66479806614716,
2117
- 2.506628277459239
2118
- ];
2119
- const b = [
2120
- -54.47609879822406,
2121
- 161.5858368580409,
2122
- -155.6989798598866,
2123
- 66.80131188771972,
2124
- -13.28068155288572
2125
- ];
2126
- const c = [
2127
- -.007784894002430293,
2128
- -.3223964580411365,
2129
- -2.400758277161838,
2130
- -2.549732539343734,
2131
- 4.374664141464968,
2132
- 2.938163982698783
2133
- ];
2134
- const d = [
2135
- .007784695709041462,
2136
- .3224671290700398,
2137
- 2.445134137142996,
2138
- 3.754408661907416
2139
- ];
2140
- const pLow = .02425;
2141
- const pHigh = 1 - pLow;
2142
- let q;
2143
- let r;
2144
- if (p < pLow) {
2145
- q = Math.sqrt(-2 * Math.log(p));
2146
- return (((((c[0] * q + c[1]) * q + c[2]) * q + c[3]) * q + c[4]) * q + c[5]) / ((((d[0] * q + d[1]) * q + d[2]) * q + d[3]) * q + 1);
2147
- }
2148
- if (p <= pHigh) {
2149
- q = p - .5;
2150
- r = q * q;
2151
- return (((((a[0] * r + a[1]) * r + a[2]) * r + a[3]) * r + a[4]) * r + a[5]) * q / (((((b[0] * r + b[1]) * r + b[2]) * r + b[3]) * r + b[4]) * r + 1);
2152
- }
2153
- q = Math.sqrt(-2 * Math.log(1 - p));
2154
- return -(((((c[0] * q + c[1]) * q + c[2]) * q + c[3]) * q + c[4]) * q + c[5]) / ((((d[0] * q + d[1]) * q + d[2]) * q + d[3]) * q + 1);
2155
- }
2156
- function medianInPlace(xs) {
2157
- if (xs.length === 0) return 0;
2158
- xs.sort((a, b) => a - b);
2159
- const mid = Math.floor(xs.length / 2);
2160
- return xs.length % 2 === 0 ? (xs[mid - 1] + xs[mid]) / 2 : xs[mid];
2161
- }
2162
- /**
2163
- * PRNG for every resampling path in this module.
2164
- *
2165
- * With no caller seed the seed is DERIVED FROM THE DATA rather than taken from
2166
- * `Math.random`, so re-running the same input reproduces the same interval —
2167
- * a gate verdict that cannot be re-derived is not evidence. Distinct data
2168
- * still gets a distinct stream. Same pattern as `promotion-gate.ts`.
2169
- */
2170
- function makeRng(seed, ...series) {
2171
- return mulberry32(seed ?? seedFromData(series));
2172
- }
2173
- /** FNV-1a over the IEEE-754 bytes of every observation. */
2174
- function seedFromData(series) {
2175
- const view = /* @__PURE__ */ new DataView(/* @__PURE__ */ new ArrayBuffer(8));
2176
- let hash = 2166136261;
2177
- for (const xs of series) {
2178
- for (const x of xs) {
2179
- view.setFloat64(0, x);
2180
- for (let byte = 0; byte < 8; byte++) hash = Math.imul(hash ^ view.getUint8(byte), 16777619);
2181
- }
2182
- hash = Math.imul(hash ^ 255, 16777619);
2183
- }
2184
- return hash | 0;
2185
- }
2186
- /** Order-independent seed for a symmetric two-sample statistic. */
2187
- function symmetricTwoSampleSeed(a, b) {
2188
- const sortedA = [...a].sort((left, right) => left - right);
2189
- const sortedB = [...b].sort((left, right) => left - right);
2190
- const forward = seedFromData([sortedA, sortedB]) >>> 0;
2191
- const reversed = seedFromData([sortedB, sortedA]) >>> 0;
2192
- return Math.min(forward, reversed);
2193
- }
2194
- /** Tiny seedable PRNG (mulberry32) — deterministic resampling/shuffling, not
2195
- * cryptographic. Exported so e-process shuffles and bootstrap resampling
2196
- * share ONE PRNG implementation. Every distinct 32-bit seed gives a distinct
2197
- * stream, including 0. */
2198
- function mulberry32(seed) {
2199
- if (!Number.isFinite(seed)) throw new ValidationError(`mulberry32: seed must be a finite number, got ${seed}`);
2200
- let s = seed | 0;
2201
- return () => {
2202
- s = s + 1831565813 | 0;
2203
- let t = s;
2204
- t = Math.imul(t ^ t >>> 15, t | 1);
2205
- t ^= t + Math.imul(t ^ t >>> 7, t | 61);
2206
- return ((t ^ t >>> 14) >>> 0) / 4294967296;
2207
- };
2208
- }
2209
- //#endregion
2210
- export { selfPreference as $, pairedRiskDifference as A, requiredSampleSize as B, mulberry32 as C, pairedCohensDz as D, pairedBootstrap as E, partialCredit as F, wilson as G, weightedComposite as H, passAtK as I, normalCdf as J, studentTCdf as K, pearsonR as L, pairedRiskDifferenceScore as M, pairedSignTest as N, pairedDeltaTieFraction as O, pairedTTest as P, positionalBias as Q, ranks as R, mcnemarRequiredN as S, pairedBinaryScale as T, weightedMean as U, spearmanR as V, wilcoxonSignedRank as W, calibrateJudgeContinuous as X, calibrateJudge as Y, continuousAgreement as Z, interpretCliffs as _, MANN_WHITNEY_EXACT_MAX_WORK as a, mcnemar as b, bonferroni as c, confidenceInterval as d, verbosityBias as et, corpusInterRaterAgreement as f, interRaterReliability as g, holm as h, MANN_WHITNEY_EXACT_MAX_STATES as i, pairedRiskDifferenceExact as j, pairedMde as k, cliffsDelta as l, eProcess as m, DECISION_PAIRED_DELTA_STATISTIC as n, WILCOXON_EXACT_MAX_N as o, corpusInterRaterAgreementFromJudgeScores as p, studentTQuantile as q, DEFAULT_PERMUTATIONS as r, benjaminiHochberg as s, BOOTSTRAP_GATE_MIN_N as t, cohensD as u, isBinaryOutcomeVector as v, normalizeScores as w, mcnemarPower as x, mannWhitneyU as y, requiredPairedSampleSize as z };
2211
-
2212
- //# sourceMappingURL=statistics-ByxzSiOM.js.map