@tangle-network/agent-eval 0.128.2 → 0.130.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (424) hide show
  1. package/CHANGELOG.md +279 -0
  2. package/README.md +19 -0
  3. package/dist/active-curriculum-C4mk67HP.js +214 -0
  4. package/dist/active-curriculum-C4mk67HP.js.map +1 -0
  5. package/dist/adversarial-smnADNFS.d.ts +21 -0
  6. package/dist/adversarial-smnADNFS.d.ts.map +1 -0
  7. package/dist/analyst/index.d.ts +83 -2932
  8. package/dist/analyst/index.d.ts.map +1 -0
  9. package/dist/analyst/index.js +319 -364
  10. package/dist/analyst/index.js.map +1 -1
  11. package/dist/analyst-BkTS3C58.d.ts +89 -0
  12. package/dist/analyst-BkTS3C58.d.ts.map +1 -0
  13. package/dist/analyst-LsnNpSkm.js +152 -0
  14. package/dist/analyst-LsnNpSkm.js.map +1 -0
  15. package/dist/analyze-runs-C1CavBMk.js +1067 -0
  16. package/dist/analyze-runs-C1CavBMk.js.map +1 -0
  17. package/dist/analyze-runs-FsgYCinh.d.ts +72 -0
  18. package/dist/analyze-runs-FsgYCinh.d.ts.map +1 -0
  19. package/dist/attribute-vocabulary-DLJ6303h.d.ts +54 -0
  20. package/dist/attribute-vocabulary-DLJ6303h.d.ts.map +1 -0
  21. package/dist/authenticity/index.d.ts +81 -79
  22. package/dist/authenticity/index.d.ts.map +1 -0
  23. package/dist/authenticity/index.js +209 -193
  24. package/dist/authenticity/index.js.map +1 -1
  25. package/dist/baseline-HsBvw_dk.js +550 -0
  26. package/dist/baseline-HsBvw_dk.js.map +1 -0
  27. package/dist/baseline-hG3K85h4.d.ts +125 -0
  28. package/dist/baseline-hG3K85h4.d.ts.map +1 -0
  29. package/dist/belief-state/index.d.ts +448 -1205
  30. package/dist/belief-state/index.d.ts.map +1 -0
  31. package/dist/belief-state/index.js +1617 -1710
  32. package/dist/belief-state/index.js.map +1 -1
  33. package/dist/benchmarks/index.d.ts +2 -894
  34. package/dist/benchmarks/index.js +2 -59
  35. package/dist/benchmarks-DviOvUNr.js +754 -0
  36. package/dist/benchmarks-DviOvUNr.js.map +1 -0
  37. package/dist/builder-eval/index.d.ts +150 -662
  38. package/dist/builder-eval/index.d.ts.map +1 -0
  39. package/dist/builder-eval/index.js +356 -345
  40. package/dist/builder-eval/index.js.map +1 -1
  41. package/dist/calibration-CNWWA6K8.js +94 -0
  42. package/dist/calibration-CNWWA6K8.js.map +1 -0
  43. package/dist/campaign/index.d.ts +5 -6390
  44. package/dist/campaign/index.js +3 -212
  45. package/dist/campaign-CBKZvQ1H.js +3885 -0
  46. package/dist/campaign-CBKZvQ1H.js.map +1 -0
  47. package/dist/cli.d.ts +1 -1
  48. package/dist/cli.js +164 -174
  49. package/dist/cli.js.map +1 -1
  50. package/dist/client-C97NMzqi.d.ts +581 -0
  51. package/dist/client-C97NMzqi.d.ts.map +1 -0
  52. package/dist/client-CYzbdJOZ.js +637 -0
  53. package/dist/client-CYzbdJOZ.js.map +1 -0
  54. package/dist/code-agent-session-BjkMTQ7H.js +1390 -0
  55. package/dist/code-agent-session-BjkMTQ7H.js.map +1 -0
  56. package/dist/code-agent-session-DqqgOJaz.d.ts +143 -0
  57. package/dist/code-agent-session-DqqgOJaz.d.ts.map +1 -0
  58. package/dist/concurrency-DIxRZF_J.js +85 -0
  59. package/dist/concurrency-DIxRZF_J.js.map +1 -0
  60. package/dist/contract/index.d.ts +645 -5605
  61. package/dist/contract/index.d.ts.map +1 -0
  62. package/dist/contract/index.js +1730 -1937
  63. package/dist/contract/index.js.map +1 -1
  64. package/dist/control.d.ts +3 -1030
  65. package/dist/control.js +2 -32
  66. package/dist/cost-ledger-DIgQUFZZ.js +801 -0
  67. package/dist/cost-ledger-DIgQUFZZ.js.map +1 -0
  68. package/dist/cost-ledger-Dye6jCgg.d.ts +236 -0
  69. package/dist/cost-ledger-Dye6jCgg.d.ts.map +1 -0
  70. package/dist/dataset-BvtnC8Dc.d.ts +115 -0
  71. package/dist/dataset-BvtnC8Dc.d.ts.map +1 -0
  72. package/dist/default-registry-C-vFCSEc.js +2579 -0
  73. package/dist/default-registry-C-vFCSEc.js.map +1 -0
  74. package/dist/default-registry-CNPo-Vsb.d.ts +540 -0
  75. package/dist/default-registry-CNPo-Vsb.d.ts.map +1 -0
  76. package/dist/emitter-CPBAhxum.js +266 -0
  77. package/dist/emitter-CPBAhxum.js.map +1 -0
  78. package/dist/emitter-DGQGoLyj.d.ts +113 -0
  79. package/dist/emitter-DGQGoLyj.d.ts.map +1 -0
  80. package/dist/errors-8YnH8WlF.js +64 -0
  81. package/dist/errors-8YnH8WlF.js.map +1 -0
  82. package/dist/errors-CEk209JS.d.ts +76 -0
  83. package/dist/errors-CEk209JS.d.ts.map +1 -0
  84. package/dist/eval-campaign-DEm6c8ru.js +349 -0
  85. package/dist/eval-campaign-DEm6c8ru.js.map +1 -0
  86. package/dist/execution-tracks-CpgFPpS5.js +93 -0
  87. package/dist/execution-tracks-CpgFPpS5.js.map +1 -0
  88. package/dist/exporters-q9iL-2Jf.js +148 -0
  89. package/dist/exporters-q9iL-2Jf.js.map +1 -0
  90. package/dist/extract-usage-BrQ8mCLX.js +155 -0
  91. package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
  92. package/dist/failure-cluster-CqcvCcdR.d.ts +59 -0
  93. package/dist/failure-cluster-CqcvCcdR.d.ts.map +1 -0
  94. package/dist/feedback-trajectory-CVaeREXV.d.ts +340 -0
  95. package/dist/feedback-trajectory-CVaeREXV.d.ts.map +1 -0
  96. package/dist/fuzz.d.ts +320 -646
  97. package/dist/fuzz.d.ts.map +1 -0
  98. package/dist/fuzz.js +670 -617
  99. package/dist/fuzz.js.map +1 -1
  100. package/dist/hf-dataset-DBJXXoY1.js +763 -0
  101. package/dist/hf-dataset-DBJXXoY1.js.map +1 -0
  102. package/dist/hosted/index.d.ts +11 -831
  103. package/dist/hosted/index.d.ts.map +1 -0
  104. package/dist/hosted/index.js +2 -37
  105. package/dist/index-2JJSA6-r2.d.ts +926 -0
  106. package/dist/index-2JJSA6-r2.d.ts.map +1 -0
  107. package/dist/index-6N0aYmpW.d.ts +217 -0
  108. package/dist/index-6N0aYmpW.d.ts.map +1 -0
  109. package/dist/index-BAvgST_9.d.ts +131 -0
  110. package/dist/index-BAvgST_9.d.ts.map +1 -0
  111. package/dist/index-BvnJuTGD.d.ts +68 -0
  112. package/dist/index-BvnJuTGD.d.ts.map +1 -0
  113. package/dist/index-C61Wi7yg.d.ts +547 -0
  114. package/dist/index-C61Wi7yg.d.ts.map +1 -0
  115. package/dist/index-CAPUUKaM.d.ts +335 -0
  116. package/dist/index-CAPUUKaM.d.ts.map +1 -0
  117. package/dist/index-DE5fb3EC.d.ts +2244 -0
  118. package/dist/index-DE5fb3EC.d.ts.map +1 -0
  119. package/dist/index-DSC51roc.d.ts +102 -0
  120. package/dist/index-DSC51roc.d.ts.map +1 -0
  121. package/dist/index.d.ts +3776 -15120
  122. package/dist/index.d.ts.map +1 -0
  123. package/dist/index.js +11185 -11191
  124. package/dist/index.js.map +1 -1
  125. package/dist/integrity-BzRbCHzi.js +424 -0
  126. package/dist/integrity-BzRbCHzi.js.map +1 -0
  127. package/dist/integrity-rmVhXWA7.d.ts +61 -0
  128. package/dist/integrity-rmVhXWA7.d.ts.map +1 -0
  129. package/dist/judge-calibration-DFtEMlde.d.ts +146 -0
  130. package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
  131. package/dist/ledger-core/index.d.ts +2 -0
  132. package/dist/ledger-core/index.js +2 -0
  133. package/dist/ledger-core-DtZz1RG0.js +388 -0
  134. package/dist/ledger-core-DtZz1RG0.js.map +1 -0
  135. package/dist/llm-client--GR4JbZE.js +687 -0
  136. package/dist/llm-client--GR4JbZE.js.map +1 -0
  137. package/dist/llm-client-B_nIBlYo.d.ts +290 -0
  138. package/dist/llm-client-B_nIBlYo.d.ts.map +1 -0
  139. package/dist/matrix/index.d.ts +3 -155
  140. package/dist/matrix/index.js +2 -8
  141. package/dist/matrix-BzQnu2S6.js +270 -0
  142. package/dist/matrix-BzQnu2S6.js.map +1 -0
  143. package/dist/meta-eval/index.d.ts +4 -1027
  144. package/dist/meta-eval/index.js +393 -390
  145. package/dist/meta-eval/index.js.map +1 -1
  146. package/dist/metrics-C9YY1OcL.js +239 -0
  147. package/dist/metrics-C9YY1OcL.js.map +1 -0
  148. package/dist/mint-yN2M2eh0.js +201 -0
  149. package/dist/mint-yN2M2eh0.js.map +1 -0
  150. package/dist/multi-layer-verifier-BHY1gWAc.d.ts +138 -0
  151. package/dist/multi-layer-verifier-BHY1gWAc.d.ts.map +1 -0
  152. package/dist/multishot/index.d.ts +273 -481
  153. package/dist/multishot/index.d.ts.map +1 -0
  154. package/dist/multishot/index.js +595 -548
  155. package/dist/multishot/index.js.map +1 -1
  156. package/dist/off-policy-DvgzvtIx.js +220 -0
  157. package/dist/off-policy-DvgzvtIx.js.map +1 -0
  158. package/dist/off-policy-mskQw8Mb.d.ts +153 -0
  159. package/dist/off-policy-mskQw8Mb.d.ts.map +1 -0
  160. package/dist/openapi.json +1 -1
  161. package/dist/opencode-sqlite-BGrHeDu3.js +318 -0
  162. package/dist/opencode-sqlite-BGrHeDu3.js.map +1 -0
  163. package/dist/outcome-store-BYHIuO0e.d.ts +65 -0
  164. package/dist/outcome-store-BYHIuO0e.d.ts.map +1 -0
  165. package/dist/outcome-store-ChBKlTd_.js +75 -0
  166. package/dist/outcome-store-ChBKlTd_.js.map +1 -0
  167. package/dist/paired-arms-D9D0wXj2.js +260 -0
  168. package/dist/paired-arms-D9D0wXj2.js.map +1 -0
  169. package/dist/pipelines/index.d.ts +95 -532
  170. package/dist/pipelines/index.d.ts.map +1 -0
  171. package/dist/pipelines/index.js +497 -478
  172. package/dist/pipelines/index.js.map +1 -1
  173. package/dist/pre-registration-DakwTRXk.js +96 -0
  174. package/dist/pre-registration-DakwTRXk.js.map +1 -0
  175. package/dist/propose-review-control-LhIWGzHl.js +1458 -0
  176. package/dist/propose-review-control-LhIWGzHl.js.map +1 -0
  177. package/dist/query-CJ_DX8vl.d.ts +28 -0
  178. package/dist/query-CJ_DX8vl.d.ts.map +1 -0
  179. package/dist/query-Di7eEQ79.js +83 -0
  180. package/dist/query-Di7eEQ79.js.map +1 -0
  181. package/dist/raw-provider-sink-BQd7mzyT.js +201 -0
  182. package/dist/raw-provider-sink-BQd7mzyT.js.map +1 -0
  183. package/dist/raw-provider-sink-BU29Sh8h.d.ts +134 -0
  184. package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +1 -0
  185. package/dist/redact-7Aq1ukl-.js +107 -0
  186. package/dist/redact-7Aq1ukl-.js.map +1 -0
  187. package/dist/release-report-Cz9NKH39.js +603 -0
  188. package/dist/release-report-Cz9NKH39.js.map +1 -0
  189. package/dist/release-report-mvB2G4_J.d.ts +244 -0
  190. package/dist/release-report-mvB2G4_J.d.ts.map +1 -0
  191. package/dist/replay-GnyotH0J.js +1741 -0
  192. package/dist/replay-GnyotH0J.js.map +1 -0
  193. package/dist/replay-OoidtG1E.d.ts +749 -0
  194. package/dist/replay-OoidtG1E.d.ts.map +1 -0
  195. package/dist/reporting.d.ts +6 -1298
  196. package/dist/reporting.js +6 -50
  197. package/dist/researcher-CwTdwXG1.d.ts +314 -0
  198. package/dist/researcher-CwTdwXG1.d.ts.map +1 -0
  199. package/dist/reward-hacking-eAnOsynk.d.ts +224 -0
  200. package/dist/reward-hacking-eAnOsynk.d.ts.map +1 -0
  201. package/dist/reward-hacking-qipEpKvY.js +596 -0
  202. package/dist/reward-hacking-qipEpKvY.js.map +1 -0
  203. package/dist/reward-nw2xZGZG.js +137 -0
  204. package/dist/reward-nw2xZGZG.js.map +1 -0
  205. package/dist/rl.d.ts +916 -3596
  206. package/dist/rl.d.ts.map +1 -0
  207. package/dist/rl.js +2362 -1751
  208. package/dist/rl.js.map +1 -1
  209. package/dist/rolldown-runtime-8H4AJuhK.js +14 -0
  210. package/dist/rollout/index.d.ts +3 -1048
  211. package/dist/rollout/index.js +8 -110
  212. package/dist/rollout-BOYjemfR.js +624 -0
  213. package/dist/rollout-BOYjemfR.js.map +1 -0
  214. package/dist/rubric-predictive-validity-B3xmbmS1.js +141 -0
  215. package/dist/rubric-predictive-validity-B3xmbmS1.js.map +1 -0
  216. package/dist/rubric-predictive-validity-Ku_clp_1.d.ts +74 -0
  217. package/dist/rubric-predictive-validity-Ku_clp_1.d.ts.map +1 -0
  218. package/dist/run-evidence-ClRX_8A9.d.ts +225 -0
  219. package/dist/run-evidence-ClRX_8A9.d.ts.map +1 -0
  220. package/dist/run-record-BuoE80Dq.js +467 -0
  221. package/dist/run-record-BuoE80Dq.js.map +1 -0
  222. package/dist/run-record-CnZu_gjl.d.ts +357 -0
  223. package/dist/run-record-CnZu_gjl.d.ts.map +1 -0
  224. package/dist/run-score-iEEAWiBY.js +41 -0
  225. package/dist/run-score-iEEAWiBY.js.map +1 -0
  226. package/dist/runtime-trajectory-1gyaTOoC.js +93 -0
  227. package/dist/runtime-trajectory-1gyaTOoC.js.map +1 -0
  228. package/dist/runtime-trajectory-BvSZcCHD.d.ts +50 -0
  229. package/dist/runtime-trajectory-BvSZcCHD.d.ts.map +1 -0
  230. package/dist/schema-BtVldJ3T.d.ts +206 -0
  231. package/dist/schema-BtVldJ3T.d.ts.map +1 -0
  232. package/dist/schema-C6DW4ZHR.js +821 -0
  233. package/dist/schema-C6DW4ZHR.js.map +1 -0
  234. package/dist/schema-CRhEY1SO.js +69 -0
  235. package/dist/schema-CRhEY1SO.js.map +1 -0
  236. package/dist/schema-Cef2cFmb.d.ts +408 -0
  237. package/dist/schema-Cef2cFmb.d.ts.map +1 -0
  238. package/dist/semantic-concept-judge-B6cWNJ2K.js +725 -0
  239. package/dist/semantic-concept-judge-B6cWNJ2K.js.map +1 -0
  240. package/dist/sequential-Br0mAPHA.js +148 -0
  241. package/dist/sequential-Br0mAPHA.js.map +1 -0
  242. package/dist/sequential-CYwq6Ff_.d.ts +141 -0
  243. package/dist/sequential-CYwq6Ff_.d.ts.map +1 -0
  244. package/dist/series-convergence-CjO2QdRW.js +43 -0
  245. package/dist/series-convergence-CjO2QdRW.js.map +1 -0
  246. package/dist/series-convergence-ofsqPWhs.d.ts +35 -0
  247. package/dist/series-convergence-ofsqPWhs.d.ts.map +1 -0
  248. package/dist/server-m5D9cvnG.js +1040 -0
  249. package/dist/server-m5D9cvnG.js.map +1 -0
  250. package/dist/skill-usage-C_pXm7lP.d.ts +534 -0
  251. package/dist/skill-usage-C_pXm7lP.d.ts.map +1 -0
  252. package/dist/skillopt-optimization-method-B7wX7XkF.d.ts +1739 -0
  253. package/dist/skillopt-optimization-method-B7wX7XkF.d.ts.map +1 -0
  254. package/dist/skillopt-optimization-method-D4ODwFVV.js +7712 -0
  255. package/dist/skillopt-optimization-method-D4ODwFVV.js.map +1 -0
  256. package/dist/statistics-Cmj6nynr.d.ts +514 -0
  257. package/dist/statistics-Cmj6nynr.d.ts.map +1 -0
  258. package/dist/statistics-CnnxdpOg.js +1437 -0
  259. package/dist/statistics-CnnxdpOg.js.map +1 -0
  260. package/dist/store-CT9YIIve.d.ts +117 -0
  261. package/dist/store-CT9YIIve.d.ts.map +1 -0
  262. package/dist/store-CxJry_cs.d.ts +229 -0
  263. package/dist/store-CxJry_cs.d.ts.map +1 -0
  264. package/dist/storyboard/index.d.ts +84 -203
  265. package/dist/storyboard/index.d.ts.map +1 -0
  266. package/dist/storyboard/index.js +609 -542
  267. package/dist/storyboard/index.js.map +1 -1
  268. package/dist/summary-report-BNs5nmXI.js +862 -0
  269. package/dist/summary-report-BNs5nmXI.js.map +1 -0
  270. package/dist/summary-report-Cj9gdw4i.d.ts +407 -0
  271. package/dist/summary-report-Cj9gdw4i.d.ts.map +1 -0
  272. package/dist/supervisor-run/index.d.ts +2 -849
  273. package/dist/supervisor-run/index.js +2 -64
  274. package/dist/supervisor-run-_lnTLM3z.js +1679 -0
  275. package/dist/supervisor-run-_lnTLM3z.js.map +1 -0
  276. package/dist/task-failure-attributes-CQZlB3et.js +311 -0
  277. package/dist/task-failure-attributes-CQZlB3et.js.map +1 -0
  278. package/dist/test-graded-scenario-BsqWLmPt.js +318 -0
  279. package/dist/test-graded-scenario-BsqWLmPt.js.map +1 -0
  280. package/dist/test-graded-scenario-D1TaI2va.d.ts +141 -0
  281. package/dist/test-graded-scenario-D1TaI2va.d.ts.map +1 -0
  282. package/dist/tools-BmuN627J.js +1085 -0
  283. package/dist/tools-BmuN627J.js.map +1 -0
  284. package/dist/trace-attributes.d.ts +2 -52
  285. package/dist/trace-attributes.js +131 -61
  286. package/dist/trace-attributes.js.map +1 -1
  287. package/dist/traces.d.ts +12 -2365
  288. package/dist/traces.js +12 -251
  289. package/dist/trajectory-D_7rLrvE.js +56 -0
  290. package/dist/trajectory-D_7rLrvE.js.map +1 -0
  291. package/dist/types-DGsxbAEd.d.ts +387 -0
  292. package/dist/types-DGsxbAEd.d.ts.map +1 -0
  293. package/dist/types-k9tZGKUg.d.ts +640 -0
  294. package/dist/types-k9tZGKUg.d.ts.map +1 -0
  295. package/dist/verdict-Dps8_okt.d.ts +37 -0
  296. package/dist/verdict-Dps8_okt.d.ts.map +1 -0
  297. package/dist/wire/index.d.ts +702 -1174
  298. package/dist/wire/index.d.ts.map +1 -0
  299. package/dist/wire/index.js +2 -81
  300. package/docs/feature-guide.md +1 -1
  301. package/docs/rollout.md +116 -2
  302. package/package.json +18 -10
  303. package/dist/benchmarks/index.js.map +0 -1
  304. package/dist/campaign/index.js.map +0 -1
  305. package/dist/chunk-2JX3CFMB.js +0 -695
  306. package/dist/chunk-2JX3CFMB.js.map +0 -1
  307. package/dist/chunk-2MKQIFS4.js +0 -183
  308. package/dist/chunk-2MKQIFS4.js.map +0 -1
  309. package/dist/chunk-3RF76KTD.js +0 -84
  310. package/dist/chunk-3RF76KTD.js.map +0 -1
  311. package/dist/chunk-5DTSBUL2.js +0 -159
  312. package/dist/chunk-5DTSBUL2.js.map +0 -1
  313. package/dist/chunk-7ZZMD7UK.js +0 -386
  314. package/dist/chunk-7ZZMD7UK.js.map +0 -1
  315. package/dist/chunk-BOD4O7OF.js +0 -40
  316. package/dist/chunk-BOD4O7OF.js.map +0 -1
  317. package/dist/chunk-BYT7ELPS.js +0 -1553
  318. package/dist/chunk-BYT7ELPS.js.map +0 -1
  319. package/dist/chunk-DJKY2TSY.js +0 -2428
  320. package/dist/chunk-DJKY2TSY.js.map +0 -1
  321. package/dist/chunk-DPUHNQLN.js +0 -232
  322. package/dist/chunk-DPUHNQLN.js.map +0 -1
  323. package/dist/chunk-DRYIUNWY.js +0 -622
  324. package/dist/chunk-DRYIUNWY.js.map +0 -1
  325. package/dist/chunk-EJGRPCO3.js +0 -617
  326. package/dist/chunk-EJGRPCO3.js.map +0 -1
  327. package/dist/chunk-EOSZT7PL.js +0 -2001
  328. package/dist/chunk-EOSZT7PL.js.map +0 -1
  329. package/dist/chunk-EZJEIH2R.js +0 -1559
  330. package/dist/chunk-EZJEIH2R.js.map +0 -1
  331. package/dist/chunk-GGE4NNQT.js +0 -65
  332. package/dist/chunk-GGE4NNQT.js.map +0 -1
  333. package/dist/chunk-HHWE3POT.js +0 -94
  334. package/dist/chunk-HHWE3POT.js.map +0 -1
  335. package/dist/chunk-IHQDPH7D.js +0 -171
  336. package/dist/chunk-IHQDPH7D.js.map +0 -1
  337. package/dist/chunk-JHCHEVET.js +0 -274
  338. package/dist/chunk-JHCHEVET.js.map +0 -1
  339. package/dist/chunk-K4DBDHLK.js +0 -158
  340. package/dist/chunk-K4DBDHLK.js.map +0 -1
  341. package/dist/chunk-K6N6XJJX.js +0 -306
  342. package/dist/chunk-K6N6XJJX.js.map +0 -1
  343. package/dist/chunk-MA6HLL3S.js +0 -65
  344. package/dist/chunk-MA6HLL3S.js.map +0 -1
  345. package/dist/chunk-MAZ26DC7.js +0 -99
  346. package/dist/chunk-MAZ26DC7.js.map +0 -1
  347. package/dist/chunk-MHELPNRP.js +0 -1212
  348. package/dist/chunk-MHELPNRP.js.map +0 -1
  349. package/dist/chunk-NACAGYSY.js +0 -1040
  350. package/dist/chunk-NACAGYSY.js.map +0 -1
  351. package/dist/chunk-NKAGIDE2.js +0 -7633
  352. package/dist/chunk-NKAGIDE2.js.map +0 -1
  353. package/dist/chunk-NPCTHQIO.js +0 -91
  354. package/dist/chunk-NPCTHQIO.js.map +0 -1
  355. package/dist/chunk-NYLOYM6N.js +0 -332
  356. package/dist/chunk-NYLOYM6N.js.map +0 -1
  357. package/dist/chunk-ONWEPEDO.js +0 -57
  358. package/dist/chunk-ONWEPEDO.js.map +0 -1
  359. package/dist/chunk-P5W7RQKK.js +0 -576
  360. package/dist/chunk-P5W7RQKK.js.map +0 -1
  361. package/dist/chunk-P6FYH6K4.js +0 -1161
  362. package/dist/chunk-P6FYH6K4.js.map +0 -1
  363. package/dist/chunk-PBE2LOSS.js +0 -669
  364. package/dist/chunk-PBE2LOSS.js.map +0 -1
  365. package/dist/chunk-PC4UYEBM.js +0 -166
  366. package/dist/chunk-PC4UYEBM.js.map +0 -1
  367. package/dist/chunk-PXE2VKMX.js +0 -140
  368. package/dist/chunk-PXE2VKMX.js.map +0 -1
  369. package/dist/chunk-PZ5AY32C.js +0 -10
  370. package/dist/chunk-PZ5AY32C.js.map +0 -1
  371. package/dist/chunk-RZTMDUO7.js +0 -49
  372. package/dist/chunk-RZTMDUO7.js.map +0 -1
  373. package/dist/chunk-S5YLIBFX.js +0 -136
  374. package/dist/chunk-S5YLIBFX.js.map +0 -1
  375. package/dist/chunk-SZLVEKMJ.js +0 -1446
  376. package/dist/chunk-SZLVEKMJ.js.map +0 -1
  377. package/dist/chunk-T4SQEITX.js +0 -95
  378. package/dist/chunk-T4SQEITX.js.map +0 -1
  379. package/dist/chunk-TBL77AUT.js +0 -355
  380. package/dist/chunk-TBL77AUT.js.map +0 -1
  381. package/dist/chunk-TSN7JT6D.js +0 -1646
  382. package/dist/chunk-TSN7JT6D.js.map +0 -1
  383. package/dist/chunk-TT4KNT67.js +0 -124
  384. package/dist/chunk-TT4KNT67.js.map +0 -1
  385. package/dist/chunk-UB2LOJ6Q.js +0 -4461
  386. package/dist/chunk-UB2LOJ6Q.js.map +0 -1
  387. package/dist/chunk-UWZZKKU7.js +0 -237
  388. package/dist/chunk-UWZZKKU7.js.map +0 -1
  389. package/dist/chunk-VBQ3CRKH.js +0 -291
  390. package/dist/chunk-VBQ3CRKH.js.map +0 -1
  391. package/dist/chunk-VGRCHJON.js +0 -163
  392. package/dist/chunk-VGRCHJON.js.map +0 -1
  393. package/dist/chunk-VI2UW6B6.js +0 -162
  394. package/dist/chunk-VI2UW6B6.js.map +0 -1
  395. package/dist/chunk-VLOATJQ2.js +0 -908
  396. package/dist/chunk-VLOATJQ2.js.map +0 -1
  397. package/dist/chunk-VQMK5FMP.js +0 -247
  398. package/dist/chunk-VQMK5FMP.js.map +0 -1
  399. package/dist/chunk-VZSRQ272.js +0 -149
  400. package/dist/chunk-VZSRQ272.js.map +0 -1
  401. package/dist/chunk-WGXIEX7P.js +0 -116
  402. package/dist/chunk-WGXIEX7P.js.map +0 -1
  403. package/dist/chunk-WS3NZZQQ.js +0 -929
  404. package/dist/chunk-WS3NZZQQ.js.map +0 -1
  405. package/dist/chunk-XDWDC2MP.js +0 -695
  406. package/dist/chunk-XDWDC2MP.js.map +0 -1
  407. package/dist/chunk-XPRT64IE.js +0 -766
  408. package/dist/chunk-XPRT64IE.js.map +0 -1
  409. package/dist/chunk-YJBNWCAA.js +0 -1056
  410. package/dist/chunk-YJBNWCAA.js.map +0 -1
  411. package/dist/chunk-ZET2UAYW.js +0 -89
  412. package/dist/chunk-ZET2UAYW.js.map +0 -1
  413. package/dist/chunk-ZUUWPZCV.js +0 -752
  414. package/dist/chunk-ZUUWPZCV.js.map +0 -1
  415. package/dist/control.js.map +0 -1
  416. package/dist/hosted/index.js.map +0 -1
  417. package/dist/matrix/index.js.map +0 -1
  418. package/dist/reporting.js.map +0 -1
  419. package/dist/rollout/index.js.map +0 -1
  420. package/dist/run-campaign-ISHFZ7FJ.js +0 -17
  421. package/dist/run-campaign-ISHFZ7FJ.js.map +0 -1
  422. package/dist/supervisor-run/index.js.map +0 -1
  423. package/dist/traces.js.map +0 -1
  424. package/dist/wire/index.js.map +0 -1
@@ -0,0 +1,1437 @@
1
+ import { s as ValidationError } from "./errors-8YnH8WlF.js";
2
+ //#region src/judge-calibration.ts
3
+ /**
4
+ * Judge calibration — measure judge quality against human gold + bias.
5
+ *
6
+ * Workflow:
7
+ * 1. Build a golden set: {itemId, humanScore}[].
8
+ * 2. Run candidate judges; each produces {itemId, score}.
9
+ * 3. `calibrateJudge(golden, candidate)` reports κ + Pearson + MAE.
10
+ * 4. `calibrateJudgeContinuous(golden, candidate)` adds quadratic-weighted
11
+ * κ over the un-rounded [0,1] scores plus ICC(2,1), Pearson, Spearman,
12
+ * and bootstrap CIs — use this for fine-grained judges where rounding
13
+ * to int discards information (e.g. 0.78 vs 0.81 both round to 1 and
14
+ * look "perfectly agreed" to integer κ).
15
+ * 5. Run bias probes (positional, verbosity, self-preference) to
16
+ * detect systematic score inflation.
17
+ * 6. For N≥2 judges on the same items, `continuousAgreement(scores)`
18
+ * reports ICC(2,1) + κ_w + Pearson + Spearman with bootstrap CIs.
19
+ *
20
+ * Returns actionable diagnostics, not a single number. Consumers then
21
+ * decide whether to trust the judge, retrain it, or add a tie-breaker.
22
+ */
23
+ /**
24
+ * Measure judge quality against human gold labels: computes Cohen's κ, Pearson correlation, and MAE over matched item ids.
25
+ */
26
+ function calibrateJudge(golden, candidate) {
27
+ const map = /* @__PURE__ */ new Map();
28
+ for (const g of golden) map.set(g.itemId, {
29
+ h: g.humanScore,
30
+ j: NaN
31
+ });
32
+ for (const c of candidate) {
33
+ const entry = map.get(c.itemId);
34
+ if (entry) entry.j = c.score;
35
+ }
36
+ const common = [...map.values()].filter((v) => Number.isFinite(v.j));
37
+ const n = common.length;
38
+ if (n < 2) return {
39
+ n,
40
+ pearson: NaN,
41
+ kappa: NaN,
42
+ mae: NaN,
43
+ worstItems: []
44
+ };
45
+ const humans = common.map((c) => c.h);
46
+ const judges = common.map((c) => c.j);
47
+ return {
48
+ n,
49
+ pearson: pearsonR(humans, judges),
50
+ kappa: weightedKappa(humans.map(Math.round), judges.map(Math.round)),
51
+ mae: common.map((c) => Math.abs(c.j - c.h)).reduce((a, b) => a + b, 0) / n,
52
+ worstItems: [...map.entries()].filter(([, v]) => Number.isFinite(v.j)).map(([itemId, v]) => ({
53
+ itemId,
54
+ judge: v.j,
55
+ human: v.h,
56
+ delta: Math.abs(v.j - v.h)
57
+ })).sort((a, b) => b.delta - a.delta).slice(0, 5)
58
+ };
59
+ }
60
+ /**
61
+ * Feed the same items to the judge twice with A/B swapped and pass all
62
+ * results here. Items that don't appear in both positions are ignored.
63
+ */
64
+ function positionalBias(scores) {
65
+ const pairs = /* @__PURE__ */ new Map();
66
+ for (const s of scores) {
67
+ const slot = pairs.get(s.itemId) ?? {};
68
+ if (s.positionOfAInput === "first") slot.first = s.score;
69
+ else if (s.positionOfAInput === "second") slot.second = s.score;
70
+ pairs.set(s.itemId, slot);
71
+ }
72
+ const deltas = [];
73
+ for (const { first, second } of pairs.values()) if (first !== void 0 && second !== void 0) deltas.push(first - second);
74
+ if (deltas.length === 0) return {
75
+ avgDelta: 0,
76
+ n: 0
77
+ };
78
+ return {
79
+ avgDelta: deltas.reduce((a, b) => a + b, 0) / deltas.length,
80
+ n: deltas.length
81
+ };
82
+ }
83
+ function verbosityBias(samples) {
84
+ const n = samples.length;
85
+ if (n < 3) return {
86
+ pearson: NaN,
87
+ n
88
+ };
89
+ return {
90
+ pearson: pearsonR(samples.map((s) => s.outputLen), samples.map((s) => s.score)),
91
+ n
92
+ };
93
+ }
94
+ /**
95
+ * Pass the same scenarios scored with judge-model X grading outputs from
96
+ * model X (in-family) and model Y (out-of-family). Non-zero delta
97
+ * indicates self-preference.
98
+ */
99
+ function selfPreference(samples) {
100
+ const inF = samples.filter((s) => s.inFamily).map((s) => s.score);
101
+ const outF = samples.filter((s) => !s.inFamily).map((s) => s.score);
102
+ if (inF.length === 0 || outF.length === 0) return {
103
+ inFamilyMean: 0,
104
+ outOfFamilyMean: 0,
105
+ deltaMean: 0,
106
+ n: 0
107
+ };
108
+ const inMean = inF.reduce((a, b) => a + b, 0) / inF.length;
109
+ const outMean = outF.reduce((a, b) => a + b, 0) / outF.length;
110
+ return {
111
+ inFamilyMean: inMean,
112
+ outOfFamilyMean: outMean,
113
+ deltaMean: inMean - outMean,
114
+ n: samples.length
115
+ };
116
+ }
117
+ /** Quadratic weighted Cohen's κ over bounded integer scores. */
118
+ function weightedKappa(a, b) {
119
+ if (a.length !== b.length || a.length === 0) return NaN;
120
+ const min = Math.min(...a, ...b);
121
+ const K = Math.max(...a, ...b) - min + 1;
122
+ if (K < 2) return 1;
123
+ const observed = Array.from({ length: K }, () => new Array(K).fill(0));
124
+ const rowMarg = new Array(K).fill(0);
125
+ const colMarg = new Array(K).fill(0);
126
+ for (let i = 0; i < a.length; i++) {
127
+ const ai = a[i] - min;
128
+ const bi = b[i] - min;
129
+ const row = observed[ai];
130
+ row[bi] = (row[bi] ?? 0) + 1;
131
+ rowMarg[ai]++;
132
+ colMarg[bi]++;
133
+ }
134
+ let num = 0;
135
+ let den = 0;
136
+ for (let i = 0; i < K; i++) for (let j = 0; j < K; j++) {
137
+ const w = (i - j) ** 2 / (K - 1) ** 2;
138
+ const expected = rowMarg[i] * colMarg[j] / a.length;
139
+ num += w * observed[i][j];
140
+ den += w * expected;
141
+ }
142
+ if (den === 0) return 1;
143
+ return 1 - num / den;
144
+ }
145
+ /**
146
+ * Inter-rater agreement on continuous (typically [0,1]) scores.
147
+ *
148
+ * `scores` has shape [n_items][n_raters]. Rows with any non-finite entry
149
+ * are dropped. Returns NaN metrics if fewer than 2 raters or 2 complete
150
+ * items remain.
151
+ */
152
+ function continuousAgreement(scores, opts = {}) {
153
+ const bootstrap = opts.bootstrap ?? 1e3;
154
+ const weights = opts.weights ?? "quadratic";
155
+ const seed = opts.seed ?? 12648430;
156
+ const ciLevel = opts.ciLevel ?? .95;
157
+ const matrix = scores.filter((row) => row.length >= 2 && row.every((v) => Number.isFinite(v)));
158
+ const raters = matrix[0]?.length ?? 0;
159
+ const clean = matrix.filter((row) => row.length === raters);
160
+ const nClean = clean.length;
161
+ if (nClean < 2 || raters < 2) return {
162
+ weightedKappa: NaN,
163
+ icc: NaN,
164
+ pearson: NaN,
165
+ spearman: NaN,
166
+ ci: {
167
+ icc: [NaN, NaN],
168
+ weightedKappa: [NaN, NaN]
169
+ },
170
+ n: nClean,
171
+ raters
172
+ };
173
+ const kappa = continuousWeightedKappa(clean, weights);
174
+ const icc = icc21(clean);
175
+ const pearson = avgPairwise(clean, pearsonR);
176
+ const spearman = avgPairwise(clean, spearmanR);
177
+ const ciIcc = [NaN, NaN];
178
+ const ciKappa = [NaN, NaN];
179
+ if (bootstrap > 0) {
180
+ const rng = mulberry32$1(seed);
181
+ const iccs = [];
182
+ const kappas = [];
183
+ for (let b = 0; b < bootstrap; b++) {
184
+ const sample = new Array(nClean);
185
+ for (let i = 0; i < nClean; i++) sample[i] = clean[Math.floor(rng() * nClean)];
186
+ const iccB = icc21(sample);
187
+ const kB = continuousWeightedKappa(sample, weights);
188
+ if (Number.isFinite(iccB)) iccs.push(iccB);
189
+ if (Number.isFinite(kB)) kappas.push(kB);
190
+ }
191
+ const [lo, hi] = percentileBounds(ciLevel);
192
+ if (iccs.length > 0) {
193
+ iccs.sort((a, b) => a - b);
194
+ ciIcc[0] = quantile(iccs, lo);
195
+ ciIcc[1] = quantile(iccs, hi);
196
+ }
197
+ if (kappas.length > 0) {
198
+ kappas.sort((a, b) => a - b);
199
+ ciKappa[0] = quantile(kappas, lo);
200
+ ciKappa[1] = quantile(kappas, hi);
201
+ }
202
+ }
203
+ return {
204
+ weightedKappa: kappa,
205
+ icc,
206
+ pearson,
207
+ spearman,
208
+ ci: {
209
+ icc: ciIcc,
210
+ weightedKappa: ciKappa
211
+ },
212
+ n: nClean,
213
+ raters
214
+ };
215
+ }
216
+ /**
217
+ * Extends `calibrateJudge` with continuous-value agreement metrics while
218
+ * retaining its base calibration summary.
219
+ */
220
+ function calibrateJudgeContinuous(golden, candidate, opts = {}) {
221
+ const base = calibrateJudge(golden, candidate);
222
+ const map = /* @__PURE__ */ new Map();
223
+ for (const g of golden) map.set(g.itemId, {
224
+ h: g.humanScore,
225
+ j: NaN
226
+ });
227
+ for (const c of candidate) {
228
+ const entry = map.get(c.itemId);
229
+ if (entry) entry.j = c.score;
230
+ }
231
+ const rows = [];
232
+ for (const v of map.values()) if (Number.isFinite(v.j)) rows.push([v.h, v.j]);
233
+ const agreement = continuousAgreement(rows, opts);
234
+ return {
235
+ ...base,
236
+ weightedKappaContinuous: agreement.weightedKappa,
237
+ icc: agreement.icc,
238
+ spearman: agreement.spearman,
239
+ ci: agreement.ci
240
+ };
241
+ }
242
+ /**
243
+ * Quadratic-weighted κ on continuous scores. With weights w(x,y) = (x-y)^2
244
+ * (or |x-y| for linear) the formula collapses to:
245
+ *
246
+ * κ_w = 1 − E_obs[w] / E_exp[w]
247
+ *
248
+ * where E_obs averages w over paired (a_i, b_i) and E_exp averages w over
249
+ * the independent product distribution (sum_{i,j} w(a_i, b_j) / n^2).
250
+ * The normalisation by (max-min)^2 in the integer version cancels in the
251
+ * ratio, so we don't need it here. Generalises to N raters by averaging κ_w
252
+ * over all rater pairs (mean pairwise weighted agreement).
253
+ */
254
+ function continuousWeightedKappa(rows, scheme) {
255
+ if (rows.length === 0) return NaN;
256
+ const raters = rows[0].length;
257
+ if (raters < 2) return NaN;
258
+ const wFn = scheme === "linear" ? (x, y) => Math.abs(x - y) : (x, y) => (x - y) ** 2;
259
+ let sum = 0;
260
+ let pairs = 0;
261
+ for (let r1 = 0; r1 < raters; r1++) for (let r2 = r1 + 1; r2 < raters; r2++) {
262
+ const a = rows.map((row) => row[r1]);
263
+ const b = rows.map((row) => row[r2]);
264
+ const n = a.length;
265
+ let obs = 0;
266
+ for (let i = 0; i < n; i++) obs += wFn(a[i], b[i]);
267
+ obs /= n;
268
+ let exp = 0;
269
+ for (let i = 0; i < n; i++) for (let j = 0; j < n; j++) exp += wFn(a[i], b[j]);
270
+ exp /= n * n;
271
+ if (exp === 0) sum += obs === 0 ? 1 : 0;
272
+ else sum += 1 - obs / exp;
273
+ pairs++;
274
+ }
275
+ return pairs === 0 ? NaN : sum / pairs;
276
+ }
277
+ /**
278
+ * ICC(2,1) — two-way random effects, absolute agreement, single rater.
279
+ *
280
+ * ICC(2,1) = (MSR − MSE) / (MSR + (k−1)·MSE + k·(MSC − MSE)/n)
281
+ *
282
+ * where MSR = between-rows MS, MSC = between-columns MS, MSE = residual MS,
283
+ * n = rows (items), k = columns (raters).
284
+ */
285
+ function icc21(rows) {
286
+ const n = rows.length;
287
+ if (n < 2) return NaN;
288
+ const k = rows[0].length;
289
+ if (k < 2) return NaN;
290
+ const rowMeans = rows.map((row) => row.reduce((s, v) => s + v, 0) / k);
291
+ const colMeans = new Array(k).fill(0);
292
+ for (let j = 0; j < k; j++) {
293
+ let s = 0;
294
+ for (let i = 0; i < n; i++) s += rows[i][j];
295
+ colMeans[j] = s / n;
296
+ }
297
+ let grand = 0;
298
+ for (let i = 0; i < n; i++) grand += rowMeans[i];
299
+ grand /= n;
300
+ let ssR = 0;
301
+ for (let i = 0; i < n; i++) ssR += (rowMeans[i] - grand) ** 2;
302
+ ssR *= k;
303
+ let ssC = 0;
304
+ for (let j = 0; j < k; j++) ssC += (colMeans[j] - grand) ** 2;
305
+ ssC *= n;
306
+ let ssT = 0;
307
+ for (let i = 0; i < n; i++) for (let j = 0; j < k; j++) ssT += (rows[i][j] - grand) ** 2;
308
+ const ssE = ssT - ssR - ssC;
309
+ const dfR = n - 1;
310
+ const dfC = k - 1;
311
+ const dfE = (n - 1) * (k - 1);
312
+ const msR = ssR / dfR;
313
+ const msC = ssC / dfC;
314
+ const msE = dfE > 0 ? ssE / dfE : 0;
315
+ const denom = msR + (k - 1) * msE + k * (msC - msE) / n;
316
+ if (denom === 0) return msR === 0 && msE === 0 ? 1 : 0;
317
+ return (msR - msE) / denom;
318
+ }
319
+ /** Average pairwise statistic over all rater pairs. */
320
+ function avgPairwise(rows, fn) {
321
+ const k = rows[0]?.length ?? 0;
322
+ if (k < 2) return NaN;
323
+ let sum = 0;
324
+ let pairs = 0;
325
+ for (let i = 0; i < k; i++) for (let j = i + 1; j < k; j++) {
326
+ const r = fn(rows.map((row) => row[i]), rows.map((row) => row[j]));
327
+ if (Number.isFinite(r)) {
328
+ sum += r;
329
+ pairs++;
330
+ }
331
+ }
332
+ return pairs === 0 ? NaN : sum / pairs;
333
+ }
334
+ /** Seeded PRNG — Mulberry32. Deterministic across platforms. */
335
+ function mulberry32$1(seed) {
336
+ let a = seed >>> 0;
337
+ return () => {
338
+ a = a + 1831565813 >>> 0;
339
+ let t = a;
340
+ t = Math.imul(t ^ t >>> 15, t | 1);
341
+ t ^= t + Math.imul(t ^ t >>> 7, t | 61);
342
+ return ((t ^ t >>> 14) >>> 0) / 4294967296;
343
+ };
344
+ }
345
+ function percentileBounds(ciLevel) {
346
+ const tail = (1 - ciLevel) / 2;
347
+ return [tail, 1 - tail];
348
+ }
349
+ /** Linear-interpolated quantile of a pre-sorted ascending array. */
350
+ function quantile(sorted, q) {
351
+ if (sorted.length === 0) return NaN;
352
+ if (sorted.length === 1) return sorted[0];
353
+ const pos = q * (sorted.length - 1);
354
+ const lo = Math.floor(pos);
355
+ const hi = Math.ceil(pos);
356
+ if (lo === hi) return sorted[lo];
357
+ const frac = pos - lo;
358
+ return sorted[lo] * (1 - frac) + sorted[hi] * frac;
359
+ }
360
+ //#endregion
361
+ //#region src/statistics.ts
362
+ /** Identity: dimensions already follow "higher = better" by prompt convention
363
+ * (inverted dims like hallucination are scored 10 = best at the source). */
364
+ const normalizeScores = (scores) => scores;
365
+ /** Weighted mean — falls back to uniform weights when omitted */
366
+ function weightedMean(scores) {
367
+ if (scores.length === 0) return 0;
368
+ let totalWeight = 0;
369
+ let weightedSum = 0;
370
+ for (const { score, weight } of scores) {
371
+ const w = weight ?? 1;
372
+ weightedSum += score * w;
373
+ totalWeight += w;
374
+ }
375
+ return totalWeight > 0 ? weightedSum / totalWeight : 0;
376
+ }
377
+ /** Bootstrap confidence interval */
378
+ function confidenceInterval(scores, confidence = .95, opts = {}) {
379
+ if (scores.length === 0) return {
380
+ mean: 0,
381
+ lower: 0,
382
+ upper: 0
383
+ };
384
+ if (scores.length === 1) return {
385
+ mean: scores[0],
386
+ lower: scores[0],
387
+ upper: scores[0]
388
+ };
389
+ const n = scores.length;
390
+ const mean = scores.reduce((a, b) => a + b, 0) / n;
391
+ const B = opts.resamples ?? 1e3;
392
+ const rng = makeRng(opts.seed);
393
+ const bootstrapMeans = [];
394
+ for (let i = 0; i < B; i++) {
395
+ let sum = 0;
396
+ for (let j = 0; j < n; j++) sum += scores[Math.floor(rng() * n)];
397
+ bootstrapMeans.push(sum / n);
398
+ }
399
+ bootstrapMeans.sort((a, b) => a - b);
400
+ const alpha = 1 - confidence;
401
+ const lowerIdx = Math.floor(alpha / 2 * B);
402
+ const upperIdx = Math.floor((1 - alpha / 2) * B) - 1;
403
+ return {
404
+ mean,
405
+ lower: bootstrapMeans[lowerIdx],
406
+ upper: bootstrapMeans[Math.min(upperIdx, B - 1)]
407
+ };
408
+ }
409
+ /**
410
+ * Inter-rater reliability — simplified Krippendorff's alpha.
411
+ *
412
+ * Each inner array is one judge's scores for all items.
413
+ * All arrays must have the same length (same items scored).
414
+ */
415
+ function interRaterReliability(judgeScores) {
416
+ if (judgeScores.length < 2) return 1;
417
+ const dimensionMap = /* @__PURE__ */ new Map();
418
+ for (const judgeSet of judgeScores) for (const s of judgeSet) {
419
+ if (!dimensionMap.has(s.dimension)) dimensionMap.set(s.dimension, []);
420
+ const arr = dimensionMap.get(s.dimension);
421
+ if (arr.length === 0 || arr[arr.length - 1].length >= judgeScores.length) arr.push([s.score]);
422
+ else arr[arr.length - 1].push(s.score);
423
+ }
424
+ const allValues = [];
425
+ const pairDiffs = [];
426
+ for (const items of dimensionMap.values()) for (const ratings of items) {
427
+ if (ratings.length < 2) continue;
428
+ for (const v of ratings) allValues.push(v);
429
+ for (let i = 0; i < ratings.length; i++) for (let j = i + 1; j < ratings.length; j++) pairDiffs.push((ratings[i] - ratings[j]) ** 2);
430
+ }
431
+ if (pairDiffs.length === 0 || allValues.length < 2) return 1;
432
+ const observedDisagreement = pairDiffs.reduce((a, b) => a + b, 0) / pairDiffs.length;
433
+ let expectedDisagreement = 0;
434
+ let expectedCount = 0;
435
+ for (let i = 0; i < allValues.length; i++) for (let j = i + 1; j < allValues.length; j++) {
436
+ expectedDisagreement += (allValues[i] - allValues[j]) ** 2;
437
+ expectedCount++;
438
+ }
439
+ expectedDisagreement = expectedCount > 0 ? expectedDisagreement / expectedCount : 0;
440
+ if (expectedDisagreement === 0) return 1;
441
+ return 1 - observedDisagreement / expectedDisagreement;
442
+ }
443
+ /**
444
+ * Mann-Whitney U test for comparing two independent groups.
445
+ * Returns U statistic and approximate p-value (normal approximation).
446
+ */
447
+ function mannWhitneyU(a, b) {
448
+ if (a.length === 0 || b.length === 0) return {
449
+ u: 0,
450
+ p: 1
451
+ };
452
+ const n1 = a.length;
453
+ const n2 = b.length;
454
+ const combined = [...a.map((v) => ({
455
+ v,
456
+ group: "a"
457
+ })), ...b.map((v) => ({
458
+ v,
459
+ group: "b"
460
+ }))].sort((x, y) => x.v - y.v);
461
+ const ranks = new Array(combined.length);
462
+ let i = 0;
463
+ while (i < combined.length) {
464
+ let j = i;
465
+ while (j < combined.length && combined[j].v === combined[i].v) j++;
466
+ const avgRank = (i + 1 + j) / 2;
467
+ for (let k = i; k < j; k++) ranks[k] = avgRank;
468
+ i = j;
469
+ }
470
+ let r1 = 0;
471
+ for (let k = 0; k < combined.length; k++) if (combined[k].group === "a") r1 += ranks[k];
472
+ const u1 = r1 - n1 * (n1 + 1) / 2;
473
+ const u2 = n1 * n2 - u1;
474
+ const u = Math.min(u1, u2);
475
+ const mu = n1 * n2 / 2;
476
+ const sigma = Math.sqrt(n1 * n2 * (n1 + n2 + 1) / 12);
477
+ if (sigma === 0) return {
478
+ u,
479
+ p: 1
480
+ };
481
+ return {
482
+ u,
483
+ p: 2 * (1 - normalCdf(Math.abs(u - mu) / sigma))
484
+ };
485
+ }
486
+ /** Partial credit: returns 0-1 ratio of current toward target */
487
+ function partialCredit(current, target) {
488
+ if (target <= 0) return 1;
489
+ return Math.min(1, Math.max(0, current / target));
490
+ }
491
+ /**
492
+ * Paired t-test — before/after measurements on the SAME items.
493
+ * Pairing removes inter-item variance, giving tighter significance than
494
+ * an unpaired test when comparing prompt v1 vs prompt v2 on identical
495
+ * scenarios.
496
+ */
497
+ function pairedTTest(before, after) {
498
+ if (before.length !== after.length) throw new ValidationError(`pairedTTest: unequal sample sizes (${before.length} vs ${after.length})`);
499
+ const n = before.length;
500
+ if (n < 2) return {
501
+ t: 0,
502
+ df: 0,
503
+ p: 1
504
+ };
505
+ const diffs = before.map((b, i) => after[i] - b);
506
+ const mean = diffs.reduce((a, b) => a + b, 0) / n;
507
+ const variance = diffs.reduce((acc, d) => acc + (d - mean) ** 2, 0) / (n - 1);
508
+ const se = Math.sqrt(variance / n);
509
+ if (se === 0) return {
510
+ t: mean === 0 ? 0 : Infinity,
511
+ df: n - 1,
512
+ p: mean === 0 ? 1 : 0
513
+ };
514
+ const t = mean / se;
515
+ const df = n - 1;
516
+ return {
517
+ t,
518
+ df,
519
+ p: 2 * (1 - studentTCdf(Math.abs(t), df))
520
+ };
521
+ }
522
+ /**
523
+ * Wilcoxon signed-rank test — paired non-parametric alternative.
524
+ * Use when the differences aren't normally distributed.
525
+ */
526
+ function wilcoxonSignedRank(before, after) {
527
+ if (before.length !== after.length) throw new ValidationError(`wilcoxonSignedRank: unequal sample sizes (${before.length} vs ${after.length})`);
528
+ const diffs = before.map((b, i) => after[i] - b).filter((d) => d !== 0);
529
+ const n = diffs.length;
530
+ if (n < 6) return {
531
+ w: 0,
532
+ p: 1
533
+ };
534
+ const absRanks = diffs.map((d, i) => ({
535
+ abs: Math.abs(d),
536
+ sign: Math.sign(d),
537
+ i
538
+ })).sort((a, b) => a.abs - b.abs);
539
+ const ranks = new Array(n);
540
+ let i = 0;
541
+ while (i < n) {
542
+ let j = i;
543
+ while (j < n && absRanks[j].abs === absRanks[i].abs) j++;
544
+ const avg = (i + 1 + j) / 2;
545
+ for (let k = i; k < j; k++) ranks[absRanks[k].i] = avg;
546
+ i = j;
547
+ }
548
+ let wPlus = 0;
549
+ for (let k = 0; k < n; k++) if (diffs[k] > 0) wPlus += ranks[k];
550
+ const mean = n * (n + 1) / 4;
551
+ const variance = n * (n + 1) * (2 * n + 1) / 24;
552
+ const z = (wPlus - mean) / Math.sqrt(variance);
553
+ const p = 2 * (1 - normalCdf(Math.abs(z)));
554
+ return {
555
+ w: wPlus,
556
+ p
557
+ };
558
+ }
559
+ /**
560
+ * Cohen's d — standardized effect size for two independent groups.
561
+ * Positive d means group b has higher mean than group a.
562
+ * Rule of thumb: |d| < 0.2 negligible, 0.2–0.5 small, 0.5–0.8 medium, > 0.8 large.
563
+ */
564
+ function cohensD(a, b) {
565
+ if (a.length < 2 || b.length < 2) return 0;
566
+ const meanA = a.reduce((x, y) => x + y, 0) / a.length;
567
+ const meanB = b.reduce((x, y) => x + y, 0) / b.length;
568
+ const varA = a.reduce((acc, x) => acc + (x - meanA) ** 2, 0) / (a.length - 1);
569
+ const varB = b.reduce((acc, x) => acc + (x - meanB) ** 2, 0) / (b.length - 1);
570
+ const pooled = Math.sqrt(((a.length - 1) * varA + (b.length - 1) * varB) / (a.length + b.length - 2));
571
+ if (pooled === 0) return 0;
572
+ return (meanB - meanA) / pooled;
573
+ }
574
+ /**
575
+ * Cohen's dz for paired observations: mean(after - before) divided by the
576
+ * sample standard deviation of those within-pair deltas.
577
+ *
578
+ * Returns null when fewer than two pairs exist or a non-zero constant delta
579
+ * has zero observed variance. In that case the standardized effect is
580
+ * undefined, not an arbitrarily large finite number.
581
+ */
582
+ function pairedCohensDz(before, after) {
583
+ if (before.length !== after.length) throw new ValidationError(`pairedCohensDz: unequal sample sizes (${before.length} vs ${after.length})`);
584
+ if (before.length < 2) return null;
585
+ const deltas = before.map((value, index) => after[index] - value);
586
+ if (deltas.some((value) => !Number.isFinite(value))) throw new ValidationError("pairedCohensDz: all paired values must be finite");
587
+ const meanDelta = deltas.reduce((sum, value) => sum + value, 0) / deltas.length;
588
+ const variance = deltas.reduce((sum, value) => sum + (value - meanDelta) ** 2, 0) / (deltas.length - 1);
589
+ const standardDeviation = Math.sqrt(variance);
590
+ const scale = Math.max(1, Math.abs(meanDelta), ...deltas.map(Math.abs));
591
+ if (standardDeviation <= Number.EPSILON * scale) return meanDelta === 0 ? 0 : null;
592
+ return meanDelta / standardDeviation;
593
+ }
594
+ /**
595
+ * Cliff's delta — a non-parametric effect size for two independent samples.
596
+ * `δ = (#(after > before) − #(after < before)) / (n_before · n_after)`,
597
+ * ranging [-1, 1]. Positive ⇒ `after` tends to exceed `before` (improvement).
598
+ *
599
+ * Distribution-free counterpart to Cohen's d: no normality assumption, robust
600
+ * to the bounded/skewed score distributions judges produce. Pairs with
601
+ * `pairedBootstrap` / `wilcoxonSignedRank` for the non-parametric reporting
602
+ * path. Returns 0 when either sample is empty.
603
+ */
604
+ function cliffsDelta(before, after) {
605
+ const n = before.length * after.length;
606
+ if (n === 0) return 0;
607
+ let dominance = 0;
608
+ for (const a of after) for (const b of before) if (a > b) dominance += 1;
609
+ else if (a < b) dominance -= 1;
610
+ return dominance / n;
611
+ }
612
+ /**
613
+ * Map a Cliff's delta to a qualitative magnitude using the standard
614
+ * Romano et al. thresholds (|δ|): <0.147 negligible, <0.33 small,
615
+ * <0.474 medium, else large.
616
+ */
617
+ function interpretCliffs(delta) {
618
+ const d = Math.abs(delta);
619
+ if (d < .147) return "negligible";
620
+ if (d < .33) return "small";
621
+ if (d < .474) return "medium";
622
+ return "large";
623
+ }
624
+ /**
625
+ * Average-rank-with-ties transform (1-indexed). Tied values receive the mean
626
+ * of the ranks they span, the standard correction for Spearman's ρ.
627
+ */
628
+ function ranks(xs) {
629
+ const indexed = xs.map((v, i) => ({
630
+ v,
631
+ i
632
+ })).sort((a, b) => a.v - b.v);
633
+ const r = new Array(xs.length);
634
+ let i = 0;
635
+ while (i < indexed.length) {
636
+ let j = i;
637
+ while (j + 1 < indexed.length && indexed[j + 1].v === indexed[i].v) j++;
638
+ const avg = (i + j) / 2 + 1;
639
+ for (let k = i; k <= j; k++) r[indexed[k].i] = avg;
640
+ i = j + 1;
641
+ }
642
+ return r;
643
+ }
644
+ /**
645
+ * Pearson product-moment correlation coefficient r ∈ [-1, 1] between two
646
+ * equal-length series. See the edge-case contract above: NaN for n < 2 or
647
+ * unequal lengths, 1 when both series are constant, 0 when exactly one is.
648
+ */
649
+ function pearsonR(a, b) {
650
+ if (a.length !== b.length || a.length < 2) return NaN;
651
+ const n = a.length;
652
+ const meanA = a.reduce((s, v) => s + v, 0) / n;
653
+ const meanB = b.reduce((s, v) => s + v, 0) / n;
654
+ let num = 0;
655
+ let varA = 0;
656
+ let varB = 0;
657
+ for (let i = 0; i < n; i++) {
658
+ const da = a[i] - meanA;
659
+ const db = b[i] - meanB;
660
+ num += da * db;
661
+ varA += da * da;
662
+ varB += db * db;
663
+ }
664
+ if (varA === 0 || varB === 0) return varA === 0 && varB === 0 ? 1 : 0;
665
+ return num / Math.sqrt(varA * varB);
666
+ }
667
+ /**
668
+ * Spearman's rank correlation ρ — Pearson over the average-rank-with-ties
669
+ * transform of each series. Same edge-case contract as {@link pearsonR}.
670
+ */
671
+ function spearmanR(a, b) {
672
+ if (a.length !== b.length || a.length < 2) return NaN;
673
+ return pearsonR(ranks(a), ranks(b));
674
+ }
675
+ /**
676
+ * Weighted composite over judge dimensions: `Σ(score_d · w_d) / Σ(w_d)` across
677
+ * the weighted dimensions. The canonical replacement for the per-consumer
678
+ * hand-rolled composite math (tax/legal/creative/gtm each ship a copy).
679
+ *
680
+ * Fail-loud: throws if a weighted dimension is missing from `dims`, if any
681
+ * weight is negative, or if the weights sum to 0 — none of which can produce
682
+ * a meaningful composite.
683
+ */
684
+ function weightedComposite(input) {
685
+ const entries = Object.entries(input.weights);
686
+ if (entries.length === 0) throw new Error("weightedComposite: `weights` is empty — nothing to combine");
687
+ let weightedSum = 0;
688
+ let weightTotal = 0;
689
+ for (const [dim, weight] of entries) {
690
+ if (weight < 0) throw new Error(`weightedComposite: weight for '${dim}' is negative (${weight})`);
691
+ if (!(dim in input.dims)) throw new Error(`weightedComposite: weighted dimension '${dim}' is absent from \`dims\` — refusing to renormalise onto a different denominator`);
692
+ weightedSum += input.dims[dim] * weight;
693
+ weightTotal += weight;
694
+ }
695
+ if (weightTotal === 0) throw new Error("weightedComposite: weights sum to 0 — composite is undefined");
696
+ const composite = weightedSum / weightTotal;
697
+ return input.threshold === void 0 ? { composite } : {
698
+ composite,
699
+ pass: composite >= input.threshold
700
+ };
701
+ }
702
+ /**
703
+ * Corpus-wide inter-rater agreement across N items × M judges × D dimensions.
704
+ *
705
+ * For each dimension, builds the [n_items][n_judges] matrix of scores
706
+ * (keeping only items every judge rated on that dimension), then runs
707
+ * `continuousAgreement` to get ICC(2,1), κ_w, Pearson, Spearman, and
708
+ * bootstrap CIs. Reports a pooled mean across dimensions as a single
709
+ * "is this judge panel reliable on this corpus?" number.
710
+ *
711
+ * Fail-loud contract:
712
+ * - Empty input throws.
713
+ * - Fewer than 2 judges or fewer than 2 items per dimension throws.
714
+ * - A judge present in some dimensions but with zero scored items on
715
+ * another dimension throws (would silently shrink the matrix).
716
+ * - Duplicate (itemId, judgeName, dimension) records throw.
717
+ */
718
+ function corpusInterRaterAgreement(records, opts = {}) {
719
+ if (records.length === 0) throw new ValidationError("corpusInterRaterAgreement: no score records supplied");
720
+ const judgesSeen = /* @__PURE__ */ new Set();
721
+ const dimsSeen = /* @__PURE__ */ new Set();
722
+ const grid = /* @__PURE__ */ new Map();
723
+ for (const r of records) {
724
+ if (!Number.isFinite(r.score)) throw new ValidationError(`corpusInterRaterAgreement: non-finite score for (item=${r.itemId}, judge=${r.judgeName}, dim=${r.dimension})`);
725
+ judgesSeen.add(r.judgeName);
726
+ dimsSeen.add(r.dimension);
727
+ const byJudge = grid.get(r.dimension) ?? /* @__PURE__ */ new Map();
728
+ const byItem = byJudge.get(r.judgeName) ?? /* @__PURE__ */ new Map();
729
+ if (byItem.has(r.itemId)) throw new ValidationError(`corpusInterRaterAgreement: duplicate record for (item=${r.itemId}, judge=${r.judgeName}, dim=${r.dimension})`);
730
+ byItem.set(r.itemId, r.score);
731
+ byJudge.set(r.judgeName, byItem);
732
+ grid.set(r.dimension, byJudge);
733
+ }
734
+ const targetDims = opts.dimensions ?? [...dimsSeen].sort();
735
+ for (const d of targetDims) if (!dimsSeen.has(d)) throw new ValidationError(`corpusInterRaterAgreement: dimension '${d}' was requested but no records carry it`);
736
+ const targetJudges = opts.judges ? [...opts.judges] : [...judgesSeen].sort();
737
+ for (const j of targetJudges) if (!judgesSeen.has(j)) throw new ValidationError(`corpusInterRaterAgreement: judge '${j}' was requested but produced no records`);
738
+ if (targetJudges.length < 2) throw new ValidationError(`corpusInterRaterAgreement: need ≥2 judges, got ${targetJudges.length}`);
739
+ const perDimension = [];
740
+ const iccs = [];
741
+ const kappas = [];
742
+ for (const dim of targetDims) {
743
+ const byJudge = grid.get(dim);
744
+ const judgeItemCounts = {};
745
+ for (const j of targetJudges) judgeItemCounts[j] = byJudge.get(j)?.size ?? 0;
746
+ const emptyJudges = targetJudges.filter((j) => judgeItemCounts[j] === 0);
747
+ if (emptyJudges.length > 0) throw new ValidationError(`corpusInterRaterAgreement: dimension '${dim}' has no scores from judge(s) ${emptyJudges.join(", ")} (counts: ${JSON.stringify(judgeItemCounts)})`);
748
+ let commonItems = null;
749
+ for (const j of targetJudges) {
750
+ const ids = new Set(byJudge.get(j).keys());
751
+ if (commonItems === null) commonItems = ids;
752
+ else commonItems = new Set([...commonItems].filter((x) => ids.has(x)));
753
+ }
754
+ const sortedItems = [...commonItems ?? /* @__PURE__ */ new Set()].sort();
755
+ if (sortedItems.length < 2) throw new ValidationError(`corpusInterRaterAgreement: dimension '${dim}' has ${sortedItems.length} item(s) rated by all ${targetJudges.length} judges (need ≥2)`);
756
+ const agreement = continuousAgreement(sortedItems.map((itemId) => targetJudges.map((j) => byJudge.get(j).get(itemId))), opts);
757
+ perDimension.push({
758
+ ...agreement,
759
+ dimension: dim,
760
+ itemIds: sortedItems,
761
+ judgeIds: [...targetJudges]
762
+ });
763
+ if (Number.isFinite(agreement.icc)) iccs.push(agreement.icc);
764
+ if (Number.isFinite(agreement.weightedKappa)) kappas.push(agreement.weightedKappa);
765
+ }
766
+ const mean = (xs) => xs.length === 0 ? NaN : xs.reduce((a, b) => a + b, 0) / xs.length;
767
+ return {
768
+ perDimension,
769
+ overallIcc: mean(iccs),
770
+ overallWeightedKappa: mean(kappas),
771
+ dimensions: targetDims,
772
+ judgeIds: targetJudges
773
+ };
774
+ }
775
+ /**
776
+ * Convenience adapter for `JudgeScore[]` data keyed externally by item.
777
+ *
778
+ * Use when you have per-item arrays of `JudgeScore[]` (e.g. one
779
+ * `ScenarioResult.judgeScores` per scenario) and want corpus-wide
780
+ * agreement without manually flattening. `itemId` must be unique per
781
+ * row of `itemsScores`.
782
+ */
783
+ function corpusInterRaterAgreementFromJudgeScores(itemsScores, opts = {}) {
784
+ const records = [];
785
+ const seen = /* @__PURE__ */ new Set();
786
+ for (const { itemId, scores } of itemsScores) {
787
+ if (seen.has(itemId)) throw new ValidationError(`corpusInterRaterAgreementFromJudgeScores: duplicate itemId '${itemId}'`);
788
+ seen.add(itemId);
789
+ for (const s of scores) records.push({
790
+ itemId,
791
+ judgeName: s.judgeName,
792
+ dimension: s.dimension,
793
+ score: s.score
794
+ });
795
+ }
796
+ return corpusInterRaterAgreement(records, opts);
797
+ }
798
+ /** Student-t CDF approximation via Abramowitz-Stegun series. */
799
+ function studentTCdf(t, df) {
800
+ if (df <= 0) return .5;
801
+ if (df > 100) return normalCdf(t);
802
+ const ib = incompleteBeta(df / (df + t * t), df / 2, .5);
803
+ return t >= 0 ? 1 - .5 * ib : .5 * ib;
804
+ }
805
+ /** Regularized incomplete beta function via continued fraction (Lentz). */
806
+ function incompleteBeta(x, a, b) {
807
+ if (x <= 0) return 0;
808
+ if (x >= 1) return 1;
809
+ const lnBeta = lnGamma(a) + lnGamma(b) - lnGamma(a + b);
810
+ const front = Math.exp(Math.log(x) * a + Math.log(1 - x) * b - lnBeta) / a;
811
+ const maxIter = 200;
812
+ const eps = 3e-7;
813
+ let c = 1;
814
+ let d = 1 - (a + b) * x / (a + 1);
815
+ if (Math.abs(d) < 1e-30) d = 1e-30;
816
+ d = 1 / d;
817
+ let f = d;
818
+ for (let m = 1; m <= maxIter; m++) {
819
+ const m2 = 2 * m;
820
+ let num = m * (b - m) * x / ((a + m2 - 1) * (a + m2));
821
+ d = 1 + num * d;
822
+ if (Math.abs(d) < 1e-30) d = 1e-30;
823
+ c = 1 + num / c;
824
+ if (Math.abs(c) < 1e-30) c = 1e-30;
825
+ d = 1 / d;
826
+ f *= d * c;
827
+ num = -((a + m) * (a + b + m) * x) / ((a + m2) * (a + m2 + 1));
828
+ d = 1 + num * d;
829
+ if (Math.abs(d) < 1e-30) d = 1e-30;
830
+ c = 1 + num / c;
831
+ if (Math.abs(c) < 1e-30) c = 1e-30;
832
+ d = 1 / d;
833
+ const delta = d * c;
834
+ f *= delta;
835
+ if (Math.abs(delta - 1) < eps) break;
836
+ }
837
+ return front * f;
838
+ }
839
+ /** Lanczos approximation to ln Γ(z). */
840
+ function lnGamma(z) {
841
+ const g = 7;
842
+ const coefs = [
843
+ .9999999999998099,
844
+ 676.5203681218851,
845
+ -1259.1392167224028,
846
+ 771.3234287776531,
847
+ -176.6150291621406,
848
+ 12.507343278686905,
849
+ -.13857109526572012,
850
+ 9984369578019572e-21,
851
+ 1.5056327351493116e-7
852
+ ];
853
+ if (z < .5) return Math.log(Math.PI / Math.sin(Math.PI * z)) - lnGamma(1 - z);
854
+ z -= 1;
855
+ let x = coefs[0];
856
+ for (let i = 1; i < 9; i++) x += coefs[i] / (z + i);
857
+ const t = z + g + .5;
858
+ return .5 * Math.log(2 * Math.PI) + (z + .5) * Math.log(t) - t + Math.log(x);
859
+ }
860
+ function normalCdf(x) {
861
+ const a1 = .254829592;
862
+ const a2 = -.284496736;
863
+ const a3 = 1.421413741;
864
+ const a4 = -1.453152027;
865
+ const a5 = 1.061405429;
866
+ const p = .3275911;
867
+ const sign = x < 0 ? -1 : 1;
868
+ const absX = Math.abs(x);
869
+ const t = 1 / (1 + p * absX);
870
+ return .5 * (1 + sign * (1 - ((((a5 * t + a4) * t + a3) * t + a2) * t + a1) * t * Math.exp(-absX * absX / 2)));
871
+ }
872
+ /**
873
+ * Required N per arm for a two-sample comparison at target effect size,
874
+ * alpha, and power. Normal-approximation formula:
875
+ * n = 2 * ( (z_{1-α/2} + z_{1-β}) / d )^2
876
+ * where d is Cohen's d. Returns Infinity for effect ≤ 0.
877
+ */
878
+ function requiredSampleSize(opts) {
879
+ const effect = opts.effect;
880
+ if (!Number.isFinite(effect) || effect <= 0) return Infinity;
881
+ const alpha = opts.alpha ?? .05;
882
+ const power = opts.power ?? .8;
883
+ const n = 2 * ((zQuantile(opts.twoSided ?? true ? 1 - alpha / 2 : 1 - alpha) + zQuantile(power)) / effect) ** 2;
884
+ return Math.ceil(n);
885
+ }
886
+ /**
887
+ * Required number of paired observations for a target Cohen's dz.
888
+ * Unlike the independent-groups formula, this has no two-arm factor of two.
889
+ */
890
+ function requiredPairedSampleSize(opts) {
891
+ const effect = opts.effect;
892
+ if (!Number.isFinite(effect) || effect <= 0) return Infinity;
893
+ const alpha = opts.alpha ?? .05;
894
+ const power = opts.power ?? .8;
895
+ const zAlpha = zQuantile(opts.twoSided ?? true ? 1 - alpha / 2 : 1 - alpha);
896
+ const zBeta = zQuantile(power);
897
+ return Math.ceil(((zAlpha + zBeta) / effect) ** 2);
898
+ }
899
+ /**
900
+ * Minimum detectable paired effect (standardised units) for a target paired
901
+ * sample size: d_min = (z_{1-α/2} + z_β) / sqrt(n_paired). Multiply by
902
+ * sd(deltas) for score units; treat as a lower bound — Wilcoxon and bootstrap
903
+ * have asymptotic relative efficiency below 1 vs the t-test on heavy tails.
904
+ */
905
+ function pairedMde(opts) {
906
+ if (!Number.isFinite(opts.nPaired) || opts.nPaired <= 0) return Infinity;
907
+ const alpha = opts.alpha ?? .05;
908
+ const power = opts.power ?? .8;
909
+ return (zQuantile(opts.twoSided ?? true ? 1 - alpha / 2 : 1 - alpha) + zQuantile(power)) / Math.sqrt(opts.nPaired);
910
+ }
911
+ /**
912
+ * Number of paired observations needed for a McNemar test to reach a target
913
+ * power — the pre-registration companion to {@link mcnemar}. Parametrised by the
914
+ * expected discordant-cell probabilities `p10` (P[treatment wins on a pair]) and
915
+ * `p01` (P[control wins]); concordant pairs carry no information, so the count
916
+ * is driven entirely by the discordant rate. Lachin's (1992) asymptotic normal
917
+ * approximation: with discordant rate `pDisc = p10 + p01` and marginal effect
918
+ * `δ = p10 − p01`,
919
+ * n = ( z_{1-α/2}·√pDisc + z_{1-β}·√(pDisc − δ²) )² / δ².
920
+ * Returns Infinity when there is no effect (p10 === p01). Asymptotic — at the
921
+ * tiny discordant counts where the exact {@link mcnemar} differs from the normal
922
+ * approximation, treat the result as a lower bound and prefer the discordant-pair
923
+ * floor.
924
+ */
925
+ function mcnemarRequiredN(opts) {
926
+ const { p10, p01 } = opts;
927
+ if (p10 < 0 || p01 < 0 || p10 + p01 > 1) throw new Error(`mcnemarRequiredN: require p10,p01 ≥ 0 and p10+p01 ≤ 1 (got ${p10}, ${p01})`);
928
+ const delta = p10 - p01;
929
+ if (delta === 0) return Infinity;
930
+ const alpha = opts.alpha ?? .05;
931
+ const power = opts.power ?? .8;
932
+ const twoSided = opts.twoSided ?? true;
933
+ const pDisc = p10 + p01;
934
+ const zAlpha = zQuantile(twoSided ? 1 - alpha / 2 : 1 - alpha);
935
+ const zBeta = zQuantile(power);
936
+ const n = (zAlpha * Math.sqrt(pDisc) + zBeta * Math.sqrt(Math.max(0, pDisc - delta * delta))) ** 2 / (delta * delta);
937
+ return Math.ceil(n);
938
+ }
939
+ /**
940
+ * Power of a McNemar test at a given number of paired observations, the inverse
941
+ * of {@link mcnemarRequiredN} (same Lachin asymptotic model, same parameters).
942
+ * Returns a value in [0, 1]; equals `alpha` when there is no effect.
943
+ */
944
+ function mcnemarPower(opts) {
945
+ const { p10, p01, nPairs } = opts;
946
+ if (p10 < 0 || p01 < 0 || p10 + p01 > 1) throw new Error(`mcnemarPower: require p10,p01 ≥ 0 and p10+p01 ≤ 1 (got ${p10}, ${p01})`);
947
+ const alpha = opts.alpha ?? .05;
948
+ const twoSided = opts.twoSided ?? true;
949
+ const delta = p10 - p01;
950
+ if (delta === 0 || nPairs <= 0) return alpha;
951
+ const pDisc = p10 + p01;
952
+ const zAlpha = zQuantile(twoSided ? 1 - alpha / 2 : 1 - alpha);
953
+ const denom = Math.sqrt(Math.max(1e-12, pDisc - delta * delta));
954
+ const zBeta = (Math.sqrt(nPairs) * Math.abs(delta) - zAlpha * Math.sqrt(pDisc)) / denom;
955
+ return Math.min(1, Math.max(0, normalCdf(zBeta)));
956
+ }
957
+ /** Bonferroni adjustment: multiply every p-value by the test count, clamp at 1. */
958
+ function bonferroni(pValues, alpha = .05) {
959
+ const k = pValues.length;
960
+ const adjusted = pValues.map((p) => Math.min(1, p * k));
961
+ return {
962
+ adjusted,
963
+ significant: adjusted.map((p) => p < alpha)
964
+ };
965
+ }
966
+ /**
967
+ * Holm step-down family-wise error adjustment.
968
+ *
969
+ * P-values are sorted from smallest to largest, multiplied by their remaining
970
+ * hypothesis count, and made monotonically non-decreasing before being mapped
971
+ * back to input order. This uniformly dominates plain Bonferroni while keeping
972
+ * strong family-wise error control under arbitrary dependence.
973
+ */
974
+ function holm(pValues, alpha = .05) {
975
+ if (!Number.isFinite(alpha) || alpha <= 0 || alpha >= 1) throw new ValidationError(`holm: alpha must be in (0,1), got ${alpha}`);
976
+ for (const [index, pValue] of pValues.entries()) if (!Number.isFinite(pValue) || pValue < 0 || pValue > 1) throw new ValidationError(`holm: pValues[${index}] must be in [0,1], got ${pValue}`);
977
+ const count = pValues.length;
978
+ if (count === 0) return {
979
+ adjusted: [],
980
+ significant: []
981
+ };
982
+ const ordered = pValues.map((pValue, index) => ({
983
+ pValue,
984
+ index
985
+ })).sort((a, b) => a.pValue - b.pValue || a.index - b.index);
986
+ const adjusted = new Array(count);
987
+ let previous = 0;
988
+ for (let rank = 0; rank < count; rank++) {
989
+ const entry = ordered[rank];
990
+ const stepAdjusted = Math.min(1, entry.pValue * (count - rank));
991
+ previous = Math.max(previous, stepAdjusted);
992
+ adjusted[entry.index] = previous;
993
+ }
994
+ return {
995
+ adjusted,
996
+ significant: adjusted.map((pValue) => pValue <= alpha)
997
+ };
998
+ }
999
+ /**
1000
+ * Benjamini–Hochberg false discovery rate. Returns adjusted q-values and
1001
+ * significance at the target FDR; handles ties and preserves q monotonicity.
1002
+ */
1003
+ function benjaminiHochberg(pValues, fdr = .05) {
1004
+ const n = pValues.length;
1005
+ if (n === 0) return {
1006
+ qValues: [],
1007
+ significant: []
1008
+ };
1009
+ const indexed = pValues.map((p, i) => ({
1010
+ p,
1011
+ i
1012
+ })).sort((a, b) => a.p - b.p);
1013
+ const q = new Array(n);
1014
+ let minRight = 1;
1015
+ for (let k = n - 1; k >= 0; k--) {
1016
+ const rank = k + 1;
1017
+ const entry = indexed[k];
1018
+ const raw = entry.p * n / rank;
1019
+ const bounded = Math.min(minRight, raw);
1020
+ minRight = bounded;
1021
+ q[entry.i] = Math.min(1, bounded);
1022
+ }
1023
+ return {
1024
+ qValues: q,
1025
+ significant: q.map((v) => v < fdr)
1026
+ };
1027
+ }
1028
+ /**
1029
+ * Paired bootstrap on (after − before) deltas. Returns a CI on the chosen
1030
+ * statistic (median by default); pairs are resampled with replacement. The
1031
+ * lower bound is what the promotion gate checks — `low > threshold` means the
1032
+ * gain is real at the confidence level. Throws on unequal sample sizes.
1033
+ */
1034
+ function pairedBootstrap(before, after, opts = {}) {
1035
+ if (before.length !== after.length) throw new Error(`pairedBootstrap: unequal sample sizes (${before.length} vs ${after.length})`);
1036
+ const confidence = opts.confidence ?? .95;
1037
+ const resamples = opts.resamples ?? 2e3;
1038
+ const statistic = opts.statistic ?? "median";
1039
+ if (confidence <= 0 || confidence >= 1) throw new Error(`pairedBootstrap: confidence must be in (0,1), got ${confidence}`);
1040
+ const n = before.length;
1041
+ const deltas = before.map((b, i) => after[i] - b);
1042
+ if (n === 0) return {
1043
+ n: 0,
1044
+ median: 0,
1045
+ mean: 0,
1046
+ low: 0,
1047
+ high: 0,
1048
+ confidence,
1049
+ resamples
1050
+ };
1051
+ if (n === 1) {
1052
+ const d = deltas[0];
1053
+ return {
1054
+ n: 1,
1055
+ median: d,
1056
+ mean: d,
1057
+ low: d,
1058
+ high: d,
1059
+ confidence,
1060
+ resamples
1061
+ };
1062
+ }
1063
+ const rng = makeRng(opts.seed);
1064
+ const samples = new Array(resamples);
1065
+ for (let b = 0; b < resamples; b++) if (statistic === "mean") {
1066
+ let sum = 0;
1067
+ for (let k = 0; k < n; k++) sum += deltas[Math.floor(rng() * n)];
1068
+ samples[b] = sum / n;
1069
+ } else {
1070
+ const acc = new Array(n);
1071
+ for (let k = 0; k < n; k++) acc[k] = deltas[Math.floor(rng() * n)];
1072
+ samples[b] = medianInPlace(acc);
1073
+ }
1074
+ samples.sort((a, b) => a - b);
1075
+ const alpha = 1 - confidence;
1076
+ const lowIdx = Math.floor(alpha / 2 * resamples);
1077
+ const highIdx = Math.min(resamples - 1, Math.ceil((1 - alpha / 2) * resamples) - 1);
1078
+ return {
1079
+ n,
1080
+ median: medianInPlace([...deltas]),
1081
+ mean: deltas.reduce((s, x) => s + x, 0) / n,
1082
+ low: samples[lowIdx],
1083
+ high: samples[Math.max(highIdx, lowIdx)],
1084
+ confidence,
1085
+ resamples
1086
+ };
1087
+ }
1088
+ /**
1089
+ * Exact one-sided sign test over paired differences.
1090
+ *
1091
+ * Pass `after[i] - before[i]` for each matched item. `alternative = 'greater'`
1092
+ * tests whether positive signs are more likely than negative signs and returns
1093
+ * `P(Binomial(nNonTies, 0.5) >= positive)`. `alternative = 'less'` treats
1094
+ * negative signs as successes instead. With a continuous difference
1095
+ * distribution this is the usual directional median test. Exact zero
1096
+ * differences are ties and do not enter the binomial denominator. All-tie and
1097
+ * empty inputs return p = 1. Every input difference must be finite, and the
1098
+ * direction must be chosen explicitly so a caller cannot select it after
1099
+ * seeing the signs.
1100
+ */
1101
+ function pairedSignTest(differences, alternative) {
1102
+ if (alternative !== "greater" && alternative !== "less") throw new ValidationError(`pairedSignTest: alternative must be 'greater' or 'less', got ${alternative}`);
1103
+ let positive = 0;
1104
+ let negative = 0;
1105
+ let ties = 0;
1106
+ for (let i = 0; i < differences.length; i++) {
1107
+ const difference = differences[i];
1108
+ if (!Number.isFinite(difference)) throw new ValidationError(`pairedSignTest: difference at index ${i} must be finite, got ${difference}`);
1109
+ if (difference > 0) positive++;
1110
+ else if (difference < 0) negative++;
1111
+ else ties++;
1112
+ }
1113
+ const nNonTies = positive + negative;
1114
+ const successes = alternative === "greater" ? positive : negative;
1115
+ return {
1116
+ n: differences.length,
1117
+ positive,
1118
+ negative,
1119
+ ties,
1120
+ nNonTies,
1121
+ alternative,
1122
+ pValue: binomialHalfUpperTail(successes, nNonTies)
1123
+ };
1124
+ }
1125
+ /**
1126
+ * Wilson score interval for a binomial proportion. Correct at small n and near
1127
+ * 0/1, where the normal (Wald) approximation produces bounds outside [0, 1] and
1128
+ * understates coverage. Use this for any pass-rate / hit-rate / realness-rate
1129
+ * CI — the continuous `confidenceInterval` assumes the wrong distribution for a
1130
+ * proportion. `n = 0 ⇒ {0, 0, 0}`.
1131
+ */
1132
+ function wilson(successes, n, confidence = .95) {
1133
+ if (n <= 0) return {
1134
+ estimate: 0,
1135
+ lower: 0,
1136
+ upper: 0
1137
+ };
1138
+ if (successes < 0 || successes > n) throw new Error(`wilson: successes (${successes}) must be in [0, ${n}]`);
1139
+ const z = zQuantile(1 - (1 - confidence) / 2);
1140
+ const p = successes / n;
1141
+ const z2 = z * z;
1142
+ const denom = 1 + z2 / n;
1143
+ const center = (p + z2 / (2 * n)) / denom;
1144
+ const half = z * Math.sqrt((p * (1 - p) + z2 / (4 * n)) / n) / denom;
1145
+ return {
1146
+ estimate: p,
1147
+ lower: Math.max(0, center - half),
1148
+ upper: Math.min(1, center + half)
1149
+ };
1150
+ }
1151
+ /**
1152
+ * McNemar's test for paired binary outcomes — the correct significance test for
1153
+ * "does treatment change the success rate vs control on the SAME items". Only
1154
+ * discordant pairs (one arm right, the other wrong) carry information; concordant
1155
+ * pairs are uninformative, so a paired t-test / two-proportion z-test on the raw
1156
+ * rates is wrong here. The p-value is exact: under H0 the b "treatment-wins" are
1157
+ * Binomial(b + c, 0.5), so the two-sided p is the doubled binomial tail — correct
1158
+ * at the small discordant counts typical of eval runs (no continuity-corrected
1159
+ * chi-square approximation needed, though it is returned as `statistic` for
1160
+ * reference). Inputs are paired 0/1 (or boolean) arrays, control first to match
1161
+ * the module's (before, after) convention. Throws on unequal lengths.
1162
+ */
1163
+ function mcnemar(control, treatment) {
1164
+ if (control.length !== treatment.length) throw new Error(`mcnemar: unequal sample sizes (${control.length} vs ${treatment.length})`);
1165
+ const n = control.length;
1166
+ let b = 0;
1167
+ let c = 0;
1168
+ for (let i = 0; i < n; i++) {
1169
+ const ctrl = control[i] ? 1 : 0;
1170
+ const treat = treatment[i] ? 1 : 0;
1171
+ if (treat === 1 && ctrl === 0) b++;
1172
+ else if (treat === 0 && ctrl === 1) c++;
1173
+ }
1174
+ const nDiscordant = b + c;
1175
+ const statistic = nDiscordant === 0 ? 0 : (Math.abs(b - c) - 1) ** 2 / nDiscordant;
1176
+ return {
1177
+ n,
1178
+ nDiscordant,
1179
+ b,
1180
+ c,
1181
+ statistic,
1182
+ pValue: binomialSignTwoSided(b, c)
1183
+ };
1184
+ }
1185
+ /**
1186
+ * Paired risk difference (the effect-size companion to {@link mcnemar}): the
1187
+ * change in success rate p(treatment) − p(control) on matched items, which for
1188
+ * paired binary data equals (b − c) / n. The CI uses the paired variance from
1189
+ * the discordant counts, not the independent-samples formula (which overstates
1190
+ * the interval by ignoring the pairing). Inputs are paired 0/1 (or boolean)
1191
+ * arrays, control first. Throws on unequal lengths.
1192
+ */
1193
+ function pairedRiskDifference(control, treatment, confidence = .95) {
1194
+ if (control.length !== treatment.length) throw new Error(`pairedRiskDifference: unequal sample sizes (${control.length} vs ${treatment.length})`);
1195
+ const n = control.length;
1196
+ if (n === 0) return {
1197
+ n: 0,
1198
+ b: 0,
1199
+ c: 0,
1200
+ riskDifference: 0,
1201
+ lower: 0,
1202
+ upper: 0,
1203
+ confidence
1204
+ };
1205
+ let b = 0;
1206
+ let c = 0;
1207
+ for (let i = 0; i < n; i++) {
1208
+ const ctrl = control[i] ? 1 : 0;
1209
+ const treat = treatment[i] ? 1 : 0;
1210
+ if (treat === 1 && ctrl === 0) b++;
1211
+ else if (treat === 0 && ctrl === 1) c++;
1212
+ }
1213
+ const rd = (b - c) / n;
1214
+ const variance = (b + c - (b - c) ** 2 / n) / (n * n);
1215
+ const half = zQuantile(1 - (1 - confidence) / 2) * Math.sqrt(Math.max(0, variance));
1216
+ return {
1217
+ n,
1218
+ b,
1219
+ c,
1220
+ riskDifference: rd,
1221
+ lower: Math.max(-1, rd - half),
1222
+ upper: Math.min(1, rd + half),
1223
+ confidence
1224
+ };
1225
+ }
1226
+ /**
1227
+ * Unbiased pass@k for code generation (Chen et al. 2021, "Evaluating Large
1228
+ * Language Models Trained on Code"). Given `n` independent samples for one
1229
+ * problem of which `c` pass, the probability that at least one of a random k of
1230
+ * them passes is 1 − C(n−c, k) / C(n, k). Estimating pass@k as "did any of the
1231
+ * first k pass" is biased high at small n; this is the variance-reduced estimator
1232
+ * averaged implicitly over all k-subsets. Average the per-problem values across
1233
+ * the suite for the corpus pass@k. Computed in the numerically stable product
1234
+ * form. Requires 1 ≤ k ≤ n and 0 ≤ c ≤ n.
1235
+ */
1236
+ function passAtK(n, c, k) {
1237
+ if (!Number.isInteger(n) || !Number.isInteger(c) || !Number.isInteger(k)) throw new Error(`passAtK: n, c, k must be integers (got n=${n}, c=${c}, k=${k})`);
1238
+ if (k < 1 || k > n || c < 0 || c > n) throw new Error(`passAtK: require 1 ≤ k ≤ n and 0 ≤ c ≤ n (got n=${n}, c=${c}, k=${k})`);
1239
+ if (n - c < k) return 1;
1240
+ let prob = 1;
1241
+ for (let i = n - c + 1; i <= n; i++) prob *= 1 - k / i;
1242
+ return 1 - prob;
1243
+ }
1244
+ /**
1245
+ * Two-sided exact p-value for b successes out of (b + c) Bernoulli(0.5) trials —
1246
+ * the exact-binomial core of {@link mcnemar}. `min(1, 2·P(X ≤ min(b,c)))`. No
1247
+ * discordant pairs ⇒ no evidence ⇒ p = 1. Summed in log space (lnGamma) so it
1248
+ * stays exact at large discordant counts without overflow.
1249
+ */
1250
+ function binomialSignTwoSided(b, c) {
1251
+ const nd = b + c;
1252
+ if (nd === 0) return 1;
1253
+ return Math.min(1, 2 * binomialHalfLowerTail(Math.min(b, c), nd));
1254
+ }
1255
+ /** P(X >= successes) for X ~ Binomial(n, 0.5). */
1256
+ function binomialHalfUpperTail(successes, n) {
1257
+ if (successes <= 0) return 1;
1258
+ if (successes > n) return 0;
1259
+ if (successes <= n / 2) return Math.max(0, 1 - binomialHalfLowerTail(successes - 1, n));
1260
+ return binomialHalfLowerTail(n - successes, n);
1261
+ }
1262
+ /** P(X <= maxSuccesses) for X ~ Binomial(n, 0.5), accumulated in log space. */
1263
+ function binomialHalfLowerTail(maxSuccesses, n) {
1264
+ if (maxSuccesses < 0) return 0;
1265
+ if (maxSuccesses >= n) return 1;
1266
+ if (maxSuccesses === 0) return 2 ** -n;
1267
+ const logHalfN = n * Math.log(.5);
1268
+ let logTail = Number.NEGATIVE_INFINITY;
1269
+ for (let i = 0; i <= maxSuccesses; i++) {
1270
+ const logChoose = lnGamma(n + 1) - lnGamma(i + 1) - lnGamma(n - i + 1);
1271
+ logTail = logAddExp(logTail, logChoose + logHalfN);
1272
+ }
1273
+ return Math.min(1, Math.exp(logTail));
1274
+ }
1275
+ function logAddExp(a, b) {
1276
+ if (a === Number.NEGATIVE_INFINITY) return b;
1277
+ if (b === Number.NEGATIVE_INFINITY) return a;
1278
+ const max = Math.max(a, b);
1279
+ return max + Math.log1p(Math.exp(Math.min(a, b) - max));
1280
+ }
1281
+ /**
1282
+ * Betting test-martingale for bounded observations — the e-process core of
1283
+ * anytime-valid sequential testing (Waudby-Smith & Ramdas, "Estimating means
1284
+ * of bounded random variables by betting", JRSS-B 2024).
1285
+ *
1286
+ * Observations x_i ∈ [0,1]; H0: E[x] ≤ m₀ (`nullMean`, default 1/2). Wealth
1287
+ *
1288
+ * W_t = Π_{i≤t} (1 + λ_i (x_i − m₀)), W_0 = 1
1289
+ *
1290
+ * with the truncated GROW-style plug-in bet computed from PRIOR observations:
1291
+ *
1292
+ * λ_i = clamp((μ̂_{i−1} − m₀) / (σ̂²_{i−1} + (μ̂_{i−1} − m₀)²), 0, maxBet)
1293
+ *
1294
+ * where μ̂/σ̂² are the shrunk running estimates μ̂_t = (1/2 + Σx_i)/(t+1),
1295
+ * σ̂²_t = (1/4 + Σ(x_i − μ̂_i)²)/(t+1).
1296
+ *
1297
+ * PREDICTABILITY INVARIANT (load-bearing): λ_i is a function of x_1..x_{i−1}
1298
+ * ONLY — it may never see x_i. With λ_i ≥ 0 predictable, each factor has
1299
+ * E[1 + λ_i(x_i − m₀) | past] ≤ 1 under H0, so W is a nonnegative
1300
+ * supermartingale and Ville's inequality gives P(∃t: W_t ≥ 1/α) ≤ α — the
1301
+ * type-I guarantee holds at ANY data-dependent stopping time. λ_1 is always 0
1302
+ * (no prior evidence), so the first observation never moves wealth.
1303
+ *
1304
+ * `decided` latches at the first crossing W_t ≥ 1/α and never un-latches;
1305
+ * wealth keeps updating after the crossing (the e-process remains valid), but
1306
+ * the decision time is the first crossing.
1307
+ */
1308
+ function eProcess(opts = {}) {
1309
+ const alpha = opts.alpha ?? .05;
1310
+ const maxBet = opts.maxBet ?? .5;
1311
+ const nullMean = opts.nullMean ?? .5;
1312
+ if (!Number.isFinite(alpha) || alpha <= 0 || alpha >= 1) throw new ValidationError(`eProcess: alpha must be in (0,1), got ${alpha}`);
1313
+ if (!Number.isFinite(nullMean) || nullMean <= 0 || nullMean >= 1) throw new ValidationError(`eProcess: nullMean must be in (0,1), got ${nullMean}`);
1314
+ if (!Number.isFinite(maxBet) || maxBet <= 0 || maxBet >= 1 / nullMean) throw new ValidationError(`eProcess: maxBet must be in (0, 1/nullMean=${(1 / nullMean).toFixed(4)}) so wealth factors stay positive, got ${maxBet}`);
1315
+ const threshold = 1 / alpha;
1316
+ let wealth = 1;
1317
+ let n = 0;
1318
+ let decided = false;
1319
+ let decidedAtN;
1320
+ let sumX = 0;
1321
+ let varSum = 0;
1322
+ return {
1323
+ update(x) {
1324
+ if (typeof x !== "number" || !Number.isFinite(x) || x < 0 || x > 1) throw new ValidationError(`eProcess: observation must be a finite number in [0,1], got ${x}`);
1325
+ const muPrev = (.5 + sumX) / (n + 1);
1326
+ const varPrev = (.25 + varSum) / (n + 1);
1327
+ const edge = muPrev - nullMean;
1328
+ const lambda = Math.min(maxBet, Math.max(0, edge / (varPrev + edge * edge)));
1329
+ wealth *= 1 + lambda * (x - nullMean);
1330
+ n += 1;
1331
+ sumX += x;
1332
+ const muNow = (.5 + sumX) / (n + 1);
1333
+ varSum += (x - muNow) ** 2;
1334
+ if (!decided && wealth >= threshold) {
1335
+ decided = true;
1336
+ decidedAtN = n;
1337
+ }
1338
+ return {
1339
+ wealth,
1340
+ n,
1341
+ decided
1342
+ };
1343
+ },
1344
+ state() {
1345
+ return {
1346
+ wealth,
1347
+ n,
1348
+ decided,
1349
+ alpha,
1350
+ maxBet,
1351
+ nullMean,
1352
+ threshold,
1353
+ decidedAtN
1354
+ };
1355
+ }
1356
+ };
1357
+ }
1358
+ /** Standard-normal inverse CDF (Acklam approximation). */
1359
+ function zQuantile(p) {
1360
+ if (p <= 0 || p >= 1) {
1361
+ if (p === 0) return -Infinity;
1362
+ if (p === 1) return Infinity;
1363
+ return NaN;
1364
+ }
1365
+ const a = [
1366
+ -39.69683028665376,
1367
+ 220.9460984245205,
1368
+ -275.9285104469687,
1369
+ 138.357751867269,
1370
+ -30.66479806614716,
1371
+ 2.506628277459239
1372
+ ];
1373
+ const b = [
1374
+ -54.47609879822406,
1375
+ 161.5858368580409,
1376
+ -155.6989798598866,
1377
+ 66.80131188771972,
1378
+ -13.28068155288572
1379
+ ];
1380
+ const c = [
1381
+ -.007784894002430293,
1382
+ -.3223964580411365,
1383
+ -2.400758277161838,
1384
+ -2.549732539343734,
1385
+ 4.374664141464968,
1386
+ 2.938163982698783
1387
+ ];
1388
+ const d = [
1389
+ .007784695709041462,
1390
+ .3224671290700398,
1391
+ 2.445134137142996,
1392
+ 3.754408661907416
1393
+ ];
1394
+ const pLow = .02425;
1395
+ const pHigh = 1 - pLow;
1396
+ let q;
1397
+ let r;
1398
+ if (p < pLow) {
1399
+ q = Math.sqrt(-2 * Math.log(p));
1400
+ return (((((c[0] * q + c[1]) * q + c[2]) * q + c[3]) * q + c[4]) * q + c[5]) / ((((d[0] * q + d[1]) * q + d[2]) * q + d[3]) * q + 1);
1401
+ }
1402
+ if (p <= pHigh) {
1403
+ q = p - .5;
1404
+ r = q * q;
1405
+ return (((((a[0] * r + a[1]) * r + a[2]) * r + a[3]) * r + a[4]) * r + a[5]) * q / (((((b[0] * r + b[1]) * r + b[2]) * r + b[3]) * r + b[4]) * r + 1);
1406
+ }
1407
+ q = Math.sqrt(-2 * Math.log(1 - p));
1408
+ return -(((((c[0] * q + c[1]) * q + c[2]) * q + c[3]) * q + c[4]) * q + c[5]) / ((((d[0] * q + d[1]) * q + d[2]) * q + d[3]) * q + 1);
1409
+ }
1410
+ function medianInPlace(xs) {
1411
+ if (xs.length === 0) return 0;
1412
+ xs.sort((a, b) => a - b);
1413
+ const mid = Math.floor(xs.length / 2);
1414
+ return xs.length % 2 === 0 ? (xs[mid - 1] + xs[mid]) / 2 : xs[mid];
1415
+ }
1416
+ function makeRng(seed) {
1417
+ if (seed === void 0) return Math.random;
1418
+ return mulberry32(seed);
1419
+ }
1420
+ /** Tiny seedable PRNG (mulberry32) — deterministic resampling/shuffling, not
1421
+ * cryptographic. Exported so e-process shuffles and bootstrap resampling
1422
+ * share ONE PRNG implementation; a seed is REQUIRED (unseeded randomness in
1423
+ * gate verdicts is non-reproducible by construction). */
1424
+ function mulberry32(seed) {
1425
+ let s = seed | 0 || 2654435769;
1426
+ return () => {
1427
+ s = s + 1831565813 | 0;
1428
+ let t = s;
1429
+ t = Math.imul(t ^ t >>> 15, t | 1);
1430
+ t ^= t + Math.imul(t ^ t >>> 7, t | 61);
1431
+ return ((t ^ t >>> 14) >>> 0) / 4294967296;
1432
+ };
1433
+ }
1434
+ //#endregion
1435
+ export { spearmanR as A, verbosityBias as B, pairedTTest as C, ranks as D, pearsonR as E, calibrateJudge as F, calibrateJudgeContinuous as I, continuousAgreement as L, weightedMean as M, wilcoxonSignedRank as N, requiredPairedSampleSize as O, wilson as P, positionalBias as R, pairedSignTest as S, passAtK as T, normalizeScores as _, confidenceInterval as a, pairedMde as b, eProcess as c, interpretCliffs as d, mannWhitneyU as f, mulberry32 as g, mcnemarRequiredN as h, cohensD as i, weightedComposite as j, requiredSampleSize as k, holm as l, mcnemarPower as m, bonferroni as n, corpusInterRaterAgreement as o, mcnemar as p, cliffsDelta as r, corpusInterRaterAgreementFromJudgeScores as s, benjaminiHochberg as t, interRaterReliability as u, pairedBootstrap as v, partialCredit as w, pairedRiskDifference as x, pairedCohensDz as y, selfPreference as z };
1436
+
1437
+ //# sourceMappingURL=statistics-CnnxdpOg.js.map