@tangle-network/agent-eval 0.128.2 → 0.130.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (424) hide show
  1. package/CHANGELOG.md +279 -0
  2. package/README.md +19 -0
  3. package/dist/active-curriculum-C4mk67HP.js +214 -0
  4. package/dist/active-curriculum-C4mk67HP.js.map +1 -0
  5. package/dist/adversarial-smnADNFS.d.ts +21 -0
  6. package/dist/adversarial-smnADNFS.d.ts.map +1 -0
  7. package/dist/analyst/index.d.ts +83 -2932
  8. package/dist/analyst/index.d.ts.map +1 -0
  9. package/dist/analyst/index.js +319 -364
  10. package/dist/analyst/index.js.map +1 -1
  11. package/dist/analyst-BkTS3C58.d.ts +89 -0
  12. package/dist/analyst-BkTS3C58.d.ts.map +1 -0
  13. package/dist/analyst-LsnNpSkm.js +152 -0
  14. package/dist/analyst-LsnNpSkm.js.map +1 -0
  15. package/dist/analyze-runs-C1CavBMk.js +1067 -0
  16. package/dist/analyze-runs-C1CavBMk.js.map +1 -0
  17. package/dist/analyze-runs-FsgYCinh.d.ts +72 -0
  18. package/dist/analyze-runs-FsgYCinh.d.ts.map +1 -0
  19. package/dist/attribute-vocabulary-DLJ6303h.d.ts +54 -0
  20. package/dist/attribute-vocabulary-DLJ6303h.d.ts.map +1 -0
  21. package/dist/authenticity/index.d.ts +81 -79
  22. package/dist/authenticity/index.d.ts.map +1 -0
  23. package/dist/authenticity/index.js +209 -193
  24. package/dist/authenticity/index.js.map +1 -1
  25. package/dist/baseline-HsBvw_dk.js +550 -0
  26. package/dist/baseline-HsBvw_dk.js.map +1 -0
  27. package/dist/baseline-hG3K85h4.d.ts +125 -0
  28. package/dist/baseline-hG3K85h4.d.ts.map +1 -0
  29. package/dist/belief-state/index.d.ts +448 -1205
  30. package/dist/belief-state/index.d.ts.map +1 -0
  31. package/dist/belief-state/index.js +1617 -1710
  32. package/dist/belief-state/index.js.map +1 -1
  33. package/dist/benchmarks/index.d.ts +2 -894
  34. package/dist/benchmarks/index.js +2 -59
  35. package/dist/benchmarks-DviOvUNr.js +754 -0
  36. package/dist/benchmarks-DviOvUNr.js.map +1 -0
  37. package/dist/builder-eval/index.d.ts +150 -662
  38. package/dist/builder-eval/index.d.ts.map +1 -0
  39. package/dist/builder-eval/index.js +356 -345
  40. package/dist/builder-eval/index.js.map +1 -1
  41. package/dist/calibration-CNWWA6K8.js +94 -0
  42. package/dist/calibration-CNWWA6K8.js.map +1 -0
  43. package/dist/campaign/index.d.ts +5 -6390
  44. package/dist/campaign/index.js +3 -212
  45. package/dist/campaign-CBKZvQ1H.js +3885 -0
  46. package/dist/campaign-CBKZvQ1H.js.map +1 -0
  47. package/dist/cli.d.ts +1 -1
  48. package/dist/cli.js +164 -174
  49. package/dist/cli.js.map +1 -1
  50. package/dist/client-C97NMzqi.d.ts +581 -0
  51. package/dist/client-C97NMzqi.d.ts.map +1 -0
  52. package/dist/client-CYzbdJOZ.js +637 -0
  53. package/dist/client-CYzbdJOZ.js.map +1 -0
  54. package/dist/code-agent-session-BjkMTQ7H.js +1390 -0
  55. package/dist/code-agent-session-BjkMTQ7H.js.map +1 -0
  56. package/dist/code-agent-session-DqqgOJaz.d.ts +143 -0
  57. package/dist/code-agent-session-DqqgOJaz.d.ts.map +1 -0
  58. package/dist/concurrency-DIxRZF_J.js +85 -0
  59. package/dist/concurrency-DIxRZF_J.js.map +1 -0
  60. package/dist/contract/index.d.ts +645 -5605
  61. package/dist/contract/index.d.ts.map +1 -0
  62. package/dist/contract/index.js +1730 -1937
  63. package/dist/contract/index.js.map +1 -1
  64. package/dist/control.d.ts +3 -1030
  65. package/dist/control.js +2 -32
  66. package/dist/cost-ledger-DIgQUFZZ.js +801 -0
  67. package/dist/cost-ledger-DIgQUFZZ.js.map +1 -0
  68. package/dist/cost-ledger-Dye6jCgg.d.ts +236 -0
  69. package/dist/cost-ledger-Dye6jCgg.d.ts.map +1 -0
  70. package/dist/dataset-BvtnC8Dc.d.ts +115 -0
  71. package/dist/dataset-BvtnC8Dc.d.ts.map +1 -0
  72. package/dist/default-registry-C-vFCSEc.js +2579 -0
  73. package/dist/default-registry-C-vFCSEc.js.map +1 -0
  74. package/dist/default-registry-CNPo-Vsb.d.ts +540 -0
  75. package/dist/default-registry-CNPo-Vsb.d.ts.map +1 -0
  76. package/dist/emitter-CPBAhxum.js +266 -0
  77. package/dist/emitter-CPBAhxum.js.map +1 -0
  78. package/dist/emitter-DGQGoLyj.d.ts +113 -0
  79. package/dist/emitter-DGQGoLyj.d.ts.map +1 -0
  80. package/dist/errors-8YnH8WlF.js +64 -0
  81. package/dist/errors-8YnH8WlF.js.map +1 -0
  82. package/dist/errors-CEk209JS.d.ts +76 -0
  83. package/dist/errors-CEk209JS.d.ts.map +1 -0
  84. package/dist/eval-campaign-DEm6c8ru.js +349 -0
  85. package/dist/eval-campaign-DEm6c8ru.js.map +1 -0
  86. package/dist/execution-tracks-CpgFPpS5.js +93 -0
  87. package/dist/execution-tracks-CpgFPpS5.js.map +1 -0
  88. package/dist/exporters-q9iL-2Jf.js +148 -0
  89. package/dist/exporters-q9iL-2Jf.js.map +1 -0
  90. package/dist/extract-usage-BrQ8mCLX.js +155 -0
  91. package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
  92. package/dist/failure-cluster-CqcvCcdR.d.ts +59 -0
  93. package/dist/failure-cluster-CqcvCcdR.d.ts.map +1 -0
  94. package/dist/feedback-trajectory-CVaeREXV.d.ts +340 -0
  95. package/dist/feedback-trajectory-CVaeREXV.d.ts.map +1 -0
  96. package/dist/fuzz.d.ts +320 -646
  97. package/dist/fuzz.d.ts.map +1 -0
  98. package/dist/fuzz.js +670 -617
  99. package/dist/fuzz.js.map +1 -1
  100. package/dist/hf-dataset-DBJXXoY1.js +763 -0
  101. package/dist/hf-dataset-DBJXXoY1.js.map +1 -0
  102. package/dist/hosted/index.d.ts +11 -831
  103. package/dist/hosted/index.d.ts.map +1 -0
  104. package/dist/hosted/index.js +2 -37
  105. package/dist/index-2JJSA6-r2.d.ts +926 -0
  106. package/dist/index-2JJSA6-r2.d.ts.map +1 -0
  107. package/dist/index-6N0aYmpW.d.ts +217 -0
  108. package/dist/index-6N0aYmpW.d.ts.map +1 -0
  109. package/dist/index-BAvgST_9.d.ts +131 -0
  110. package/dist/index-BAvgST_9.d.ts.map +1 -0
  111. package/dist/index-BvnJuTGD.d.ts +68 -0
  112. package/dist/index-BvnJuTGD.d.ts.map +1 -0
  113. package/dist/index-C61Wi7yg.d.ts +547 -0
  114. package/dist/index-C61Wi7yg.d.ts.map +1 -0
  115. package/dist/index-CAPUUKaM.d.ts +335 -0
  116. package/dist/index-CAPUUKaM.d.ts.map +1 -0
  117. package/dist/index-DE5fb3EC.d.ts +2244 -0
  118. package/dist/index-DE5fb3EC.d.ts.map +1 -0
  119. package/dist/index-DSC51roc.d.ts +102 -0
  120. package/dist/index-DSC51roc.d.ts.map +1 -0
  121. package/dist/index.d.ts +3776 -15120
  122. package/dist/index.d.ts.map +1 -0
  123. package/dist/index.js +11185 -11191
  124. package/dist/index.js.map +1 -1
  125. package/dist/integrity-BzRbCHzi.js +424 -0
  126. package/dist/integrity-BzRbCHzi.js.map +1 -0
  127. package/dist/integrity-rmVhXWA7.d.ts +61 -0
  128. package/dist/integrity-rmVhXWA7.d.ts.map +1 -0
  129. package/dist/judge-calibration-DFtEMlde.d.ts +146 -0
  130. package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
  131. package/dist/ledger-core/index.d.ts +2 -0
  132. package/dist/ledger-core/index.js +2 -0
  133. package/dist/ledger-core-DtZz1RG0.js +388 -0
  134. package/dist/ledger-core-DtZz1RG0.js.map +1 -0
  135. package/dist/llm-client--GR4JbZE.js +687 -0
  136. package/dist/llm-client--GR4JbZE.js.map +1 -0
  137. package/dist/llm-client-B_nIBlYo.d.ts +290 -0
  138. package/dist/llm-client-B_nIBlYo.d.ts.map +1 -0
  139. package/dist/matrix/index.d.ts +3 -155
  140. package/dist/matrix/index.js +2 -8
  141. package/dist/matrix-BzQnu2S6.js +270 -0
  142. package/dist/matrix-BzQnu2S6.js.map +1 -0
  143. package/dist/meta-eval/index.d.ts +4 -1027
  144. package/dist/meta-eval/index.js +393 -390
  145. package/dist/meta-eval/index.js.map +1 -1
  146. package/dist/metrics-C9YY1OcL.js +239 -0
  147. package/dist/metrics-C9YY1OcL.js.map +1 -0
  148. package/dist/mint-yN2M2eh0.js +201 -0
  149. package/dist/mint-yN2M2eh0.js.map +1 -0
  150. package/dist/multi-layer-verifier-BHY1gWAc.d.ts +138 -0
  151. package/dist/multi-layer-verifier-BHY1gWAc.d.ts.map +1 -0
  152. package/dist/multishot/index.d.ts +273 -481
  153. package/dist/multishot/index.d.ts.map +1 -0
  154. package/dist/multishot/index.js +595 -548
  155. package/dist/multishot/index.js.map +1 -1
  156. package/dist/off-policy-DvgzvtIx.js +220 -0
  157. package/dist/off-policy-DvgzvtIx.js.map +1 -0
  158. package/dist/off-policy-mskQw8Mb.d.ts +153 -0
  159. package/dist/off-policy-mskQw8Mb.d.ts.map +1 -0
  160. package/dist/openapi.json +1 -1
  161. package/dist/opencode-sqlite-BGrHeDu3.js +318 -0
  162. package/dist/opencode-sqlite-BGrHeDu3.js.map +1 -0
  163. package/dist/outcome-store-BYHIuO0e.d.ts +65 -0
  164. package/dist/outcome-store-BYHIuO0e.d.ts.map +1 -0
  165. package/dist/outcome-store-ChBKlTd_.js +75 -0
  166. package/dist/outcome-store-ChBKlTd_.js.map +1 -0
  167. package/dist/paired-arms-D9D0wXj2.js +260 -0
  168. package/dist/paired-arms-D9D0wXj2.js.map +1 -0
  169. package/dist/pipelines/index.d.ts +95 -532
  170. package/dist/pipelines/index.d.ts.map +1 -0
  171. package/dist/pipelines/index.js +497 -478
  172. package/dist/pipelines/index.js.map +1 -1
  173. package/dist/pre-registration-DakwTRXk.js +96 -0
  174. package/dist/pre-registration-DakwTRXk.js.map +1 -0
  175. package/dist/propose-review-control-LhIWGzHl.js +1458 -0
  176. package/dist/propose-review-control-LhIWGzHl.js.map +1 -0
  177. package/dist/query-CJ_DX8vl.d.ts +28 -0
  178. package/dist/query-CJ_DX8vl.d.ts.map +1 -0
  179. package/dist/query-Di7eEQ79.js +83 -0
  180. package/dist/query-Di7eEQ79.js.map +1 -0
  181. package/dist/raw-provider-sink-BQd7mzyT.js +201 -0
  182. package/dist/raw-provider-sink-BQd7mzyT.js.map +1 -0
  183. package/dist/raw-provider-sink-BU29Sh8h.d.ts +134 -0
  184. package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +1 -0
  185. package/dist/redact-7Aq1ukl-.js +107 -0
  186. package/dist/redact-7Aq1ukl-.js.map +1 -0
  187. package/dist/release-report-Cz9NKH39.js +603 -0
  188. package/dist/release-report-Cz9NKH39.js.map +1 -0
  189. package/dist/release-report-mvB2G4_J.d.ts +244 -0
  190. package/dist/release-report-mvB2G4_J.d.ts.map +1 -0
  191. package/dist/replay-GnyotH0J.js +1741 -0
  192. package/dist/replay-GnyotH0J.js.map +1 -0
  193. package/dist/replay-OoidtG1E.d.ts +749 -0
  194. package/dist/replay-OoidtG1E.d.ts.map +1 -0
  195. package/dist/reporting.d.ts +6 -1298
  196. package/dist/reporting.js +6 -50
  197. package/dist/researcher-CwTdwXG1.d.ts +314 -0
  198. package/dist/researcher-CwTdwXG1.d.ts.map +1 -0
  199. package/dist/reward-hacking-eAnOsynk.d.ts +224 -0
  200. package/dist/reward-hacking-eAnOsynk.d.ts.map +1 -0
  201. package/dist/reward-hacking-qipEpKvY.js +596 -0
  202. package/dist/reward-hacking-qipEpKvY.js.map +1 -0
  203. package/dist/reward-nw2xZGZG.js +137 -0
  204. package/dist/reward-nw2xZGZG.js.map +1 -0
  205. package/dist/rl.d.ts +916 -3596
  206. package/dist/rl.d.ts.map +1 -0
  207. package/dist/rl.js +2362 -1751
  208. package/dist/rl.js.map +1 -1
  209. package/dist/rolldown-runtime-8H4AJuhK.js +14 -0
  210. package/dist/rollout/index.d.ts +3 -1048
  211. package/dist/rollout/index.js +8 -110
  212. package/dist/rollout-BOYjemfR.js +624 -0
  213. package/dist/rollout-BOYjemfR.js.map +1 -0
  214. package/dist/rubric-predictive-validity-B3xmbmS1.js +141 -0
  215. package/dist/rubric-predictive-validity-B3xmbmS1.js.map +1 -0
  216. package/dist/rubric-predictive-validity-Ku_clp_1.d.ts +74 -0
  217. package/dist/rubric-predictive-validity-Ku_clp_1.d.ts.map +1 -0
  218. package/dist/run-evidence-ClRX_8A9.d.ts +225 -0
  219. package/dist/run-evidence-ClRX_8A9.d.ts.map +1 -0
  220. package/dist/run-record-BuoE80Dq.js +467 -0
  221. package/dist/run-record-BuoE80Dq.js.map +1 -0
  222. package/dist/run-record-CnZu_gjl.d.ts +357 -0
  223. package/dist/run-record-CnZu_gjl.d.ts.map +1 -0
  224. package/dist/run-score-iEEAWiBY.js +41 -0
  225. package/dist/run-score-iEEAWiBY.js.map +1 -0
  226. package/dist/runtime-trajectory-1gyaTOoC.js +93 -0
  227. package/dist/runtime-trajectory-1gyaTOoC.js.map +1 -0
  228. package/dist/runtime-trajectory-BvSZcCHD.d.ts +50 -0
  229. package/dist/runtime-trajectory-BvSZcCHD.d.ts.map +1 -0
  230. package/dist/schema-BtVldJ3T.d.ts +206 -0
  231. package/dist/schema-BtVldJ3T.d.ts.map +1 -0
  232. package/dist/schema-C6DW4ZHR.js +821 -0
  233. package/dist/schema-C6DW4ZHR.js.map +1 -0
  234. package/dist/schema-CRhEY1SO.js +69 -0
  235. package/dist/schema-CRhEY1SO.js.map +1 -0
  236. package/dist/schema-Cef2cFmb.d.ts +408 -0
  237. package/dist/schema-Cef2cFmb.d.ts.map +1 -0
  238. package/dist/semantic-concept-judge-B6cWNJ2K.js +725 -0
  239. package/dist/semantic-concept-judge-B6cWNJ2K.js.map +1 -0
  240. package/dist/sequential-Br0mAPHA.js +148 -0
  241. package/dist/sequential-Br0mAPHA.js.map +1 -0
  242. package/dist/sequential-CYwq6Ff_.d.ts +141 -0
  243. package/dist/sequential-CYwq6Ff_.d.ts.map +1 -0
  244. package/dist/series-convergence-CjO2QdRW.js +43 -0
  245. package/dist/series-convergence-CjO2QdRW.js.map +1 -0
  246. package/dist/series-convergence-ofsqPWhs.d.ts +35 -0
  247. package/dist/series-convergence-ofsqPWhs.d.ts.map +1 -0
  248. package/dist/server-m5D9cvnG.js +1040 -0
  249. package/dist/server-m5D9cvnG.js.map +1 -0
  250. package/dist/skill-usage-C_pXm7lP.d.ts +534 -0
  251. package/dist/skill-usage-C_pXm7lP.d.ts.map +1 -0
  252. package/dist/skillopt-optimization-method-B7wX7XkF.d.ts +1739 -0
  253. package/dist/skillopt-optimization-method-B7wX7XkF.d.ts.map +1 -0
  254. package/dist/skillopt-optimization-method-D4ODwFVV.js +7712 -0
  255. package/dist/skillopt-optimization-method-D4ODwFVV.js.map +1 -0
  256. package/dist/statistics-Cmj6nynr.d.ts +514 -0
  257. package/dist/statistics-Cmj6nynr.d.ts.map +1 -0
  258. package/dist/statistics-CnnxdpOg.js +1437 -0
  259. package/dist/statistics-CnnxdpOg.js.map +1 -0
  260. package/dist/store-CT9YIIve.d.ts +117 -0
  261. package/dist/store-CT9YIIve.d.ts.map +1 -0
  262. package/dist/store-CxJry_cs.d.ts +229 -0
  263. package/dist/store-CxJry_cs.d.ts.map +1 -0
  264. package/dist/storyboard/index.d.ts +84 -203
  265. package/dist/storyboard/index.d.ts.map +1 -0
  266. package/dist/storyboard/index.js +609 -542
  267. package/dist/storyboard/index.js.map +1 -1
  268. package/dist/summary-report-BNs5nmXI.js +862 -0
  269. package/dist/summary-report-BNs5nmXI.js.map +1 -0
  270. package/dist/summary-report-Cj9gdw4i.d.ts +407 -0
  271. package/dist/summary-report-Cj9gdw4i.d.ts.map +1 -0
  272. package/dist/supervisor-run/index.d.ts +2 -849
  273. package/dist/supervisor-run/index.js +2 -64
  274. package/dist/supervisor-run-_lnTLM3z.js +1679 -0
  275. package/dist/supervisor-run-_lnTLM3z.js.map +1 -0
  276. package/dist/task-failure-attributes-CQZlB3et.js +311 -0
  277. package/dist/task-failure-attributes-CQZlB3et.js.map +1 -0
  278. package/dist/test-graded-scenario-BsqWLmPt.js +318 -0
  279. package/dist/test-graded-scenario-BsqWLmPt.js.map +1 -0
  280. package/dist/test-graded-scenario-D1TaI2va.d.ts +141 -0
  281. package/dist/test-graded-scenario-D1TaI2va.d.ts.map +1 -0
  282. package/dist/tools-BmuN627J.js +1085 -0
  283. package/dist/tools-BmuN627J.js.map +1 -0
  284. package/dist/trace-attributes.d.ts +2 -52
  285. package/dist/trace-attributes.js +131 -61
  286. package/dist/trace-attributes.js.map +1 -1
  287. package/dist/traces.d.ts +12 -2365
  288. package/dist/traces.js +12 -251
  289. package/dist/trajectory-D_7rLrvE.js +56 -0
  290. package/dist/trajectory-D_7rLrvE.js.map +1 -0
  291. package/dist/types-DGsxbAEd.d.ts +387 -0
  292. package/dist/types-DGsxbAEd.d.ts.map +1 -0
  293. package/dist/types-k9tZGKUg.d.ts +640 -0
  294. package/dist/types-k9tZGKUg.d.ts.map +1 -0
  295. package/dist/verdict-Dps8_okt.d.ts +37 -0
  296. package/dist/verdict-Dps8_okt.d.ts.map +1 -0
  297. package/dist/wire/index.d.ts +702 -1174
  298. package/dist/wire/index.d.ts.map +1 -0
  299. package/dist/wire/index.js +2 -81
  300. package/docs/feature-guide.md +1 -1
  301. package/docs/rollout.md +116 -2
  302. package/package.json +18 -10
  303. package/dist/benchmarks/index.js.map +0 -1
  304. package/dist/campaign/index.js.map +0 -1
  305. package/dist/chunk-2JX3CFMB.js +0 -695
  306. package/dist/chunk-2JX3CFMB.js.map +0 -1
  307. package/dist/chunk-2MKQIFS4.js +0 -183
  308. package/dist/chunk-2MKQIFS4.js.map +0 -1
  309. package/dist/chunk-3RF76KTD.js +0 -84
  310. package/dist/chunk-3RF76KTD.js.map +0 -1
  311. package/dist/chunk-5DTSBUL2.js +0 -159
  312. package/dist/chunk-5DTSBUL2.js.map +0 -1
  313. package/dist/chunk-7ZZMD7UK.js +0 -386
  314. package/dist/chunk-7ZZMD7UK.js.map +0 -1
  315. package/dist/chunk-BOD4O7OF.js +0 -40
  316. package/dist/chunk-BOD4O7OF.js.map +0 -1
  317. package/dist/chunk-BYT7ELPS.js +0 -1553
  318. package/dist/chunk-BYT7ELPS.js.map +0 -1
  319. package/dist/chunk-DJKY2TSY.js +0 -2428
  320. package/dist/chunk-DJKY2TSY.js.map +0 -1
  321. package/dist/chunk-DPUHNQLN.js +0 -232
  322. package/dist/chunk-DPUHNQLN.js.map +0 -1
  323. package/dist/chunk-DRYIUNWY.js +0 -622
  324. package/dist/chunk-DRYIUNWY.js.map +0 -1
  325. package/dist/chunk-EJGRPCO3.js +0 -617
  326. package/dist/chunk-EJGRPCO3.js.map +0 -1
  327. package/dist/chunk-EOSZT7PL.js +0 -2001
  328. package/dist/chunk-EOSZT7PL.js.map +0 -1
  329. package/dist/chunk-EZJEIH2R.js +0 -1559
  330. package/dist/chunk-EZJEIH2R.js.map +0 -1
  331. package/dist/chunk-GGE4NNQT.js +0 -65
  332. package/dist/chunk-GGE4NNQT.js.map +0 -1
  333. package/dist/chunk-HHWE3POT.js +0 -94
  334. package/dist/chunk-HHWE3POT.js.map +0 -1
  335. package/dist/chunk-IHQDPH7D.js +0 -171
  336. package/dist/chunk-IHQDPH7D.js.map +0 -1
  337. package/dist/chunk-JHCHEVET.js +0 -274
  338. package/dist/chunk-JHCHEVET.js.map +0 -1
  339. package/dist/chunk-K4DBDHLK.js +0 -158
  340. package/dist/chunk-K4DBDHLK.js.map +0 -1
  341. package/dist/chunk-K6N6XJJX.js +0 -306
  342. package/dist/chunk-K6N6XJJX.js.map +0 -1
  343. package/dist/chunk-MA6HLL3S.js +0 -65
  344. package/dist/chunk-MA6HLL3S.js.map +0 -1
  345. package/dist/chunk-MAZ26DC7.js +0 -99
  346. package/dist/chunk-MAZ26DC7.js.map +0 -1
  347. package/dist/chunk-MHELPNRP.js +0 -1212
  348. package/dist/chunk-MHELPNRP.js.map +0 -1
  349. package/dist/chunk-NACAGYSY.js +0 -1040
  350. package/dist/chunk-NACAGYSY.js.map +0 -1
  351. package/dist/chunk-NKAGIDE2.js +0 -7633
  352. package/dist/chunk-NKAGIDE2.js.map +0 -1
  353. package/dist/chunk-NPCTHQIO.js +0 -91
  354. package/dist/chunk-NPCTHQIO.js.map +0 -1
  355. package/dist/chunk-NYLOYM6N.js +0 -332
  356. package/dist/chunk-NYLOYM6N.js.map +0 -1
  357. package/dist/chunk-ONWEPEDO.js +0 -57
  358. package/dist/chunk-ONWEPEDO.js.map +0 -1
  359. package/dist/chunk-P5W7RQKK.js +0 -576
  360. package/dist/chunk-P5W7RQKK.js.map +0 -1
  361. package/dist/chunk-P6FYH6K4.js +0 -1161
  362. package/dist/chunk-P6FYH6K4.js.map +0 -1
  363. package/dist/chunk-PBE2LOSS.js +0 -669
  364. package/dist/chunk-PBE2LOSS.js.map +0 -1
  365. package/dist/chunk-PC4UYEBM.js +0 -166
  366. package/dist/chunk-PC4UYEBM.js.map +0 -1
  367. package/dist/chunk-PXE2VKMX.js +0 -140
  368. package/dist/chunk-PXE2VKMX.js.map +0 -1
  369. package/dist/chunk-PZ5AY32C.js +0 -10
  370. package/dist/chunk-PZ5AY32C.js.map +0 -1
  371. package/dist/chunk-RZTMDUO7.js +0 -49
  372. package/dist/chunk-RZTMDUO7.js.map +0 -1
  373. package/dist/chunk-S5YLIBFX.js +0 -136
  374. package/dist/chunk-S5YLIBFX.js.map +0 -1
  375. package/dist/chunk-SZLVEKMJ.js +0 -1446
  376. package/dist/chunk-SZLVEKMJ.js.map +0 -1
  377. package/dist/chunk-T4SQEITX.js +0 -95
  378. package/dist/chunk-T4SQEITX.js.map +0 -1
  379. package/dist/chunk-TBL77AUT.js +0 -355
  380. package/dist/chunk-TBL77AUT.js.map +0 -1
  381. package/dist/chunk-TSN7JT6D.js +0 -1646
  382. package/dist/chunk-TSN7JT6D.js.map +0 -1
  383. package/dist/chunk-TT4KNT67.js +0 -124
  384. package/dist/chunk-TT4KNT67.js.map +0 -1
  385. package/dist/chunk-UB2LOJ6Q.js +0 -4461
  386. package/dist/chunk-UB2LOJ6Q.js.map +0 -1
  387. package/dist/chunk-UWZZKKU7.js +0 -237
  388. package/dist/chunk-UWZZKKU7.js.map +0 -1
  389. package/dist/chunk-VBQ3CRKH.js +0 -291
  390. package/dist/chunk-VBQ3CRKH.js.map +0 -1
  391. package/dist/chunk-VGRCHJON.js +0 -163
  392. package/dist/chunk-VGRCHJON.js.map +0 -1
  393. package/dist/chunk-VI2UW6B6.js +0 -162
  394. package/dist/chunk-VI2UW6B6.js.map +0 -1
  395. package/dist/chunk-VLOATJQ2.js +0 -908
  396. package/dist/chunk-VLOATJQ2.js.map +0 -1
  397. package/dist/chunk-VQMK5FMP.js +0 -247
  398. package/dist/chunk-VQMK5FMP.js.map +0 -1
  399. package/dist/chunk-VZSRQ272.js +0 -149
  400. package/dist/chunk-VZSRQ272.js.map +0 -1
  401. package/dist/chunk-WGXIEX7P.js +0 -116
  402. package/dist/chunk-WGXIEX7P.js.map +0 -1
  403. package/dist/chunk-WS3NZZQQ.js +0 -929
  404. package/dist/chunk-WS3NZZQQ.js.map +0 -1
  405. package/dist/chunk-XDWDC2MP.js +0 -695
  406. package/dist/chunk-XDWDC2MP.js.map +0 -1
  407. package/dist/chunk-XPRT64IE.js +0 -766
  408. package/dist/chunk-XPRT64IE.js.map +0 -1
  409. package/dist/chunk-YJBNWCAA.js +0 -1056
  410. package/dist/chunk-YJBNWCAA.js.map +0 -1
  411. package/dist/chunk-ZET2UAYW.js +0 -89
  412. package/dist/chunk-ZET2UAYW.js.map +0 -1
  413. package/dist/chunk-ZUUWPZCV.js +0 -752
  414. package/dist/chunk-ZUUWPZCV.js.map +0 -1
  415. package/dist/control.js.map +0 -1
  416. package/dist/hosted/index.js.map +0 -1
  417. package/dist/matrix/index.js.map +0 -1
  418. package/dist/reporting.js.map +0 -1
  419. package/dist/rollout/index.js.map +0 -1
  420. package/dist/run-campaign-ISHFZ7FJ.js +0 -17
  421. package/dist/run-campaign-ISHFZ7FJ.js.map +0 -1
  422. package/dist/supervisor-run/index.js.map +0 -1
  423. package/dist/traces.js.map +0 -1
  424. package/dist/wire/index.js.map +0 -1
@@ -0,0 +1,514 @@
1
+ import { g as JudgeScore } from "./types-DGsxbAEd.js";
2
+ import { i as ContinuousAgreementOptions, r as ContinuousAgreement } from "./judge-calibration-DFtEMlde.js";
3
+ //#region src/statistics.d.ts
4
+ /** Identity: dimensions already follow "higher = better" by prompt convention
5
+ * (inverted dims like hallucination are scored 10 = best at the source). */
6
+ declare const normalizeScores: (scores: JudgeScore[]) => JudgeScore[];
7
+ /** Weighted mean — falls back to uniform weights when omitted */
8
+ declare function weightedMean(scores: {
9
+ score: number;
10
+ weight?: number;
11
+ }[]): number;
12
+ /** Bootstrap confidence interval */
13
+ declare function confidenceInterval(scores: number[], confidence?: number, opts?: {
14
+ seed?: number;
15
+ resamples?: number;
16
+ }): {
17
+ mean: number;
18
+ lower: number;
19
+ upper: number;
20
+ };
21
+ /**
22
+ * Inter-rater reliability — simplified Krippendorff's alpha.
23
+ *
24
+ * Each inner array is one judge's scores for all items.
25
+ * All arrays must have the same length (same items scored).
26
+ */
27
+ declare function interRaterReliability(judgeScores: JudgeScore[][]): number;
28
+ /**
29
+ * Mann-Whitney U test for comparing two independent groups.
30
+ * Returns U statistic and approximate p-value (normal approximation).
31
+ */
32
+ declare function mannWhitneyU(a: number[], b: number[]): {
33
+ u: number;
34
+ p: number;
35
+ };
36
+ /** Partial credit: returns 0-1 ratio of current toward target */
37
+ declare function partialCredit(current: number, target: number): number;
38
+ /**
39
+ * Paired t-test — before/after measurements on the SAME items.
40
+ * Pairing removes inter-item variance, giving tighter significance than
41
+ * an unpaired test when comparing prompt v1 vs prompt v2 on identical
42
+ * scenarios.
43
+ */
44
+ declare function pairedTTest(before: number[], after: number[]): {
45
+ t: number;
46
+ df: number;
47
+ p: number;
48
+ };
49
+ /**
50
+ * Wilcoxon signed-rank test — paired non-parametric alternative.
51
+ * Use when the differences aren't normally distributed.
52
+ */
53
+ declare function wilcoxonSignedRank(before: number[], after: number[]): {
54
+ w: number;
55
+ p: number;
56
+ };
57
+ /**
58
+ * Cohen's d — standardized effect size for two independent groups.
59
+ * Positive d means group b has higher mean than group a.
60
+ * Rule of thumb: |d| < 0.2 negligible, 0.2–0.5 small, 0.5–0.8 medium, > 0.8 large.
61
+ */
62
+ declare function cohensD(a: number[], b: number[]): number;
63
+ /**
64
+ * Cohen's dz for paired observations: mean(after - before) divided by the
65
+ * sample standard deviation of those within-pair deltas.
66
+ *
67
+ * Returns null when fewer than two pairs exist or a non-zero constant delta
68
+ * has zero observed variance. In that case the standardized effect is
69
+ * undefined, not an arbitrarily large finite number.
70
+ */
71
+ declare function pairedCohensDz(before: number[], after: number[]): number | null;
72
+ type CliffsMagnitude = 'negligible' | 'small' | 'medium' | 'large';
73
+ /**
74
+ * Cliff's delta — a non-parametric effect size for two independent samples.
75
+ * `δ = (#(after > before) − #(after < before)) / (n_before · n_after)`,
76
+ * ranging [-1, 1]. Positive ⇒ `after` tends to exceed `before` (improvement).
77
+ *
78
+ * Distribution-free counterpart to Cohen's d: no normality assumption, robust
79
+ * to the bounded/skewed score distributions judges produce. Pairs with
80
+ * `pairedBootstrap` / `wilcoxonSignedRank` for the non-parametric reporting
81
+ * path. Returns 0 when either sample is empty.
82
+ */
83
+ declare function cliffsDelta(before: number[], after: number[]): number;
84
+ /**
85
+ * Map a Cliff's delta to a qualitative magnitude using the standard
86
+ * Romano et al. thresholds (|δ|): <0.147 negligible, <0.33 small,
87
+ * <0.474 medium, else large.
88
+ */
89
+ declare function interpretCliffs(delta: number): CliffsMagnitude;
90
+ /**
91
+ * Average-rank-with-ties transform (1-indexed). Tied values receive the mean
92
+ * of the ranks they span, the standard correction for Spearman's ρ.
93
+ */
94
+ declare function ranks(xs: number[]): number[];
95
+ /**
96
+ * Pearson product-moment correlation coefficient r ∈ [-1, 1] between two
97
+ * equal-length series. See the edge-case contract above: NaN for n < 2 or
98
+ * unequal lengths, 1 when both series are constant, 0 when exactly one is.
99
+ */
100
+ declare function pearsonR(a: number[], b: number[]): number;
101
+ /**
102
+ * Spearman's rank correlation ρ — Pearson over the average-rank-with-ties
103
+ * transform of each series. Same edge-case contract as {@link pearsonR}.
104
+ */
105
+ declare function spearmanR(a: number[], b: number[]): number;
106
+ interface WeightedCompositeInput {
107
+ /** Per-dimension scores (typically 0..1). */
108
+ dims: Record<string, number>;
109
+ /** Weight per dimension. Every weighted dimension MUST be present in
110
+ * `dims` — a weight for an absent dimension is a config error and throws,
111
+ * because silently dropping it would renormalise the composite onto a
112
+ * different denominator than intended. */
113
+ weights: Record<string, number>;
114
+ /** Optional pass threshold; when set, the result reports `pass`. */
115
+ threshold?: number;
116
+ }
117
+ interface WeightedCompositeResult {
118
+ composite: number;
119
+ pass?: boolean;
120
+ }
121
+ /**
122
+ * Weighted composite over judge dimensions: `Σ(score_d · w_d) / Σ(w_d)` across
123
+ * the weighted dimensions. The canonical replacement for the per-consumer
124
+ * hand-rolled composite math (tax/legal/creative/gtm each ship a copy).
125
+ *
126
+ * Fail-loud: throws if a weighted dimension is missing from `dims`, if any
127
+ * weight is negative, or if the weights sum to 0 — none of which can produce
128
+ * a meaningful composite.
129
+ */
130
+ declare function weightedComposite(input: WeightedCompositeInput): WeightedCompositeResult;
131
+ interface CorpusScoreRecord {
132
+ /** Stable identifier for the rated item (scenario, span, turn, …). */
133
+ itemId: string;
134
+ /** Identifier for the judge that produced this score. */
135
+ judgeName: string;
136
+ /** Dimension name (matches `JudgeScore.dimension`). */
137
+ dimension: string;
138
+ /** Numeric score; must be finite. */
139
+ score: number;
140
+ }
141
+ interface CorpusAgreementPerDimension extends ContinuousAgreement {
142
+ dimension: string;
143
+ /** Item IDs that contributed to this dimension's matrix (every judge scored them). */
144
+ itemIds: string[];
145
+ /** Judge IDs that contributed to this dimension's matrix. */
146
+ judgeIds: string[];
147
+ }
148
+ interface CorpusAgreementReport {
149
+ /** Per-dimension ICC(2,1) + κ_w + Pearson + Spearman + bootstrap CIs. */
150
+ perDimension: CorpusAgreementPerDimension[];
151
+ /** Mean ICC across dimensions (NaN if no dimension yielded a finite ICC). */
152
+ overallIcc: number;
153
+ /** Mean weighted κ across dimensions (NaN if none finite). */
154
+ overallWeightedKappa: number;
155
+ /** Dimensions evaluated (sorted). */
156
+ dimensions: string[];
157
+ /** Judges seen across the corpus (sorted). */
158
+ judgeIds: string[];
159
+ }
160
+ interface CorpusAgreementOptions extends ContinuousAgreementOptions {
161
+ /**
162
+ * Restrict the audit to these dimensions. Default = every dimension
163
+ * that appears in the input. A dimension named here but absent from
164
+ * the input throws — silent omission would corrupt the overall metric.
165
+ */
166
+ dimensions?: string[];
167
+ /**
168
+ * Restrict the audit to these judges. Default = every judge that
169
+ * appears in the input. A judge named here but absent from a
170
+ * dimension throws (see "fail loud" below).
171
+ */
172
+ judges?: string[];
173
+ }
174
+ /**
175
+ * Corpus-wide inter-rater agreement across N items × M judges × D dimensions.
176
+ *
177
+ * For each dimension, builds the [n_items][n_judges] matrix of scores
178
+ * (keeping only items every judge rated on that dimension), then runs
179
+ * `continuousAgreement` to get ICC(2,1), κ_w, Pearson, Spearman, and
180
+ * bootstrap CIs. Reports a pooled mean across dimensions as a single
181
+ * "is this judge panel reliable on this corpus?" number.
182
+ *
183
+ * Fail-loud contract:
184
+ * - Empty input throws.
185
+ * - Fewer than 2 judges or fewer than 2 items per dimension throws.
186
+ * - A judge present in some dimensions but with zero scored items on
187
+ * another dimension throws (would silently shrink the matrix).
188
+ * - Duplicate (itemId, judgeName, dimension) records throw.
189
+ */
190
+ declare function corpusInterRaterAgreement(records: CorpusScoreRecord[], opts?: CorpusAgreementOptions): CorpusAgreementReport;
191
+ /**
192
+ * Convenience adapter for `JudgeScore[]` data keyed externally by item.
193
+ *
194
+ * Use when you have per-item arrays of `JudgeScore[]` (e.g. one
195
+ * `ScenarioResult.judgeScores` per scenario) and want corpus-wide
196
+ * agreement without manually flattening. `itemId` must be unique per
197
+ * row of `itemsScores`.
198
+ */
199
+ declare function corpusInterRaterAgreementFromJudgeScores(itemsScores: Array<{
200
+ itemId: string;
201
+ scores: JudgeScore[];
202
+ }>, opts?: CorpusAgreementOptions): CorpusAgreementReport;
203
+ /**
204
+ * Required N per arm for a two-sample comparison at target effect size,
205
+ * alpha, and power. Normal-approximation formula:
206
+ * n = 2 * ( (z_{1-α/2} + z_{1-β}) / d )^2
207
+ * where d is Cohen's d. Returns Infinity for effect ≤ 0.
208
+ */
209
+ declare function requiredSampleSize(opts: {
210
+ effect: number;
211
+ alpha?: number;
212
+ power?: number;
213
+ twoSided?: boolean;
214
+ }): number;
215
+ /**
216
+ * Required number of paired observations for a target Cohen's dz.
217
+ * Unlike the independent-groups formula, this has no two-arm factor of two.
218
+ */
219
+ declare function requiredPairedSampleSize(opts: {
220
+ effect: number;
221
+ alpha?: number;
222
+ power?: number;
223
+ twoSided?: boolean;
224
+ }): number;
225
+ /**
226
+ * Minimum detectable paired effect (standardised units) for a target paired
227
+ * sample size: d_min = (z_{1-α/2} + z_β) / sqrt(n_paired). Multiply by
228
+ * sd(deltas) for score units; treat as a lower bound — Wilcoxon and bootstrap
229
+ * have asymptotic relative efficiency below 1 vs the t-test on heavy tails.
230
+ */
231
+ declare function pairedMde(opts: {
232
+ nPaired: number;
233
+ alpha?: number;
234
+ power?: number;
235
+ twoSided?: boolean;
236
+ }): number;
237
+ /**
238
+ * Number of paired observations needed for a McNemar test to reach a target
239
+ * power — the pre-registration companion to {@link mcnemar}. Parametrised by the
240
+ * expected discordant-cell probabilities `p10` (P[treatment wins on a pair]) and
241
+ * `p01` (P[control wins]); concordant pairs carry no information, so the count
242
+ * is driven entirely by the discordant rate. Lachin's (1992) asymptotic normal
243
+ * approximation: with discordant rate `pDisc = p10 + p01` and marginal effect
244
+ * `δ = p10 − p01`,
245
+ * n = ( z_{1-α/2}·√pDisc + z_{1-β}·√(pDisc − δ²) )² / δ².
246
+ * Returns Infinity when there is no effect (p10 === p01). Asymptotic — at the
247
+ * tiny discordant counts where the exact {@link mcnemar} differs from the normal
248
+ * approximation, treat the result as a lower bound and prefer the discordant-pair
249
+ * floor.
250
+ */
251
+ declare function mcnemarRequiredN(opts: {
252
+ p10: number;
253
+ p01: number;
254
+ alpha?: number;
255
+ power?: number;
256
+ twoSided?: boolean;
257
+ }): number;
258
+ /**
259
+ * Power of a McNemar test at a given number of paired observations, the inverse
260
+ * of {@link mcnemarRequiredN} (same Lachin asymptotic model, same parameters).
261
+ * Returns a value in [0, 1]; equals `alpha` when there is no effect.
262
+ */
263
+ declare function mcnemarPower(opts: {
264
+ p10: number;
265
+ p01: number;
266
+ nPairs: number;
267
+ alpha?: number;
268
+ twoSided?: boolean;
269
+ }): number;
270
+ /** Bonferroni adjustment: multiply every p-value by the test count, clamp at 1. */
271
+ declare function bonferroni(pValues: number[], alpha?: number): {
272
+ adjusted: number[];
273
+ significant: boolean[];
274
+ };
275
+ /**
276
+ * Holm step-down family-wise error adjustment.
277
+ *
278
+ * P-values are sorted from smallest to largest, multiplied by their remaining
279
+ * hypothesis count, and made monotonically non-decreasing before being mapped
280
+ * back to input order. This uniformly dominates plain Bonferroni while keeping
281
+ * strong family-wise error control under arbitrary dependence.
282
+ */
283
+ declare function holm(pValues: readonly number[], alpha?: number): {
284
+ adjusted: number[];
285
+ significant: boolean[];
286
+ };
287
+ /**
288
+ * Benjamini–Hochberg false discovery rate. Returns adjusted q-values and
289
+ * significance at the target FDR; handles ties and preserves q monotonicity.
290
+ */
291
+ declare function benjaminiHochberg(pValues: number[], fdr?: number): {
292
+ qValues: number[];
293
+ significant: boolean[];
294
+ };
295
+ interface PairedBootstrapResult {
296
+ /** Number of paired observations. */
297
+ n: number;
298
+ /** Median of paired deltas (after − before). */
299
+ median: number;
300
+ /** Mean of paired deltas. */
301
+ mean: number;
302
+ /** Lower bound of the bootstrap CI on the chosen statistic. */
303
+ low: number;
304
+ /** Upper bound of the bootstrap CI on the chosen statistic. */
305
+ high: number;
306
+ /** Confidence level used (e.g. 0.95). */
307
+ confidence: number;
308
+ /** Number of bootstrap resamples used. */
309
+ resamples: number;
310
+ }
311
+ interface PairedBootstrapOptions {
312
+ /** Confidence level. Default 0.95. */
313
+ confidence?: number;
314
+ /** Bootstrap resample count. Default 2000. */
315
+ resamples?: number;
316
+ /** Statistic to bootstrap. Default 'median'. */
317
+ statistic?: 'median' | 'mean';
318
+ /** Deterministic seed. If omitted, uses Math.random(). */
319
+ seed?: number;
320
+ }
321
+ /**
322
+ * Paired bootstrap on (after − before) deltas. Returns a CI on the chosen
323
+ * statistic (median by default); pairs are resampled with replacement. The
324
+ * lower bound is what the promotion gate checks — `low > threshold` means the
325
+ * gain is real at the confidence level. Throws on unequal sample sizes.
326
+ */
327
+ declare function pairedBootstrap(before: number[], after: number[], opts?: PairedBootstrapOptions): PairedBootstrapResult;
328
+ /** Pre-registered direction for a one-sided paired sign test. */
329
+ type SignTestAlternative = 'greater' | 'less';
330
+ /** Exact one-sided sign-test result for paired numeric differences. */
331
+ interface PairedSignTestResult {
332
+ /** Total supplied differences, including zero ties. */
333
+ n: number;
334
+ /** Strictly positive differences. */
335
+ positive: number;
336
+ /** Strictly negative differences. */
337
+ negative: number;
338
+ /** Zero differences excluded from the binomial test. */
339
+ ties: number;
340
+ /** Non-zero differences used by the binomial test. */
341
+ nNonTies: number;
342
+ /** Direction of the pre-registered alternative hypothesis. */
343
+ alternative: SignTestAlternative;
344
+ /** Exact one-sided p-value under P(positive) = P(negative) = 0.5. */
345
+ pValue: number;
346
+ }
347
+ /**
348
+ * Exact one-sided sign test over paired differences.
349
+ *
350
+ * Pass `after[i] - before[i]` for each matched item. `alternative = 'greater'`
351
+ * tests whether positive signs are more likely than negative signs and returns
352
+ * `P(Binomial(nNonTies, 0.5) >= positive)`. `alternative = 'less'` treats
353
+ * negative signs as successes instead. With a continuous difference
354
+ * distribution this is the usual directional median test. Exact zero
355
+ * differences are ties and do not enter the binomial denominator. All-tie and
356
+ * empty inputs return p = 1. Every input difference must be finite, and the
357
+ * direction must be chosen explicitly so a caller cannot select it after
358
+ * seeing the signs.
359
+ */
360
+ declare function pairedSignTest(differences: readonly number[], alternative: SignTestAlternative): PairedSignTestResult;
361
+ /** A binomial proportion estimate with a confidence interval. */
362
+ interface ProportionInterval {
363
+ /** Point estimate successes / n (0 when n = 0). */
364
+ estimate: number;
365
+ /** Lower bound, clamped to [0, 1]. */
366
+ lower: number;
367
+ /** Upper bound, clamped to [0, 1]. */
368
+ upper: number;
369
+ }
370
+ /**
371
+ * Wilson score interval for a binomial proportion. Correct at small n and near
372
+ * 0/1, where the normal (Wald) approximation produces bounds outside [0, 1] and
373
+ * understates coverage. Use this for any pass-rate / hit-rate / realness-rate
374
+ * CI — the continuous `confidenceInterval` assumes the wrong distribution for a
375
+ * proportion. `n = 0 ⇒ {0, 0, 0}`.
376
+ */
377
+ declare function wilson(successes: number, n: number, confidence?: number): ProportionInterval;
378
+ /** Result of a McNemar paired-binary significance test. */
379
+ interface McNemarResult {
380
+ /** Total paired observations. */
381
+ n: number;
382
+ /** Discordant pairs (b + c) — the only ones that carry signal. */
383
+ nDiscordant: number;
384
+ /** Pairs where treatment succeeded and control failed ("newly correct"). */
385
+ b: number;
386
+ /** Pairs where control succeeded and treatment failed ("newly wrong"). */
387
+ c: number;
388
+ /** Continuity-corrected chi-square statistic (reference; exact p drives the call). */
389
+ statistic: number;
390
+ /** Two-sided p-value. Exact (binomial sign test on discordant pairs). */
391
+ pValue: number;
392
+ }
393
+ /**
394
+ * McNemar's test for paired binary outcomes — the correct significance test for
395
+ * "does treatment change the success rate vs control on the SAME items". Only
396
+ * discordant pairs (one arm right, the other wrong) carry information; concordant
397
+ * pairs are uninformative, so a paired t-test / two-proportion z-test on the raw
398
+ * rates is wrong here. The p-value is exact: under H0 the b "treatment-wins" are
399
+ * Binomial(b + c, 0.5), so the two-sided p is the doubled binomial tail — correct
400
+ * at the small discordant counts typical of eval runs (no continuity-corrected
401
+ * chi-square approximation needed, though it is returned as `statistic` for
402
+ * reference). Inputs are paired 0/1 (or boolean) arrays, control first to match
403
+ * the module's (before, after) convention. Throws on unequal lengths.
404
+ */
405
+ declare function mcnemar(control: ArrayLike<number | boolean>, treatment: ArrayLike<number | boolean>): McNemarResult;
406
+ /** A paired binary effect size (treatment rate − control rate) with a CI. */
407
+ interface RiskDifferenceResult {
408
+ /** Total paired observations. */
409
+ n: number;
410
+ /** Discordant pairs: treatment-win count. */
411
+ b: number;
412
+ /** Discordant pairs: control-win count. */
413
+ c: number;
414
+ /** Paired risk difference p(treatment) − p(control) = (b − c) / n. */
415
+ riskDifference: number;
416
+ /** Lower bound of the CI, clamped to [-1, 1]. */
417
+ lower: number;
418
+ /** Upper bound of the CI, clamped to [-1, 1]. */
419
+ upper: number;
420
+ /** Confidence level used. */
421
+ confidence: number;
422
+ }
423
+ /**
424
+ * Paired risk difference (the effect-size companion to {@link mcnemar}): the
425
+ * change in success rate p(treatment) − p(control) on matched items, which for
426
+ * paired binary data equals (b − c) / n. The CI uses the paired variance from
427
+ * the discordant counts, not the independent-samples formula (which overstates
428
+ * the interval by ignoring the pairing). Inputs are paired 0/1 (or boolean)
429
+ * arrays, control first. Throws on unequal lengths.
430
+ */
431
+ declare function pairedRiskDifference(control: ArrayLike<number | boolean>, treatment: ArrayLike<number | boolean>, confidence?: number): RiskDifferenceResult;
432
+ /**
433
+ * Unbiased pass@k for code generation (Chen et al. 2021, "Evaluating Large
434
+ * Language Models Trained on Code"). Given `n` independent samples for one
435
+ * problem of which `c` pass, the probability that at least one of a random k of
436
+ * them passes is 1 − C(n−c, k) / C(n, k). Estimating pass@k as "did any of the
437
+ * first k pass" is biased high at small n; this is the variance-reduced estimator
438
+ * averaged implicitly over all k-subsets. Average the per-problem values across
439
+ * the suite for the corpus pass@k. Computed in the numerically stable product
440
+ * form. Requires 1 ≤ k ≤ n and 0 ≤ c ≤ n.
441
+ */
442
+ declare function passAtK(n: number, c: number, k: number): number;
443
+ interface EProcessOptions {
444
+ /** Type-I error budget. The process decides when wealth ≥ 1/alpha
445
+ * (Ville's inequality). Default 0.05. */
446
+ alpha?: number;
447
+ /** Truncation bound on the predictable bet λ ∈ [0, maxBet]. Must satisfy
448
+ * maxBet < 1/nullMean so every wealth factor stays strictly positive.
449
+ * Default 0.5. */
450
+ maxBet?: number;
451
+ /** The null boundary m₀ for H0: E[x] ≤ m₀ on x ∈ [0,1]. Default 0.5
452
+ * (the paired-delta encoding x = (d+1)/2 maps "no effect" to 1/2).
453
+ * A pre-registered minEffect shifts this — see `sequentialPairedGate`. */
454
+ nullMean?: number;
455
+ }
456
+ interface EProcessStep {
457
+ /** Current wealth W_n — the e-value against H0 after n observations. */
458
+ wealth: number;
459
+ /** Observations consumed so far. */
460
+ n: number;
461
+ /** True from the first n where W_n ≥ 1/alpha onward (sticky). */
462
+ decided: boolean;
463
+ }
464
+ interface EProcessState extends EProcessStep {
465
+ alpha: number;
466
+ maxBet: number;
467
+ nullMean: number;
468
+ /** The decision boundary 1/alpha. */
469
+ threshold: number;
470
+ /** Observation count at the first threshold crossing; undefined until decided. */
471
+ decidedAtN?: number;
472
+ }
473
+ interface EProcess {
474
+ /** Consume one observation x ∈ [0,1]. Throws on non-finite / out-of-range
475
+ * input — a silent clamp would corrupt the type-I guarantee. */
476
+ update(x: number): EProcessStep;
477
+ state(): EProcessState;
478
+ }
479
+ /**
480
+ * Betting test-martingale for bounded observations — the e-process core of
481
+ * anytime-valid sequential testing (Waudby-Smith & Ramdas, "Estimating means
482
+ * of bounded random variables by betting", JRSS-B 2024).
483
+ *
484
+ * Observations x_i ∈ [0,1]; H0: E[x] ≤ m₀ (`nullMean`, default 1/2). Wealth
485
+ *
486
+ * W_t = Π_{i≤t} (1 + λ_i (x_i − m₀)), W_0 = 1
487
+ *
488
+ * with the truncated GROW-style plug-in bet computed from PRIOR observations:
489
+ *
490
+ * λ_i = clamp((μ̂_{i−1} − m₀) / (σ̂²_{i−1} + (μ̂_{i−1} − m₀)²), 0, maxBet)
491
+ *
492
+ * where μ̂/σ̂² are the shrunk running estimates μ̂_t = (1/2 + Σx_i)/(t+1),
493
+ * σ̂²_t = (1/4 + Σ(x_i − μ̂_i)²)/(t+1).
494
+ *
495
+ * PREDICTABILITY INVARIANT (load-bearing): λ_i is a function of x_1..x_{i−1}
496
+ * ONLY — it may never see x_i. With λ_i ≥ 0 predictable, each factor has
497
+ * E[1 + λ_i(x_i − m₀) | past] ≤ 1 under H0, so W is a nonnegative
498
+ * supermartingale and Ville's inequality gives P(∃t: W_t ≥ 1/α) ≤ α — the
499
+ * type-I guarantee holds at ANY data-dependent stopping time. λ_1 is always 0
500
+ * (no prior evidence), so the first observation never moves wealth.
501
+ *
502
+ * `decided` latches at the first crossing W_t ≥ 1/α and never un-latches;
503
+ * wealth keeps updating after the crossing (the e-process remains valid), but
504
+ * the decision time is the first crossing.
505
+ */
506
+ declare function eProcess(opts?: EProcessOptions): EProcess;
507
+ /** Tiny seedable PRNG (mulberry32) — deterministic resampling/shuffling, not
508
+ * cryptographic. Exported so e-process shuffles and bootstrap resampling
509
+ * share ONE PRNG implementation; a seed is REQUIRED (unseeded randomness in
510
+ * gate verdicts is non-reproducible by construction). */
511
+ declare function mulberry32(seed: number): () => number;
512
+ //#endregion
513
+ export { mannWhitneyU as A, pairedSignTest as B, confidenceInterval as C, holm as D, eProcess as E, normalizeScores as F, ranks as G, partialCredit as H, pairedBootstrap as I, spearmanR as J, requiredPairedSampleSize as K, pairedCohensDz as L, mcnemarPower as M, mcnemarRequiredN as N, interRaterReliability as O, mulberry32 as P, wilson as Q, pairedMde as R, cohensD as S, corpusInterRaterAgreementFromJudgeScores as T, passAtK as U, pairedTTest as V, pearsonR as W, weightedMean as X, weightedComposite as Y, wilcoxonSignedRank as Z, WeightedCompositeInput as _, CorpusScoreRecord as a, bonferroni as b, EProcessState as c, PairedBootstrapOptions as d, PairedBootstrapResult as f, SignTestAlternative as g, RiskDifferenceResult as h, CorpusAgreementReport as i, mcnemar as j, interpretCliffs as k, EProcessStep as l, ProportionInterval as m, CorpusAgreementOptions as n, EProcess as o, PairedSignTestResult as p, requiredSampleSize as q, CorpusAgreementPerDimension as r, EProcessOptions as s, CliffsMagnitude as t, McNemarResult as u, WeightedCompositeResult as v, corpusInterRaterAgreement as w, cliffsDelta as x, benjaminiHochberg as y, pairedRiskDifference as z };
514
+ //# sourceMappingURL=statistics-Cmj6nynr.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"statistics-Cmj6nynr.d.ts","names":[],"sources":["../src/statistics.ts"],"mappings":";;;;;cAUa,kBAAe,QAAY,iBAAe;;iBAGvC,aAAa;EAAU;EAAe;;;iBAatC,mBACd,kBACA,qBACA;EAAQ;EAAe;;EACpB;EAAc;EAAe;;;;;;;;iBAsClB,sBAAsB,aAAa;;;;;iBAwDnC,aAAa,aAAa;EAAgB;EAAW;;;iBA+CrD,cAAc,iBAAiB;;;;;;;iBAW/B,YACd,kBACA;EACG;EAAW;EAAY;;;;;;iBAyBZ,mBAAmB,kBAAkB;EAAoB;EAAW;;;;;;;iBAqCpE,QAAQ,aAAa;;;;;;;;;iBAqBrB,eAAe,kBAAkB;KAoBrC;;;;;;;;;;;iBAYI,YAAY,kBAAkB;;;;;;iBAkB9B,gBAAgB,gBAAgB;;;;;iBAuBhC,MAAM;;;;;;iBAmBN,SAAS,aAAa;;;;;iBAuBtB,UAAU,aAAa;UAKtB;;EAEf,MAAM;;;;;EAKN,SAAS;;EAET;;UAGe;EACf;EACA;;;;;;;;;;;iBAYc,kBAAkB,OAAO,yBAAyB;UA4CjD;;EAEf;;EAEA;;EAEA;;EAEA;;UAGe,oCAAoC;EACnD;;EAEA;;EAEA;;UAGe;;EAEf,cAAc;;EAEd;;EAEA;;EAEA;;EAEA;;UAGe,+BAA+B;;;;;;EAM9C;;;;;;EAMA;;;;;;;;;;;;;;;;;;iBAmBc,0BACd,SAAS,qBACT,OAAM,yBACL;;;;;;;;;iBA0Ha,yCACd,aAAa;EAAQ;EAAgB,QAAQ;IAC7C,OAAM,yBACL;;;;;;;iBA+Ga,mBAAmB;EACjC;EACA;EACA;EACA;;;;;;iBAiBc,yBAAyB;EACvC;EACA;EACA;EACA;;;;;;;;iBAkBc,UAAU;EACxB;EACA;EACA;EACA;;;;;;;;;;;;;;;;iBAyBc,iBAAiB;EAC/B;EACA;EACA;EACA;EACA;;;;;;;iBAyBc,aAAa;EAC3B;EACA;EACA;EACA;EACA;;;iBAkBc,WACd,mBACA;EACG;EAAoB;;;;;;;;;;iBAeT,KACd,4BACA;EACG;EAAoB;;;;;;iBA+BT,kBACd,mBACA;EACG;EAAmB;;UAoBP;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;UAGe;;EAEf;;EAEA;;EAEA;;EAEA;;;;;;;;iBASc,gBACd,kBACA,iBACA,OAAM,yBACL;;KAwDS;;UAGK;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA,aAAa;;EAEb;;;;;;;;;;;;;;;iBAgBc,eACd,gCACA,aAAa,sBACZ;;UAgDc;;EAEf;;EAEA;;EAEA;;;;;;;;;iBAUc,OAAO,mBAAmB,WAAW,sBAAoB;;UAmBxD;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;;;;;;;iBAec,QACd,SAAS,6BACT,WAAW,8BACV;;UAmBc;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;;;iBAWc,qBACd,SAAS,6BACT,WAAW,6BACX,sBACC;;;;;;;;;;;iBAyCa,QAAQ,WAAW,WAAW;UAgE7B;;;EAGf;;;;EAIA;;;;EAIA;;UAGe;;EAEf;;EAEA;;EAEA;;UAGe,sBAAsB;EACrC;EACA;EACA;;EAEA;;EAEA;;UAGe;;;EAGf,OAAO,YAAY;EACnB,SAAS;;;;;;;;;;;;;;;;;;;;;;;;;;;;;iBA8BK,SAAS,OAAM,kBAAuB;;;;;iBAsHtC,WAAW"}