@tangle-network/agent-eval 0.144.11 → 0.144.13

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (378) hide show
  1. package/CHANGELOG.md +11 -0
  2. package/dist/agent-profile-B9_GGsG8.d.ts +84 -0
  3. package/dist/agent-profile-B9_GGsG8.d.ts.map +1 -0
  4. package/dist/{agent-profile-cell-BOP-iA9Q.d.ts → agent-profile-cell-BkcRDikH.d.ts} +2 -2
  5. package/dist/{agent-profile-cell-BOP-iA9Q.d.ts.map → agent-profile-cell-BkcRDikH.d.ts.map} +1 -1
  6. package/dist/analyst/index.d.ts +134 -16
  7. package/dist/analyst/index.d.ts.map +1 -1
  8. package/dist/analyst/index.js +364 -10
  9. package/dist/analyst/index.js.map +1 -1
  10. package/dist/backend-integrity-CsVin_Wb.d.ts +280 -0
  11. package/dist/backend-integrity-CsVin_Wb.d.ts.map +1 -0
  12. package/dist/{baseline-C-GocmIW.js → baseline-BhPRQBVn.js} +3 -2
  13. package/dist/{baseline-C-GocmIW.js.map → baseline-BhPRQBVn.js.map} +1 -1
  14. package/dist/{benchmark-CWeqGl7x.js → benchmark-BhT16ep9.js} +2 -2
  15. package/dist/{benchmark-CWeqGl7x.js.map → benchmark-BhT16ep9.js.map} +1 -1
  16. package/dist/{agent-profile-DPi7IZg7.d.ts → benchmark-Ceoan7vk.d.ts} +4 -91
  17. package/dist/benchmark-Ceoan7vk.d.ts.map +1 -0
  18. package/dist/{benchmark-command-BteMFN62.js → benchmark-command-CY6cbhOB.js} +36 -79
  19. package/dist/benchmark-command-CY6cbhOB.js.map +1 -0
  20. package/dist/benchmarks/index.d.ts +244 -2
  21. package/dist/benchmarks/index.d.ts.map +1 -0
  22. package/dist/benchmarks/index.js +733 -1
  23. package/dist/benchmarks/index.js.map +1 -0
  24. package/dist/builder-eval/index.d.ts +23 -2
  25. package/dist/builder-eval/index.d.ts.map +1 -1
  26. package/dist/builder-eval/index.js +227 -3
  27. package/dist/builder-eval/index.js.map +1 -1
  28. package/dist/campaign/index.d.ts +10 -8
  29. package/dist/campaign/index.js +9 -6
  30. package/dist/{campaign-C2TTzQII.js → campaign-BYjBAypg.js} +21 -634
  31. package/dist/campaign-BYjBAypg.js.map +1 -0
  32. package/dist/{canonical-D011XM8r.js → canonical-D-XsTQ6_.js} +2 -2
  33. package/dist/{canonical-D011XM8r.js.map → canonical-D-XsTQ6_.js.map} +1 -1
  34. package/dist/{index-CQTZ-4XN.d.ts → capture-fetch-DDvpjVRU.d.ts} +3 -3
  35. package/dist/capture-fetch-DDvpjVRU.d.ts.map +1 -0
  36. package/dist/{default-registry-BmktKy8r.js → chat-client-Bp4Ebuuc.js} +1276 -1276
  37. package/dist/chat-client-Bp4Ebuuc.js.map +1 -0
  38. package/dist/cli.js +2 -2
  39. package/dist/{client-C9gzZE59.d.ts → client-KUbGulm_.d.ts} +4 -4
  40. package/dist/{client-C9gzZE59.d.ts.map → client-KUbGulm_.d.ts.map} +1 -1
  41. package/dist/contract/index.d.ts +13 -390
  42. package/dist/contract/index.d.ts.map +1 -1
  43. package/dist/contract/index.js +18 -542
  44. package/dist/contract/index.js.map +1 -1
  45. package/dist/{cost-ledger-DMFxsLKr.js → cost-ledger-BSe92yAV.js} +3 -3
  46. package/dist/{cost-ledger-DMFxsLKr.js.map → cost-ledger-BSe92yAV.js.map} +1 -1
  47. package/dist/{cost-ledger-Bv_e8XHY.d.ts → cost-ledger-DbQdN3nO.d.ts} +2 -2
  48. package/dist/{cost-ledger-Bv_e8XHY.d.ts.map → cost-ledger-DbQdN3nO.d.ts.map} +1 -1
  49. package/dist/{counterfactual-CxmxAONP.d.ts → counterfactual--bpysZF0.d.ts} +2 -14
  50. package/dist/{counterfactual-CxmxAONP.d.ts.map → counterfactual--bpysZF0.d.ts.map} +1 -1
  51. package/dist/{counterfactual-CWPTrMH7.js → counterfactual-lDfCx0Uz.js} +3 -31
  52. package/dist/{counterfactual-CWPTrMH7.js.map → counterfactual-lDfCx0Uz.js.map} +1 -1
  53. package/dist/{dataset-C8xaLXdY.d.ts → dataset-CJjKqQfA.d.ts} +2 -8
  54. package/dist/dataset-CJjKqQfA.d.ts.map +1 -0
  55. package/dist/{default-registry-Bf8Woqmq.d.ts → default-registry-Di6HP6pG.d.ts} +7 -12
  56. package/dist/default-registry-Di6HP6pG.d.ts.map +1 -0
  57. package/dist/{analyze-runs-C30yljDJ.js → define-agent-eval-iqjT--ZZ.js} +542 -130
  58. package/dist/define-agent-eval-iqjT--ZZ.js.map +1 -0
  59. package/dist/define-agent-eval-sH24zBfM.d.ts +388 -0
  60. package/dist/define-agent-eval-sH24zBfM.d.ts.map +1 -0
  61. package/dist/descriptive-B5MwKfbf.js +144 -0
  62. package/dist/descriptive-B5MwKfbf.js.map +1 -0
  63. package/dist/{dspy-rlm-engine-DbTk4JdR.js → dspy-rlm-engine-B_qhSc21.js} +4 -4
  64. package/dist/{dspy-rlm-engine-DbTk4JdR.js.map → dspy-rlm-engine-B_qhSc21.js.map} +1 -1
  65. package/dist/effect-sizes-DiH8MGOH.js +82 -0
  66. package/dist/effect-sizes-DiH8MGOH.js.map +1 -0
  67. package/dist/{engine-3hL-XqwJ.d.ts → engine-otFpE2gF.d.ts} +10 -38
  68. package/dist/engine-otFpE2gF.d.ts.map +1 -0
  69. package/dist/{errors-CKPfb2aH.d.ts → errors-DEE6u6ot.d.ts} +2 -14
  70. package/dist/{errors-CKPfb2aH.d.ts.map → errors-DEE6u6ot.d.ts.map} +1 -1
  71. package/dist/{errors-D-LKuDhb.js → errors-Dngq5h35.js} +2 -8
  72. package/dist/{errors-D-LKuDhb.js.map → errors-Dngq5h35.js.map} +1 -1
  73. package/dist/{eval-campaign-B_7wcnav.js → eval-campaign-UB-usSQ2.js} +6 -6
  74. package/dist/{eval-campaign-B_7wcnav.js.map → eval-campaign-UB-usSQ2.js.map} +1 -1
  75. package/dist/{exact-types-DSFFpLLI.d.ts → exact-types-BH1twmAJ.d.ts} +2 -2
  76. package/dist/{exact-types-DSFFpLLI.d.ts.map → exact-types-BH1twmAJ.d.ts.map} +1 -1
  77. package/dist/experiment/index.d.ts +9 -6
  78. package/dist/experiment/index.d.ts.map +1 -1
  79. package/dist/experiment/index.js +11 -7
  80. package/dist/experiment/index.js.map +1 -1
  81. package/dist/experiment-tracker-C29gXM4B.js +269 -0
  82. package/dist/experiment-tracker-C29gXM4B.js.map +1 -0
  83. package/dist/{experiment-tracker-IMntXr6J.d.ts → experiment-tracker-DWHZBAYL.d.ts} +77 -121
  84. package/dist/experiment-tracker-DWHZBAYL.d.ts.map +1 -0
  85. package/dist/{external-optimizer-contracts-CZuJNcT5.d.ts → external-optimizer-contracts-lixrOZdX.d.ts} +3 -3
  86. package/dist/{external-optimizer-contracts-CZuJNcT5.d.ts.map → external-optimizer-contracts-lixrOZdX.d.ts.map} +1 -1
  87. package/dist/external-optimizer-process-BTiNB-RH.js +301 -0
  88. package/dist/external-optimizer-process-BTiNB-RH.js.map +1 -0
  89. package/dist/{single-run-lock-DFWHEB09.js → external-optimizer-subprocess-DrJ9hR8u.js} +144 -438
  90. package/dist/external-optimizer-subprocess-DrJ9hR8u.js.map +1 -0
  91. package/dist/extract-usage-BrQ8mCLX.js +155 -0
  92. package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
  93. package/dist/{failure-cluster-CqcvCcdR.d.ts → failure-cluster-BLURuWG4.d.ts} +2 -3
  94. package/dist/failure-cluster-BLURuWG4.d.ts.map +1 -0
  95. package/dist/{feedback-trajectory-CacHpxsp.d.ts → feedback-trajectory-juOozjAc.d.ts} +5 -36
  96. package/dist/feedback-trajectory-juOozjAc.d.ts.map +1 -0
  97. package/dist/fuzz.d.ts +2 -2
  98. package/dist/fuzz.js +2 -2
  99. package/dist/hosted/index.d.ts +3 -3
  100. package/dist/{index-CvSN3IG1.d.ts → index-B8Ui1mr1.d.ts} +2 -2
  101. package/dist/{index-CvSN3IG1.d.ts.map → index-B8Ui1mr1.d.ts.map} +1 -1
  102. package/dist/{skill-usage-CO9OLRBx.d.ts → index-BWDrSVfw.d.ts} +11 -136
  103. package/dist/index-BWDrSVfw.d.ts.map +1 -0
  104. package/dist/index-Ba3YrbAL.d.ts +1 -0
  105. package/dist/{index-BnEuDAK2.d.ts → index-COQYtuRF.d.ts} +3 -3
  106. package/dist/{index-BnEuDAK2.d.ts.map → index-COQYtuRF.d.ts.map} +1 -1
  107. package/dist/{index-C3ssXVLv.d.ts → index-CvjYbU0D.d.ts} +2 -2
  108. package/dist/{index-C3ssXVLv.d.ts.map → index-CvjYbU0D.d.ts.map} +1 -1
  109. package/dist/{index-DPPGNJ_R.d.ts → index-DSmEylT9.d.ts} +17 -115
  110. package/dist/index-DSmEylT9.d.ts.map +1 -0
  111. package/dist/index.d.ts +2397 -5308
  112. package/dist/index.d.ts.map +1 -1
  113. package/dist/index.js +5914 -10496
  114. package/dist/index.js.map +1 -1
  115. package/dist/{insight-report-CgX_s0Ez.d.ts → insight-report-CRi-Ufrj.d.ts} +4 -4
  116. package/dist/{insight-report-CgX_s0Ez.d.ts.map → insight-report-CRi-Ufrj.d.ts.map} +1 -1
  117. package/dist/{integrity-MLzHOfV9.js → integrity-Cy9WHAtb.js} +2 -2
  118. package/dist/{integrity-MLzHOfV9.js.map → integrity-Cy9WHAtb.js.map} +1 -1
  119. package/dist/{integrity-DY6tIbl0.js → integrity-DysDBWDu.js} +2 -2
  120. package/dist/{integrity-DY6tIbl0.js.map → integrity-DysDBWDu.js.map} +1 -1
  121. package/dist/{integrity-BuqEKu-x.d.ts → integrity-OrcI9Nau.d.ts} +3 -3
  122. package/dist/{integrity-BuqEKu-x.d.ts.map → integrity-OrcI9Nau.d.ts.map} +1 -1
  123. package/dist/internal-BDHPCnjk.js +230 -0
  124. package/dist/internal-BDHPCnjk.js.map +1 -0
  125. package/dist/judge-calibration-C5CbMYce.d.ts +117 -0
  126. package/dist/judge-calibration-C5CbMYce.d.ts.map +1 -0
  127. package/dist/judge-calibration-DZkWrm5H.js +317 -0
  128. package/dist/judge-calibration-DZkWrm5H.js.map +1 -0
  129. package/dist/{kind-factory-B8-r8-y8.js → kind-factory-CPmSd58s.js} +208 -208
  130. package/dist/{kind-factory-B8-r8-y8.js.map → kind-factory-CPmSd58s.js.map} +1 -1
  131. package/dist/ledger-core/index.d.ts +1 -1
  132. package/dist/ledger-core/index.js +2 -2
  133. package/dist/{ledger-core-DXZIqu17.js → ledger-core-BmZt19oQ.js} +110 -110
  134. package/dist/{ledger-core-DXZIqu17.js.map → ledger-core-BmZt19oQ.js.map} +1 -1
  135. package/dist/{llm-client-DzvMUsS_.js → llm-client-d0-2TT1g.js} +4 -64
  136. package/dist/{llm-client-DzvMUsS_.js.map → llm-client-d0-2TT1g.js.map} +1 -1
  137. package/dist/{skillopt-optimization-method-CQdVeM8k.js → llm-judge-dZ8P6nGI.js} +3349 -5401
  138. package/dist/llm-judge-dZ8P6nGI.js.map +1 -0
  139. package/dist/matrix/index.d.ts +2 -2
  140. package/dist/meta-eval/index.d.ts +3 -3
  141. package/dist/meta-eval/index.js +3 -3
  142. package/dist/{metrics-C9YY1OcL.js → metrics-Cl0L1KUy.js} +2 -108
  143. package/dist/{metrics-C9YY1OcL.js.map → metrics-Cl0L1KUy.js.map} +1 -1
  144. package/dist/{mint-B2O60ACG.js → mint-Dj9Ww_3I.js} +4 -4
  145. package/dist/{mint-B2O60ACG.js.map → mint-Dj9Ww_3I.js.map} +1 -1
  146. package/dist/{multi-layer-verifier-DnAqwl0h.d.ts → multi-layer-verifier-DIguZc8Z.d.ts} +3 -3
  147. package/dist/{multi-layer-verifier-DnAqwl0h.d.ts.map → multi-layer-verifier-DIguZc8Z.d.ts.map} +1 -1
  148. package/dist/multiplicity-DIWHvysC.d.ts +43 -0
  149. package/dist/multiplicity-DIWHvysC.d.ts.map +1 -0
  150. package/dist/multishot/index.d.ts +3 -3
  151. package/dist/multishot/index.js +1 -1
  152. package/dist/openapi.json +1 -1
  153. package/dist/{opencode-sqlite-8r6WUfHc.js → opencode-sqlite-DJWAXLms.js} +2 -2
  154. package/dist/{opencode-sqlite-8r6WUfHc.js.map → opencode-sqlite-DJWAXLms.js.map} +1 -1
  155. package/dist/package-version-D7lQHt_-.js +34 -0
  156. package/dist/package-version-D7lQHt_-.js.map +1 -0
  157. package/dist/paired-arms-D-XRF_fy.js +1045 -0
  158. package/dist/paired-arms-D-XRF_fy.js.map +1 -0
  159. package/dist/paired-promotion-decision-CGzg0cI_.d.ts +251 -0
  160. package/dist/paired-promotion-decision-CGzg0cI_.d.ts.map +1 -0
  161. package/dist/paired-tests-BHIhYVdu.js +213 -0
  162. package/dist/paired-tests-BHIhYVdu.js.map +1 -0
  163. package/dist/pareto-BqNW3LJR.d.ts +117 -0
  164. package/dist/pareto-BqNW3LJR.d.ts.map +1 -0
  165. package/dist/pipelines/index.d.ts +3 -64
  166. package/dist/pipelines/index.d.ts.map +1 -1
  167. package/dist/pipelines/index.js +4 -284
  168. package/dist/pipelines/index.js.map +1 -1
  169. package/dist/power-and-mde-CHIrXJll.js +195 -0
  170. package/dist/power-and-mde-CHIrXJll.js.map +1 -0
  171. package/dist/{promotion-policy-CrLrmys8.js → power-preflight-DEw-uC7q.js} +4 -184
  172. package/dist/power-preflight-DEw-uC7q.js.map +1 -0
  173. package/dist/pre-registration-CZwSFQS4.d.ts +577 -0
  174. package/dist/pre-registration-CZwSFQS4.d.ts.map +1 -0
  175. package/dist/{prime-protocol-BfSalTfR.js → prime-protocol-6tZTVsWm.js} +72 -21
  176. package/dist/prime-protocol-6tZTVsWm.js.map +1 -0
  177. package/dist/produced-state-DU79a81m.js +586 -0
  178. package/dist/produced-state-DU79a81m.js.map +1 -0
  179. package/dist/profile-cell.d.ts +1 -1
  180. package/dist/profile-cell.js +1 -1
  181. package/dist/promotion-policy-D0nPhkSy.d.ts +134 -0
  182. package/dist/promotion-policy-D0nPhkSy.d.ts.map +1 -0
  183. package/dist/promotion-policy-xzA40Evo.js +186 -0
  184. package/dist/promotion-policy-xzA40Evo.js.map +1 -0
  185. package/dist/{skillopt-optimization-method-USDKhxSA.d.ts → provenance-CCdxgLDT.d.ts} +57 -478
  186. package/dist/provenance-CCdxgLDT.d.ts.map +1 -0
  187. package/dist/registry-oJeeI4-a.d.ts +178 -0
  188. package/dist/registry-oJeeI4-a.d.ts.map +1 -0
  189. package/dist/{release-report-DA2BCu5a.d.ts → release-confidence-BFRE5WSp.d.ts} +180 -112
  190. package/dist/release-confidence-BFRE5WSp.d.ts.map +1 -0
  191. package/dist/{release-report-BUYmoKo2.js → release-confidence-CxDuiAev.js} +125 -217
  192. package/dist/release-confidence-CxDuiAev.js.map +1 -0
  193. package/dist/reporting.d.ts +6 -5
  194. package/dist/reporting.js +7 -5
  195. package/dist/{researcher-BchpD55R.d.ts → researcher-DaN4GST-.d.ts} +7 -40
  196. package/dist/{researcher-BchpD55R.d.ts.map → researcher-DaN4GST-.d.ts.map} +1 -1
  197. package/dist/{reward-hacking-BDToousL.js → reward-hacking-DNgjilrV.js} +3 -3
  198. package/dist/reward-hacking-DNgjilrV.js.map +1 -0
  199. package/dist/{reward-hacking-BI0OMAlo.d.ts → reward-hacking-DSSTuI9r.d.ts} +5 -5
  200. package/dist/{reward-hacking-BI0OMAlo.d.ts.map → reward-hacking-DSSTuI9r.d.ts.map} +1 -1
  201. package/dist/rl.d.ts +7 -7
  202. package/dist/rl.js +11 -10
  203. package/dist/rl.js.map +1 -1
  204. package/dist/rollout/index.d.ts +2 -2
  205. package/dist/rollout/index.js +4 -4
  206. package/dist/{rollout-DRrksrcV.js → rollout-BWtw0I_6.js} +3 -3
  207. package/dist/{rollout-DRrksrcV.js.map → rollout-BWtw0I_6.js.map} +1 -1
  208. package/dist/{rubric-predictive-validity-DvkjPYCe.d.ts → rubric-predictive-validity-BgxtKe4G.d.ts} +2 -2
  209. package/dist/{rubric-predictive-validity-DvkjPYCe.d.ts.map → rubric-predictive-validity-BgxtKe4G.d.ts.map} +1 -1
  210. package/dist/{rubric-predictive-validity-BRR632r1.js → rubric-predictive-validity-Cwwyd7ah.js} +2 -2
  211. package/dist/{rubric-predictive-validity-BRR632r1.js.map → rubric-predictive-validity-Cwwyd7ah.js.map} +1 -1
  212. package/dist/{run-record-BmSPWXJR.js → run-record-BvHPVS-i.js} +3 -3
  213. package/dist/{run-record-BmSPWXJR.js.map → run-record-BvHPVS-i.js.map} +1 -1
  214. package/dist/{run-record-CF4Dwpxr.d.ts → run-record-CKiihE6f.d.ts} +4 -4
  215. package/dist/{run-record-CF4Dwpxr.d.ts.map → run-record-CKiihE6f.d.ts.map} +1 -1
  216. package/dist/{proposal-findings-2GIUo1et.js → run-score-lDzV0X8j.js} +31 -31
  217. package/dist/run-score-lDzV0X8j.js.map +1 -0
  218. package/dist/sandbox-harness-BlSOu4LX.d.ts +70 -0
  219. package/dist/sandbox-harness-BlSOu4LX.d.ts.map +1 -0
  220. package/dist/{schema-Cef2cFmb2.d.ts → schema-Cef2cFmb.d.ts} +1 -1
  221. package/dist/schema-Cef2cFmb.d.ts.map +1 -0
  222. package/dist/semantic-concept-judge-BiJxScqe.js +406 -0
  223. package/dist/semantic-concept-judge-BiJxScqe.js.map +1 -0
  224. package/dist/{sequential-D-BLJBKU.js → sequential-C458DXNf.js} +4 -3
  225. package/dist/{sequential-D-BLJBKU.js.map → sequential-C458DXNf.js.map} +1 -1
  226. package/dist/sequential-eprocess-CbUt2htw.js +83 -0
  227. package/dist/sequential-eprocess-CbUt2htw.js.map +1 -0
  228. package/dist/series-convergence-D1cL1f-4.d.ts +129 -0
  229. package/dist/series-convergence-D1cL1f-4.d.ts.map +1 -0
  230. package/dist/{server-iu0ede49.js → server-ulsOdrTI.js} +5 -21
  231. package/dist/server-ulsOdrTI.js.map +1 -0
  232. package/dist/skillopt-optimization-method-C5cotF4E.d.ts +427 -0
  233. package/dist/skillopt-optimization-method-C5cotF4E.d.ts.map +1 -0
  234. package/dist/skillopt-optimization-method-jjdnc3YK.js +1999 -0
  235. package/dist/skillopt-optimization-method-jjdnc3YK.js.map +1 -0
  236. package/dist/{statistical-heldout-TQ-4CYiN.d.ts → statistical-heldout-DoFF5KJX.d.ts} +5 -278
  237. package/dist/statistical-heldout-DoFF5KJX.d.ts.map +1 -0
  238. package/dist/{store-otlp-Dw8PPIlL.js → store-otlp-CsptLYpN.js} +3 -3
  239. package/dist/{store-otlp-Dw8PPIlL.js.map → store-otlp-CsptLYpN.js.map} +1 -1
  240. package/dist/store-tool-spans-Cq9mFd-q.js +667 -0
  241. package/dist/store-tool-spans-Cq9mFd-q.js.map +1 -0
  242. package/dist/store-tool-spans-DbDLOBOb.d.ts +342 -0
  243. package/dist/store-tool-spans-DbDLOBOb.d.ts.map +1 -0
  244. package/dist/student-t-CvBq2mve.js +38 -0
  245. package/dist/student-t-CvBq2mve.js.map +1 -0
  246. package/dist/{summary-report-Lf-5I7xh.js → summary-report-Blysd6Z2.js} +6 -3
  247. package/dist/{summary-report-Lf-5I7xh.js.map → summary-report-Blysd6Z2.js.map} +1 -1
  248. package/dist/{summary-report-BNR7DWTj.d.ts → summary-report-CaL-Hnxt.d.ts} +5 -5
  249. package/dist/{summary-report-BNR7DWTj.d.ts.map → summary-report-CaL-Hnxt.d.ts.map} +1 -1
  250. package/dist/supervisor-run/index.d.ts +391 -3
  251. package/dist/supervisor-run/index.d.ts.map +1 -0
  252. package/dist/supervisor-run/index.js +1689 -2
  253. package/dist/{supervisor-run-D_sokXcO.js.map → supervisor-run/index.js.map} +1 -1
  254. package/dist/{extract-usage-CdZdoj1s.js → task-failure-attributes-CpQ4y5RD.js} +5 -157
  255. package/dist/task-failure-attributes-CpQ4y5RD.js.map +1 -0
  256. package/dist/{tool-groups-CZPGGlHf.d.ts → tool-groups-Cteb03Ps.d.ts} +3 -3
  257. package/dist/tool-groups-Cteb03Ps.d.ts.map +1 -0
  258. package/dist/tool-waste-BDdBZG1F.js +803 -0
  259. package/dist/tool-waste-BDdBZG1F.js.map +1 -0
  260. package/dist/tool-waste-DjRDEsuI.d.ts +128 -0
  261. package/dist/tool-waste-DjRDEsuI.d.ts.map +1 -0
  262. package/dist/trace-repair/index.d.ts +14 -5
  263. package/dist/trace-repair/index.d.ts.map +1 -1
  264. package/dist/trace-repair/index.js +35 -7
  265. package/dist/trace-repair/index.js.map +1 -1
  266. package/dist/traces.d.ts +406 -7
  267. package/dist/traces.d.ts.map +1 -0
  268. package/dist/traces.js +1011 -10
  269. package/dist/traces.js.map +1 -0
  270. package/dist/trajectory-replay/index.d.ts +16 -3
  271. package/dist/trajectory-replay/index.d.ts.map +1 -1
  272. package/dist/trajectory-replay/index.js +52 -5
  273. package/dist/trajectory-replay/index.js.map +1 -1
  274. package/dist/types-BEPZc6eo.d.ts +93 -0
  275. package/dist/types-BEPZc6eo.d.ts.map +1 -0
  276. package/dist/{usage-receipt-t7vAzCRQ.js → types-BI4fT3HN.js} +67 -67
  277. package/dist/types-BI4fT3HN.js.map +1 -0
  278. package/dist/{types-BnjdJ70P.d.ts → types-BZ59Ahr8.d.ts} +5 -5
  279. package/dist/{types-BnjdJ70P.d.ts.map → types-BZ59Ahr8.d.ts.map} +1 -1
  280. package/dist/{types-D216SgwM.d.ts → types-Cx3YUh2r.d.ts} +4 -241
  281. package/dist/types-Cx3YUh2r.d.ts.map +1 -0
  282. package/dist/{types-BvDKaULh.d.ts → types-DLQx4mKU.d.ts} +4 -4
  283. package/dist/{types-BvDKaULh.d.ts.map → types-DLQx4mKU.d.ts.map} +1 -1
  284. package/dist/{types-D4mog56g.d.ts → types-yLK8gXE9.d.ts} +2 -2
  285. package/dist/{types-D4mog56g.d.ts.map → types-yLK8gXE9.d.ts.map} +1 -1
  286. package/dist/verdict-BndeTAh_.js +61 -0
  287. package/dist/verdict-BndeTAh_.js.map +1 -0
  288. package/dist/verdict-E4eRNf7-.d.ts +392 -0
  289. package/dist/verdict-E4eRNf7-.d.ts.map +1 -0
  290. package/dist/{verdict-cache-BCcOh0kF.js → verdict-cache-mZf5FEiY.js} +3 -55
  291. package/dist/{verdict-cache-BCcOh0kF.js.map → verdict-cache-mZf5FEiY.js.map} +1 -1
  292. package/dist/wire/index.d.ts +3 -3
  293. package/dist/wire/index.d.ts.map +1 -1
  294. package/dist/wire/index.js +1 -1
  295. package/docs/charter.md +3 -3
  296. package/docs/control-runtime.md +3 -42
  297. package/docs/experiment.md +0 -1
  298. package/docs/feature-guide.md +2 -2
  299. package/docs/trace-repair-grader.md +1 -0
  300. package/docs/trajectory-replay.md +1 -0
  301. package/docs/verdicts.md +43 -0
  302. package/docs/verification-strategies.md +3 -2
  303. package/package.json +6 -11
  304. package/dist/agent-profile-DPi7IZg7.d.ts.map +0 -1
  305. package/dist/analyze-runs-C30yljDJ.js.map +0 -1
  306. package/dist/baseline-CavEbRyH.d.ts +0 -136
  307. package/dist/baseline-CavEbRyH.d.ts.map +0 -1
  308. package/dist/benchmark-command-BteMFN62.js.map +0 -1
  309. package/dist/benchmarks-Dzs8CKb1.js +0 -755
  310. package/dist/benchmarks-Dzs8CKb1.js.map +0 -1
  311. package/dist/campaign-C2TTzQII.js.map +0 -1
  312. package/dist/completion-verifier-DJA5BhPb.d.ts +0 -414
  313. package/dist/completion-verifier-DJA5BhPb.d.ts.map +0 -1
  314. package/dist/control.d.ts +0 -3
  315. package/dist/control.js +0 -2
  316. package/dist/dataset-C8xaLXdY.d.ts.map +0 -1
  317. package/dist/default-registry-Bf8Woqmq.d.ts.map +0 -1
  318. package/dist/default-registry-BmktKy8r.js.map +0 -1
  319. package/dist/engine-3hL-XqwJ.d.ts.map +0 -1
  320. package/dist/experiment-tracker-CnRICnMl.js +0 -500
  321. package/dist/experiment-tracker-CnRICnMl.js.map +0 -1
  322. package/dist/experiment-tracker-IMntXr6J.d.ts.map +0 -1
  323. package/dist/extract-usage-CdZdoj1s.js.map +0 -1
  324. package/dist/failure-cluster-CqcvCcdR.d.ts.map +0 -1
  325. package/dist/feedback-trajectory-CacHpxsp.d.ts.map +0 -1
  326. package/dist/index-BZ3-y4YL.d.ts +0 -391
  327. package/dist/index-BZ3-y4YL.d.ts.map +0 -1
  328. package/dist/index-CQTZ-4XN.d.ts.map +0 -1
  329. package/dist/index-DPPGNJ_R.d.ts.map +0 -1
  330. package/dist/index-YE4KdKbO2.d.ts +0 -335
  331. package/dist/index-YE4KdKbO2.d.ts.map +0 -1
  332. package/dist/paired-arms-iZ08VFMN.js +0 -260
  333. package/dist/paired-arms-iZ08VFMN.js.map +0 -1
  334. package/dist/paired-promotion-decision-B6zJ3gYM.d.ts +0 -114
  335. package/dist/paired-promotion-decision-B6zJ3gYM.d.ts.map +0 -1
  336. package/dist/prime-protocol-BfSalTfR.js.map +0 -1
  337. package/dist/promotion-policy-Ckjhzg_4.d.ts +0 -289
  338. package/dist/promotion-policy-Ckjhzg_4.d.ts.map +0 -1
  339. package/dist/promotion-policy-CrLrmys8.js.map +0 -1
  340. package/dist/proposal-findings-2GIUo1et.js.map +0 -1
  341. package/dist/propose-review-control-dSNPjFUH.js +0 -1458
  342. package/dist/propose-review-control-dSNPjFUH.js.map +0 -1
  343. package/dist/release-report-BUYmoKo2.js.map +0 -1
  344. package/dist/release-report-DA2BCu5a.d.ts.map +0 -1
  345. package/dist/replay-CohS93nE.js +0 -1859
  346. package/dist/replay-CohS93nE.js.map +0 -1
  347. package/dist/replay-DbhZ4Ked.d.ts +0 -834
  348. package/dist/replay-DbhZ4Ked.d.ts.map +0 -1
  349. package/dist/reward-hacking-BDToousL.js.map +0 -1
  350. package/dist/run-evidence-Dhi3C81V.d.ts +0 -225
  351. package/dist/run-evidence-Dhi3C81V.d.ts.map +0 -1
  352. package/dist/schema-Cef2cFmb2.d.ts.map +0 -1
  353. package/dist/semantic-concept-judge-D1z-KepS.js +0 -767
  354. package/dist/semantic-concept-judge-D1z-KepS.js.map +0 -1
  355. package/dist/series-convergence-ofsqPWhs.d.ts +0 -35
  356. package/dist/series-convergence-ofsqPWhs.d.ts.map +0 -1
  357. package/dist/server-iu0ede49.js.map +0 -1
  358. package/dist/single-run-lock-DFWHEB09.js.map +0 -1
  359. package/dist/skill-usage-CO9OLRBx.d.ts.map +0 -1
  360. package/dist/skillopt-optimization-method-CQdVeM8k.js.map +0 -1
  361. package/dist/skillopt-optimization-method-USDKhxSA.d.ts.map +0 -1
  362. package/dist/statistical-heldout-TQ-4CYiN.d.ts.map +0 -1
  363. package/dist/statistics-ByxzSiOM.js +0 -2212
  364. package/dist/statistics-ByxzSiOM.js.map +0 -1
  365. package/dist/statistics-D6Uebe_4.d.ts +0 -968
  366. package/dist/statistics-D6Uebe_4.d.ts.map +0 -1
  367. package/dist/supervisor-run-D_sokXcO.js +0 -1690
  368. package/dist/test-graded-scenario-D1TaI2va.d.ts +0 -141
  369. package/dist/test-graded-scenario-D1TaI2va.d.ts.map +0 -1
  370. package/dist/test-graded-scenario-JHcKQNpq.js +0 -318
  371. package/dist/test-graded-scenario-JHcKQNpq.js.map +0 -1
  372. package/dist/tool-groups-CZPGGlHf.d.ts.map +0 -1
  373. package/dist/tool-use-metrics-DEGMKycK.js +0 -370
  374. package/dist/tool-use-metrics-DEGMKycK.js.map +0 -1
  375. package/dist/types-D216SgwM.d.ts.map +0 -1
  376. package/dist/usage-receipt-t7vAzCRQ.js.map +0 -1
  377. package/dist/verdict-DExhxfgR.d.ts +0 -201
  378. package/dist/verdict-DExhxfgR.d.ts.map +0 -1
@@ -1,968 +0,0 @@
1
- import { g as JudgeScore } from "./types-D216SgwM.js";
2
- //#region src/judge-calibration.d.ts
3
- /**
4
- * Judge calibration — measure judge quality against human gold + bias.
5
- *
6
- * Workflow:
7
- * 1. Build a golden set: {itemId, humanScore}[].
8
- * 2. Run candidate judges; each produces {itemId, score}.
9
- * 3. `calibrateJudge(golden, candidate)` reports κ + Pearson + MAE.
10
- * 4. `calibrateJudgeContinuous(golden, candidate)` adds quadratic-weighted
11
- * κ over the un-rounded [0,1] scores plus ICC(2,1), Pearson, Spearman,
12
- * and bootstrap CIs — use this for fine-grained judges where rounding
13
- * to int discards information (e.g. 0.78 vs 0.81 both round to 1 and
14
- * look "perfectly agreed" to integer κ).
15
- * 5. Run bias probes (positional, verbosity, self-preference) to
16
- * detect systematic score inflation.
17
- * 6. For N≥2 judges on the same items, `continuousAgreement(scores)`
18
- * reports ICC(2,1) + κ_w + Pearson + Spearman with bootstrap CIs.
19
- *
20
- * Returns actionable diagnostics, not a single number. Consumers then
21
- * decide whether to trust the judge, retrain it, or add a tie-breaker.
22
- */
23
- interface GoldenItem {
24
- itemId: string;
25
- humanScore: number;
26
- /** Optional group used for per-group bias audits (e.g. model-of-output family). */
27
- group?: string;
28
- }
29
- interface CandidateScore {
30
- itemId: string;
31
- score: number;
32
- /** Optional — enables positional-bias analysis (did order matter?). */
33
- positionOfAInput?: 'first' | 'second';
34
- }
35
- interface CalibrationResult {
36
- n: number;
37
- pearson: number;
38
- /** Cohen's κ with quadratic weights over integer-rounded scores. */
39
- kappa: number;
40
- /** Mean absolute error vs human. */
41
- mae: number;
42
- /** Worst-5 miscalibrations (largest |judge - human|). */
43
- worstItems: Array<{
44
- itemId: string;
45
- judge: number;
46
- human: number;
47
- delta: number;
48
- }>;
49
- }
50
- /**
51
- * Measure judge quality against human gold labels: computes Cohen's κ, Pearson correlation, and MAE over matched item ids.
52
- */
53
- declare function calibrateJudge(golden: GoldenItem[], candidate: CandidateScore[]): CalibrationResult;
54
- interface PositionalBiasResult {
55
- /**
56
- * Score delta (first-position - second-position) averaged across items
57
- * presented in both positions. Non-zero = positional bias.
58
- */
59
- avgDelta: number;
60
- n: number;
61
- }
62
- /**
63
- * Feed the same items to the judge twice with A/B swapped and pass all
64
- * results here. Items that don't appear in both positions are ignored.
65
- */
66
- declare function positionalBias(scores: CandidateScore[]): PositionalBiasResult;
67
- interface VerbosityBiasResult {
68
- /** Pearson correlation between output length and score. Strong positive = verbosity bias. */
69
- pearson: number;
70
- n: number;
71
- }
72
- declare function verbosityBias(samples: Array<{
73
- outputLen: number;
74
- score: number;
75
- }>): VerbosityBiasResult;
76
- interface SelfPreferenceResult {
77
- /** Mean judge score when judge's family matches output's family. */
78
- inFamilyMean: number;
79
- outOfFamilyMean: number;
80
- deltaMean: number;
81
- n: number;
82
- }
83
- /**
84
- * Pass the same scenarios scored with judge-model X grading outputs from
85
- * model X (in-family) and model Y (out-of-family). Non-zero delta
86
- * indicates self-preference.
87
- */
88
- declare function selfPreference(samples: Array<{
89
- score: number;
90
- inFamily: boolean;
91
- }>): SelfPreferenceResult;
92
- interface ContinuousAgreement {
93
- /** Cohen's κ_w with quadratic weights, computed on raw [0,1] scores. */
94
- weightedKappa: number;
95
- /** ICC(2,1): two-way random effects, absolute agreement, single rater. */
96
- icc: number;
97
- /** Pearson product-moment correlation (averaged over rater pairs if N>2). */
98
- pearson: number;
99
- /** Spearman rank correlation (averaged over rater pairs if N>2). */
100
- spearman: number;
101
- /** 95% bootstrap percentile CIs over items. */
102
- ci: {
103
- icc: [number, number];
104
- weightedKappa: [number, number];
105
- };
106
- /** Number of complete items (no NaN across raters). */
107
- n: number;
108
- /** Number of raters. */
109
- raters: number;
110
- }
111
- interface ContinuousAgreementOptions {
112
- /** Bootstrap iterations. Default 1000. Set to 0 to skip CIs (CI = [NaN, NaN]). */
113
- bootstrap?: number;
114
- /** κ weighting scheme. Default 'quadratic'. */
115
- weights?: 'linear' | 'quadratic';
116
- /** PRNG seed for reproducible bootstrap. Default 0xC0FFEE. */
117
- seed?: number;
118
- /** Confidence level for percentile CI. Default 0.95. */
119
- ciLevel?: number;
120
- }
121
- /**
122
- * Inter-rater agreement on continuous (typically [0,1]) scores.
123
- *
124
- * `scores` has shape [n_items][n_raters]. Rows with any non-finite entry
125
- * are dropped. Returns NaN metrics if fewer than 2 raters or 2 complete
126
- * items remain.
127
- */
128
- declare function continuousAgreement(scores: number[][], opts?: ContinuousAgreementOptions): ContinuousAgreement;
129
- interface ContinuousCalibrationResult extends CalibrationResult {
130
- /** Cohen's κ_w computed on raw (un-rounded) scores. */
131
- weightedKappaContinuous: number;
132
- /** ICC(2,1) treating golden + candidate as two raters. */
133
- icc: number;
134
- spearman: number;
135
- ci: {
136
- icc: [number, number];
137
- weightedKappa: [number, number];
138
- };
139
- }
140
- /**
141
- * Extends `calibrateJudge` with continuous-value agreement metrics while
142
- * retaining its base calibration summary.
143
- */
144
- declare function calibrateJudgeContinuous(golden: GoldenItem[], candidate: CandidateScore[], opts?: ContinuousAgreementOptions): ContinuousCalibrationResult;
145
- //#endregion
146
- //#region src/statistics.d.ts
147
- /** Identity: dimensions already follow "higher = better" by prompt convention
148
- * (inverted dims like hallucination are scored 10 = best at the source). */
149
- declare const normalizeScores: (scores: JudgeScore[]) => JudgeScore[];
150
- /** Weighted mean — falls back to uniform weights when omitted */
151
- declare function weightedMean(scores: {
152
- score: number;
153
- weight?: number;
154
- }[]): number;
155
- /**
156
- * Percentile bootstrap confidence interval on the mean of `scores`.
157
- *
158
- * Descriptive spread. It is not a significance test, and at small n its bounds
159
- * are anti-conservative in the same way {@link pairedBootstrap}'s are — see
160
- * {@link BOOTSTRAP_GATE_MIN_N}. With no `seed` the resampling is seeded from
161
- * the scores themselves, so the interval is reproducible either way.
162
- */
163
- declare function confidenceInterval(scores: number[], confidence?: number, opts?: {
164
- seed?: number;
165
- resamples?: number;
166
- }): {
167
- mean: number;
168
- lower: number;
169
- upper: number;
170
- };
171
- /**
172
- * Inter-rater reliability — Krippendorff's α under the squared-difference
173
- * metric, pooled across dimensions.
174
- *
175
- * Each inner array is one judge's scores. Items are matched by position
176
- * WITHIN a dimension: the k-th score a judge supplies carrying dimension
177
- * `d` is item k of `d`, and the ratings compared against each other are
178
- * the ones different judges gave to the same item. Every judge that scores
179
- * a dimension at all must supply the same number of scores for it —
180
- * ragged input cannot be aligned into items and throws rather than
181
- * comparing mismatched items.
182
- *
183
- * α = 1 − D_observed / D_expected: D_observed averages the squared
184
- * difference over within-item judge pairs, D_expected over every pair of
185
- * ratings irrespective of item. α = 1 is perfect agreement, 0 is chance,
186
- * negative is systematic disagreement.
187
- */
188
- declare function interRaterReliability(judgeScores: JudgeScore[][]): number;
189
- /** How a rank test's p-value was actually computed. */
190
- type RankTestMethod = 'exact' | 'permutation' | 'asymptotic';
191
- /**
192
- * What the caller asks for. `'auto'` selects `'exact'` inside the enumeration
193
- * threshold and `'permutation'` above it, and never selects `'asymptotic'`.
194
- */
195
- type RankTestMethodRequest = 'auto' | 'exact' | 'asymptotic';
196
- interface RankTestOptions {
197
- /** Default `'auto'`. `'asymptotic'` inside the exact-feasible range throws. */
198
- method?: RankTestMethodRequest;
199
- /** Resamples on the Monte Carlo permutation path. Default 100000. */
200
- permutations?: number;
201
- /** Seed for the permutation path. Omitted ⇒ derived from the data itself, so
202
- * the result is reproducible either way. */
203
- seed?: number;
204
- }
205
- /** Maximum dynamic-programming cells used by an exact two-sample rank test. */
206
- declare const MANN_WHITNEY_EXACT_MAX_STATES = 8192;
207
- /** Maximum inner-loop transitions used by an exact two-sample rank test. */
208
- declare const MANN_WHITNEY_EXACT_MAX_WORK = 250000;
209
- /** Non-zero differences up to which the signed-rank null is enumerated exactly. */
210
- declare const WILCOXON_EXACT_MAX_N = 20;
211
- /** Resamples used when a rank test falls back to Monte Carlo permutation. */
212
- declare const DEFAULT_PERMUTATIONS = 100000;
213
- interface MannWhitneyResult {
214
- /** `min(U_a, U_b)` — the conventional reported statistic. */
215
- u: number;
216
- /** U for sample `a`. Carries the direction of the effect, which `u` discards. */
217
- uA: number;
218
- /** Two-sided p-value. */
219
- p: number;
220
- /** How `p` was computed. */
221
- method: RankTestMethod;
222
- /** Smallest two-sided p this design can produce. `p` can never be below it. */
223
- pFloor: number;
224
- }
225
- /**
226
- * Mann-Whitney U — two independent samples, no distributional assumption.
227
- *
228
- * Exact conditional (permutation) p by default when the dynamic program fits
229
- * {@link MANN_WHITNEY_EXACT_MAX_STATES} cells and
230
- * {@link MANN_WHITNEY_EXACT_MAX_WORK} transitions, seeded Monte Carlo
231
- * permutation above those limits. This keeps imbalanced designs such as 1+24
232
- * exact without admitting expensive balanced designs merely because they have
233
- * the same total size. Throws on non-finite input and on `method:
234
- * 'asymptotic'` where an exact answer is available. Empty input yields `p = 1,
235
- * pFloor = 1` — no design, no attainable evidence.
236
- */
237
- declare function mannWhitneyU(a: number[], b: number[], opts?: RankTestOptions): MannWhitneyResult;
238
- /** Partial credit: returns 0-1 ratio of current toward target */
239
- declare function partialCredit(current: number, target: number): number;
240
- interface PairedTTestResult {
241
- /** Null when the statistic is undefined — see {@link pairedTTest}. */
242
- t: number | null;
243
- df: number;
244
- /** Null exactly when `t` is null. */
245
- p: number | null;
246
- }
247
- /**
248
- * Paired t-test — before/after measurements on the SAME items.
249
- * Pairing removes inter-item variance, giving tighter significance than
250
- * an unpaired test when comparing prompt v1 vs prompt v2 on identical
251
- * scenarios.
252
- *
253
- * Returns `t = p = null` where the statistic is undefined: fewer than two
254
- * pairs, or a non-zero constant delta whose observed variance is zero. A
255
- * constant shift carries no information about the variance it would have to
256
- * be compared against, so the honest answer is "undefined", not `p = 0` —
257
- * three observations cannot buy absolute certainty. This is the same contract
258
- * {@link pairedCohensDz} states for the same condition. An all-zero delta is
259
- * different: it is a measured null, and returns `t = 0, p = 1`.
260
- */
261
- declare function pairedTTest(before: number[], after: number[]): PairedTTestResult;
262
- interface WilcoxonSignedRankResult {
263
- /** W⁺, the rank sum of the positive differences. (scipy reports
264
- * `min(W⁺, W⁻)`; compare statistics only after converting.) */
265
- w: number;
266
- /** Two-sided p-value. */
267
- p: number;
268
- /** How `p` was computed. */
269
- method: RankTestMethod;
270
- /** Smallest two-sided p this design can produce. */
271
- pFloor: number;
272
- /** Non-zero differences — zero differences are dropped and carry no rank. */
273
- nNonZero: number;
274
- }
275
- /**
276
- * Wilcoxon signed-rank — paired, no distributional assumption on the deltas.
277
- *
278
- * Exact conditional (sign-flip) p by default at `n ≤
279
- * {@link WILCOXON_EXACT_MAX_N}` non-zero differences, seeded Monte Carlo
280
- * permutation above it. Throws on non-finite input and on `method:
281
- * 'asymptotic'` where an exact answer is available.
282
- *
283
- * `n` is the count of NON-ZERO differences: exact ties are dropped before
284
- * ranking, so a run of tied pairs shrinks the design and raises `pFloor`.
285
- * All-tied input yields `p = 1, pFloor = 1` — no attainable evidence, which
286
- * `pFloor` states rather than leaving `p = 1` to be read as a measured null.
287
- */
288
- declare function wilcoxonSignedRank(before: number[], after: number[], opts?: RankTestOptions): WilcoxonSignedRankResult;
289
- /**
290
- * Cohen's d — standardized effect size for two independent groups.
291
- * Positive d means group b has higher mean than group a.
292
- * Rule of thumb: |d| < 0.2 negligible, 0.2–0.5 small, 0.5–0.8 medium, > 0.8 large.
293
- *
294
- * Returns null where the standardized effect is undefined: fewer than two
295
- * observations in either group, or a zero pooled standard deviation with
296
- * unequal means. Null is NOT "no effect" — zero within-group spread across a
297
- * real mean gap is an unbounded effect, the opposite of negligible. Equal
298
- * means with zero spread is a genuine 0. Same contract as
299
- * {@link pairedCohensDz}.
300
- */
301
- declare function cohensD(a: number[], b: number[]): number | null;
302
- /**
303
- * Cohen's dz for paired observations: mean(after - before) divided by the
304
- * sample standard deviation of those within-pair deltas.
305
- *
306
- * Returns null when fewer than two pairs exist or a non-zero constant delta
307
- * has zero observed variance. In that case the standardized effect is
308
- * undefined, not an arbitrarily large finite number.
309
- */
310
- declare function pairedCohensDz(before: number[], after: number[]): number | null;
311
- type CliffsMagnitude = 'negligible' | 'small' | 'medium' | 'large';
312
- /**
313
- * Cliff's delta — a non-parametric effect size for two independent samples.
314
- * `δ = (#(after > before) − #(after < before)) / (n_before · n_after)`,
315
- * ranging [-1, 1]. Positive ⇒ `after` tends to exceed `before` (improvement).
316
- *
317
- * Distribution-free counterpart to Cohen's d: no normality assumption, robust
318
- * to the bounded/skewed score distributions judges produce. Pairs with
319
- * `pairedBootstrap` / `wilcoxonSignedRank` for the non-parametric reporting
320
- * path. Returns 0 when either sample is empty.
321
- */
322
- declare function cliffsDelta(before: number[], after: number[]): number;
323
- /**
324
- * Map a Cliff's delta to a qualitative magnitude using the standard
325
- * Romano et al. thresholds (|δ|): <0.147 negligible, <0.33 small,
326
- * <0.474 medium, else large.
327
- */
328
- declare function interpretCliffs(delta: number): CliffsMagnitude;
329
- /**
330
- * Average-rank-with-ties transform (1-indexed). Tied values receive the mean
331
- * of the ranks they span, the standard correction for Spearman's ρ.
332
- */
333
- declare function ranks(xs: number[]): number[];
334
- /**
335
- * Pearson product-moment correlation coefficient r ∈ [-1, 1] between two
336
- * equal-length series. See the edge-case contract above: NaN for n < 2 or
337
- * unequal lengths, 1 when both series are constant, 0 when exactly one is.
338
- */
339
- declare function pearsonR(a: number[], b: number[]): number;
340
- /**
341
- * Spearman's rank correlation ρ — Pearson over the average-rank-with-ties
342
- * transform of each series. Same edge-case contract as {@link pearsonR}.
343
- */
344
- declare function spearmanR(a: number[], b: number[]): number;
345
- interface WeightedCompositeInput {
346
- /** Per-dimension scores (typically 0..1). */
347
- dims: Record<string, number>;
348
- /** Weight per dimension. Every weighted dimension MUST be present in
349
- * `dims` — a weight for an absent dimension is a config error and throws,
350
- * because silently dropping it would renormalise the composite onto a
351
- * different denominator than intended. */
352
- weights: Record<string, number>;
353
- /** Optional pass threshold; when set, the result reports `pass`. */
354
- threshold?: number;
355
- }
356
- interface WeightedCompositeResult {
357
- composite: number;
358
- pass?: boolean;
359
- }
360
- /**
361
- * Weighted composite over judge dimensions: `Σ(score_d · w_d) / Σ(w_d)` across
362
- * the weighted dimensions. The canonical replacement for the per-consumer
363
- * hand-rolled composite math (tax/legal/creative/gtm each ship a copy).
364
- *
365
- * Fail-loud: throws if a weighted dimension is missing from `dims`, if any
366
- * weight is negative, or if the weights sum to 0 — none of which can produce
367
- * a meaningful composite.
368
- */
369
- declare function weightedComposite(input: WeightedCompositeInput): WeightedCompositeResult;
370
- interface CorpusScoreRecord {
371
- /** Stable identifier for the rated item (scenario, span, turn, …). */
372
- itemId: string;
373
- /** Identifier for the judge that produced this score. */
374
- judgeName: string;
375
- /** Dimension name (matches `JudgeScore.dimension`). */
376
- dimension: string;
377
- /** Numeric score; must be finite. */
378
- score: number;
379
- }
380
- interface CorpusAgreementPerDimension extends ContinuousAgreement {
381
- dimension: string;
382
- /** Item IDs that contributed to this dimension's matrix (every judge scored them). */
383
- itemIds: string[];
384
- /** Judge IDs that contributed to this dimension's matrix. */
385
- judgeIds: string[];
386
- }
387
- interface CorpusAgreementReport {
388
- /** Per-dimension ICC(2,1) + κ_w + Pearson + Spearman + bootstrap CIs. */
389
- perDimension: CorpusAgreementPerDimension[];
390
- /** Mean ICC across dimensions (NaN if no dimension yielded a finite ICC). */
391
- overallIcc: number;
392
- /** Mean weighted κ across dimensions (NaN if none finite). */
393
- overallWeightedKappa: number;
394
- /** Dimensions evaluated (sorted). */
395
- dimensions: string[];
396
- /** Judges seen across the corpus (sorted). */
397
- judgeIds: string[];
398
- }
399
- interface CorpusAgreementOptions extends ContinuousAgreementOptions {
400
- /**
401
- * Restrict the audit to these dimensions. Default = every dimension
402
- * that appears in the input. A dimension named here but absent from
403
- * the input throws — silent omission would corrupt the overall metric.
404
- */
405
- dimensions?: string[];
406
- /**
407
- * Restrict the audit to these judges. Default = every judge that
408
- * appears in the input. A judge named here but absent from a
409
- * dimension throws (see "fail loud" below).
410
- */
411
- judges?: string[];
412
- }
413
- /**
414
- * Corpus-wide inter-rater agreement across N items × M judges × D dimensions.
415
- *
416
- * For each dimension, builds the [n_items][n_judges] matrix of scores
417
- * (keeping only items every judge rated on that dimension), then runs
418
- * `continuousAgreement` to get ICC(2,1), κ_w, Pearson, Spearman, and
419
- * bootstrap CIs. Reports a pooled mean across dimensions as a single
420
- * "is this judge panel reliable on this corpus?" number.
421
- *
422
- * Fail-loud contract:
423
- * - Empty input throws.
424
- * - Fewer than 2 judges or fewer than 2 items per dimension throws.
425
- * - A judge present in some dimensions but with zero scored items on
426
- * another dimension throws (would silently shrink the matrix).
427
- * - Duplicate (itemId, judgeName, dimension) records throw.
428
- */
429
- declare function corpusInterRaterAgreement(records: CorpusScoreRecord[], opts?: CorpusAgreementOptions): CorpusAgreementReport;
430
- /**
431
- * Convenience adapter for `JudgeScore[]` data keyed externally by item.
432
- *
433
- * Use when you have per-item arrays of `JudgeScore[]` (e.g. one
434
- * `ScenarioResult.judgeScores` per scenario) and want corpus-wide
435
- * agreement without manually flattening. `itemId` must be unique per
436
- * row of `itemsScores`.
437
- */
438
- declare function corpusInterRaterAgreementFromJudgeScores(itemsScores: Array<{
439
- itemId: string;
440
- scores: JudgeScore[];
441
- }>, opts?: CorpusAgreementOptions): CorpusAgreementReport;
442
- /**
443
- * Required N per arm for a two-sample comparison at target effect size,
444
- * alpha, and power. Normal-approximation formula:
445
- * n = 2 * ( (z_{1-α/2} + z_{1-β}) / d )^2
446
- * where d is Cohen's d. Returns Infinity for effect ≤ 0.
447
- */
448
- declare function requiredSampleSize(opts: {
449
- effect: number;
450
- alpha?: number;
451
- power?: number;
452
- twoSided?: boolean;
453
- }): number;
454
- /**
455
- * Required number of paired observations for a target Cohen's dz.
456
- * Unlike the independent-groups formula, this has no two-arm factor of two.
457
- *
458
- * Normal quantiles with no t correction, so treat the result as a LOWER bound:
459
- * it returns 32 where the exact t-based answer is 34 at dz = 0.5, and 13 where
460
- * it is 15 at dz = 0.8 — a 6–13 % shortfall precisely in the range a caller
461
- * consults to decide whether 3–10 repetitions suffice.
462
- */
463
- declare function requiredPairedSampleSize(opts: {
464
- effect: number;
465
- alpha?: number;
466
- power?: number;
467
- twoSided?: boolean;
468
- }): number;
469
- /**
470
- * Minimum detectable paired effect (standardised units) for a target paired
471
- * sample size: d_min = (z_{1-α/2} + z_β) / sqrt(n_paired). Multiply by
472
- * sd(deltas) for score units; treat as a lower bound — Wilcoxon and bootstrap
473
- * have asymptotic relative efficiency below 1 vs the t-test on heavy tails.
474
- */
475
- declare function pairedMde(opts: {
476
- nPaired: number;
477
- alpha?: number;
478
- power?: number;
479
- twoSided?: boolean;
480
- }): number;
481
- /**
482
- * Number of paired observations needed for a McNemar test to reach a target
483
- * power — the pre-registration companion to {@link mcnemar}. Parametrised by the
484
- * expected discordant-cell probabilities `p10` (P[treatment wins on a pair]) and
485
- * `p01` (P[control wins]); concordant pairs carry no information, so the count
486
- * is driven entirely by the discordant rate. Lachin's (1992) asymptotic normal
487
- * approximation: with discordant rate `pDisc = p10 + p01` and marginal effect
488
- * `δ = p10 − p01`,
489
- * n = ( z_{1-α/2}·√pDisc + z_{1-β}·√(pDisc − δ²) )² / δ².
490
- * Returns Infinity when there is no effect (p10 === p01). Asymptotic — at the
491
- * tiny discordant counts where the exact {@link mcnemar} differs from the normal
492
- * approximation, treat the result as a lower bound and prefer the discordant-pair
493
- * floor.
494
- */
495
- declare function mcnemarRequiredN(opts: {
496
- p10: number;
497
- p01: number;
498
- alpha?: number;
499
- power?: number;
500
- twoSided?: boolean;
501
- }): number;
502
- /**
503
- * Power of a McNemar test at a given number of paired observations, the inverse
504
- * of {@link mcnemarRequiredN} (same Lachin asymptotic model, same parameters).
505
- * Returns a value in [0, 1]; equals `alpha` when there is no effect.
506
- */
507
- declare function mcnemarPower(opts: {
508
- p10: number;
509
- p01: number;
510
- nPairs: number;
511
- alpha?: number;
512
- twoSided?: boolean;
513
- }): number;
514
- /**
515
- * Bonferroni adjustment: multiply every p-value by the test count, clamp at 1.
516
- *
517
- * Rejects at `p_adjusted ≤ alpha` — the boundary is inclusive, matching
518
- * {@link holm}, which uniformly dominates this correction and must therefore
519
- * never reject less. Validates its inputs on the same terms.
520
- */
521
- declare function bonferroni(pValues: readonly number[], alpha?: number): {
522
- adjusted: number[];
523
- significant: boolean[];
524
- };
525
- /**
526
- * Holm step-down family-wise error adjustment.
527
- *
528
- * P-values are sorted from smallest to largest, multiplied by their remaining
529
- * hypothesis count, and made monotonically non-decreasing before being mapped
530
- * back to input order. This uniformly dominates plain Bonferroni while keeping
531
- * strong family-wise error control under arbitrary dependence.
532
- */
533
- declare function holm(pValues: readonly number[], alpha?: number): {
534
- adjusted: number[];
535
- significant: boolean[];
536
- };
537
- /**
538
- * Benjamini–Hochberg false discovery rate. Returns adjusted q-values and
539
- * significance at the target FDR; handles ties and preserves q monotonicity.
540
- *
541
- * Rejects at `q ≤ fdr` — the BH rule is inclusive at the boundary, so an
542
- * exactly-`fdr` q-value is a discovery.
543
- */
544
- declare function benjaminiHochberg(pValues: readonly number[], fdr?: number): {
545
- qValues: number[];
546
- significant: boolean[];
547
- };
548
- interface PairedBootstrapResult {
549
- /** Number of paired observations. */
550
- n: number;
551
- /** Median of paired deltas (after − before). */
552
- median: number;
553
- /** Mean of paired deltas. */
554
- mean: number;
555
- /** Lower bound of the bootstrap CI on the chosen statistic. */
556
- low: number;
557
- /** Upper bound of the bootstrap CI on the chosen statistic. */
558
- high: number;
559
- /** Confidence level used (e.g. 0.95). */
560
- confidence: number;
561
- /** Number of bootstrap resamples used. */
562
- resamples: number;
563
- /** False below {@link BOOTSTRAP_GATE_MIN_N}. See {@link pairedBootstrap}. */
564
- gateEligible: boolean;
565
- }
566
- /**
567
- * Pairs below which a percentile bootstrap interval is descriptive spread only.
568
- *
569
- * `P(low > 0)` under a true null, against a nominal 2.5 %, measured over 4000
570
- * seeded trials: 13.53 % at n = 3, 3.52 % at n = 10, 3.10 % at n = 20 on the
571
- * median; 13.85 %, 4.90 %, 3.80 % on the mean. This is intrinsic to resampling
572
- * three points, not an implementation error — scipy's BCa gives 16.0 % on the
573
- * same n = 3 data — so no change to the estimator moves it. Below this floor
574
- * the decision belongs to the exact sign test or exact signed-rank test.
575
- */
576
- declare const BOOTSTRAP_GATE_MIN_N = 20;
577
- interface PairedBootstrapOptions {
578
- /** Confidence level. Default 0.95. */
579
- confidence?: number;
580
- /** Bootstrap resample count. Default 2000. */
581
- resamples?: number;
582
- /** Statistic to bootstrap. Default 'median'. */
583
- statistic?: 'median' | 'mean';
584
- /** Deterministic seed. If omitted, derived from the deltas so the interval
585
- * is reproducible regardless. */
586
- seed?: number;
587
- }
588
- /**
589
- * Paired bootstrap on (after − before) deltas. Returns a CI on the chosen
590
- * statistic (median by default); pairs are resampled with replacement. Throws
591
- * on unequal sample sizes.
592
- *
593
- * `low > threshold` carries the stated confidence ONLY at `n ≥
594
- * {@link BOOTSTRAP_GATE_MIN_N}`, which `gateEligible` reports. Below it the
595
- * check fires under a true null several times more often than nominal, so the
596
- * interval is descriptive spread and a promotion must not turn on it.
597
- */
598
- declare function pairedBootstrap(before: number[], after: number[], opts?: PairedBootstrapOptions): PairedBootstrapResult;
599
- /** Pre-registered direction for a one-sided paired sign test. */
600
- type SignTestAlternative = 'greater' | 'less';
601
- /** Exact one-sided sign-test result for paired numeric differences. */
602
- interface PairedSignTestResult {
603
- /** Total supplied differences, including zero ties. */
604
- n: number;
605
- /** Strictly positive differences. */
606
- positive: number;
607
- /** Strictly negative differences. */
608
- negative: number;
609
- /** Zero differences excluded from the binomial test. */
610
- ties: number;
611
- /** Non-zero differences used by the binomial test. */
612
- nNonTies: number;
613
- /** Direction of the pre-registered alternative hypothesis. */
614
- alternative: SignTestAlternative;
615
- /** Exact one-sided p-value under P(positive) = P(negative) = 0.5. */
616
- pValue: number;
617
- }
618
- /**
619
- * Exact one-sided sign test over paired differences.
620
- *
621
- * Pass `after[i] - before[i]` for each matched item. `alternative = 'greater'`
622
- * tests whether positive signs are more likely than negative signs and returns
623
- * `P(Binomial(nNonTies, 0.5) >= positive)`. `alternative = 'less'` treats
624
- * negative signs as successes instead. With a continuous difference
625
- * distribution this is the usual directional median test. Exact zero
626
- * differences are ties and do not enter the binomial denominator. All-tie and
627
- * empty inputs return p = 1. Every input difference must be finite, and the
628
- * direction must be chosen explicitly so a caller cannot select it after
629
- * seeing the signs.
630
- */
631
- declare function pairedSignTest(differences: readonly number[], alternative: SignTestAlternative): PairedSignTestResult;
632
- /** A binomial proportion estimate with a confidence interval. */
633
- interface ProportionInterval {
634
- /** Point estimate successes / n (0 when n = 0). */
635
- estimate: number;
636
- /** Lower bound, clamped to [0, 1]. */
637
- lower: number;
638
- /** Upper bound, clamped to [0, 1]. */
639
- upper: number;
640
- }
641
- /**
642
- * Wilson score interval for a binomial proportion. Correct at small n and near
643
- * 0/1, where the normal (Wald) approximation produces bounds outside [0, 1] and
644
- * understates coverage. Use this for any pass-rate / hit-rate / realness-rate
645
- * CI — the continuous `confidenceInterval` assumes the wrong distribution for a
646
- * proportion. `n = 0 ⇒ {0, 0, 0}`.
647
- */
648
- declare function wilson(successes: number, n: number, confidence?: number): ProportionInterval;
649
- /**
650
- * Are these per-item outcomes binary (every value exactly 0 or 1)?
651
- *
652
- * The discriminator a promotion gate needs before choosing a paired statistic.
653
- * On binary outcomes the paired delta vector lives in {-1, 0, +1} and is
654
- * normally dominated by zeros (both arms solve, or both arms miss, most items),
655
- * so its MEDIAN is pinned at exactly 0 no matter how large the real shift in
656
- * success rate is — and a bootstrap CI on that median collapses to [0, 0].
657
- * A gate keying on `ci.low > threshold` is then structurally unable to see
658
- * either a gain or a regression. Detect this shape and switch to the
659
- * paired-binary estimators ({@link mcnemar}, {@link pairedRiskDifference})
660
- * instead of silently answering "no" forever.
661
- *
662
- * Empty input is NOT binary: there is no evidence of the outcome's shape, and
663
- * defaulting an empty vector into the binary branch would pick a statistic on
664
- * no data at all.
665
- *
666
- * NOT the right discriminator for a gate. It recognises the literal {0, 1}
667
- * encoding and nothing else, so a pass/fail dimension emitted on 0-100 — which
668
- * judges in this codebase do routinely — reads as non-binary, and a single
669
- * partial-credit score in an otherwise pass/fail vector flips it to false while
670
- * leaving the median just as blind. Gates want {@link pairedBinaryScale} (any
671
- * two-point encoding). This predicate remains for callers that specifically
672
- * mean "literally 0/1".
673
- */
674
- declare function isBinaryOutcomeVector(values: ArrayLike<number>): boolean;
675
- /** Result of a McNemar paired-binary significance test. */
676
- interface McNemarResult {
677
- /** Total paired observations. */
678
- n: number;
679
- /** Discordant pairs (b + c) — the only ones that carry signal. */
680
- nDiscordant: number;
681
- /** Pairs where treatment succeeded and control failed ("newly correct"). */
682
- b: number;
683
- /** Pairs where control succeeded and treatment failed ("newly wrong"). */
684
- c: number;
685
- /** Continuity-corrected chi-square statistic (reference; exact p drives the call). */
686
- statistic: number;
687
- /** Two-sided p-value. Exact (binomial sign test on discordant pairs). */
688
- pValue: number;
689
- }
690
- /**
691
- * McNemar's test for paired binary outcomes — the correct significance test for
692
- * "does treatment change the success rate vs control on the SAME items". Only
693
- * discordant pairs (one arm right, the other wrong) carry information; concordant
694
- * pairs are uninformative, so a paired t-test / two-proportion z-test on the raw
695
- * rates is wrong here. The p-value is exact: under H0 the b "treatment-wins" are
696
- * Binomial(b + c, 0.5), so the two-sided p is the doubled binomial tail — correct
697
- * at the small discordant counts typical of eval runs (no continuity-corrected
698
- * chi-square approximation needed, though it is returned as `statistic` for
699
- * reference). Inputs are paired 0/1 (or boolean) arrays, control first to match
700
- * the module's (before, after) convention. Throws on unequal lengths.
701
- */
702
- declare function mcnemar(control: ArrayLike<number | boolean>, treatment: ArrayLike<number | boolean>): McNemarResult;
703
- /** A paired binary effect size (treatment rate − control rate) with a CI. */
704
- interface RiskDifferenceResult {
705
- /** Total paired observations. */
706
- n: number;
707
- /** Discordant pairs: treatment-win count. */
708
- b: number;
709
- /** Discordant pairs: control-win count. */
710
- c: number;
711
- /** Paired risk difference p(treatment) − p(control) = (b − c) / n. */
712
- riskDifference: number;
713
- /** Lower bound of the CI, clamped to [-1, 1]. */
714
- lower: number;
715
- /** Upper bound of the CI, clamped to [-1, 1]. */
716
- upper: number;
717
- /** Confidence level used. */
718
- confidence: number;
719
- }
720
- /**
721
- * Paired risk difference (the effect-size companion to {@link mcnemar}): the
722
- * change in success rate p(treatment) − p(control) on matched items, which for
723
- * paired binary data equals (b − c) / n. The CI uses the paired variance from
724
- * the discordant counts, not the independent-samples formula (which overstates
725
- * the interval by ignoring the pairing). Inputs are paired 0/1 (or boolean)
726
- * arrays, control first. Throws on unequal lengths.
727
- *
728
- * REPORTING ONLY — do NOT decide a promotion on this interval. The CI is a Wald
729
- * normal approximation, which badly UNDERCOVERS when only a handful of pairs are
730
- * discordant: at n = 3 with b = 2, c = 0 it returns [0.133, 1.000], excluding 0,
731
- * while McNemar's exact test on the same data gives p = 0.50. A gate keying on
732
- * `lower > 0` would promote noise. Use {@link pairedRiskDifferenceExact}, whose
733
- * interval is dual to the exact test by construction, for any decision.
734
- */
735
- declare function pairedRiskDifference(control: ArrayLike<number | boolean>, treatment: ArrayLike<number | boolean>, confidence?: number): RiskDifferenceResult;
736
- /** A paired binary effect size with an EXACT interval and the exact test that
737
- * bounds it — one object so a caller cannot read the estimate without the
738
- * significance it is entitled to. */
739
- interface ExactRiskDifferenceResult {
740
- /** Total paired observations. */
741
- n: number;
742
- /** Discordant pairs: treatment-win count. */
743
- b: number;
744
- /** Discordant pairs: control-win count. */
745
- c: number;
746
- /** Discordant pairs (b + c) — the only ones carrying information. */
747
- nDiscordant: number;
748
- /** Paired risk difference p(treatment) − p(control) = (b − c) / n. */
749
- riskDifference: number;
750
- /** Exact conditional CI lower bound. 0 when there are no discordant pairs. */
751
- lower: number;
752
- /** Exact conditional CI upper bound. 0 when there are no discordant pairs. */
753
- upper: number;
754
- /** Confidence level used. */
755
- confidence: number;
756
- /** McNemar's exact two-sided p-value on the same discordant counts. */
757
- pValue: number;
758
- }
759
- /**
760
- * Paired risk difference with the EXACT CONDITIONAL interval — the estimator a
761
- * promotion gate may decide on.
762
- *
763
- * Conditional on the number of discordant pairs m = b + c, the treatment-win
764
- * count b is Binomial(m, π) with π = P(treatment wins | discordant), and the
765
- * risk difference is an exact reparameterisation: RD = (2π − 1)·m/n. So a
766
- * Clopper-Pearson exact interval for π maps straight onto RD. This buys the
767
- * property the Wald interval in {@link pairedRiskDifference} does not have:
768
- *
769
- * **`lower > 0` ⟺ McNemar's exact test rejects at α = 1 − confidence.**
770
- *
771
- * Clopper-Pearson excludes π = 0.5 exactly when the two-sided exact binomial
772
- * test of π = 0.5 rejects, and that test IS {@link mcnemar}'s p-value — so the
773
- * interval and the test can never disagree, and a gate keyed on `lower` cannot
774
- * promote what the exact test refuses. The exact p is returned in the same
775
- * object so the two are impossible to compute apart.
776
- *
777
- * The interval is conservative (exact intervals over-cover; conditioning on m
778
- * discards the concordant pairs' information about m itself). That is the
779
- * correct direction for a promotion gate: it refuses more often, never less.
780
- *
781
- * With m = 0 there are no discordant pairs and π is not identified: the result
782
- * is the degenerate [0, 0] with p = 1. That is NOT evidence of equivalence —
783
- * callers must treat a zero-width interval as "cannot decide", not as "no
784
- * difference". Inputs are paired 0/1 (or boolean) arrays, control first.
785
- * Throws on unequal lengths.
786
- */
787
- declare function pairedRiskDifferenceExact(control: ArrayLike<number | boolean>, treatment: ArrayLike<number | boolean>, confidence?: number): ExactRiskDifferenceResult;
788
- /** A paired binary effect size with an interval that is valid at a NONZERO
789
- * margin — the estimator a noninferiority decision may be made on. */
790
- interface ScoreRiskDifferenceResult {
791
- /** Total paired observations. */
792
- n: number;
793
- /** Discordant pairs: treatment-win count. */
794
- b: number;
795
- /** Discordant pairs: control-win count. */
796
- c: number;
797
- /** Discordant pairs (b + c). */
798
- nDiscordant: number;
799
- /** Paired risk difference p(treatment) − p(control) = (b − c) / n. */
800
- riskDifference: number;
801
- /** Score-interval lower bound on the population risk difference. */
802
- lower: number;
803
- /** Score-interval upper bound on the population risk difference. */
804
- upper: number;
805
- /** Confidence level used. */
806
- confidence: number;
807
- }
808
- /**
809
- * Paired risk difference with TANGO'S (1998) SCORE INTERVAL — the estimator a
810
- * promotion gate may decide on **at a nonzero margin**.
811
- *
812
- * {@link pairedRiskDifferenceExact} conditions on the observed discordant count
813
- * `m = b + c`, builds a Clopper-Pearson interval for the win share among those
814
- * `m` pairs, and multiplies by the observed `m/n`. That is exact for testing
815
- * RD = 0 — it is dual to McNemar — but it is NOT a confidence interval for the
816
- * population risk difference at a nonzero margin, because the sampling
817
- * variability of `m/n` itself is discarded. The gap is not academic: with the
818
- * production caller's `pairedDeltaThreshold: -0.05`, a process whose true risk
819
- * difference sits exactly on that margin clears a nominal-95 % `lower > margin`
820
- * check 24.75 % of the time at n = 40 and 43.95 % at n = 76 (2000 replicates
821
- * each) when the conditional interval decides.
822
- *
823
- * Tango's interval inverts the score test of RD = delta, which estimates the
824
- * nuisance loss rate under each hypothesised delta instead of fixing it at the
825
- * observed value, so `m` contributes its own uncertainty. It is the method
826
- * `ratesci::scorepairci` uses for paired risk-difference noninferiority, and it
827
- * is not conditional, so it stays valid as the margin moves away from zero.
828
- *
829
- * The bounds are found by bisecting `tangoScore(delta) = ±z` — the score is
830
- * monotone decreasing in delta, so each crossing is unique. Inputs are paired
831
- * 0/1 (or boolean) arrays, control first. Throws on unequal lengths.
832
- */
833
- declare function pairedRiskDifferenceScore(control: ArrayLike<number | boolean>, treatment: ArrayLike<number | boolean>, confidence?: number): ScoreRiskDifferenceResult;
834
- /**
835
- * The common positive level `s` such that EVERY value across both paired arms is
836
- * exactly 0 or `s` — i.e. the outcome is pass/fail, whatever encoding it arrived
837
- * in. Returns null when the outcomes are not two-point, when the two arms use
838
- * different levels, or when no positive value was observed at all (all-zero
839
- * arms: the level is not identified, and there is nothing to decide anyway).
840
- *
841
- * This is the scale-aware successor to {@link isBinaryOutcomeVector}, which only
842
- * recognises literal {0, 1}. Judges in this codebase emit dimensions on 0-100 as
843
- * well as [0,1] (see `detectScale` in `campaign/gates/statistical-heldout.ts`),
844
- * so a pass/fail dimension routinely arrives as {0, 100} and a {0,1}-only test
845
- * silently sends it down the median path that cannot see it. Any positive level
846
- * is accepted, not just 1 and 100: for a two-point {0, s} outcome the mean paired
847
- * delta is exactly s·(b − c)/n, so the binary estimators apply after dividing by
848
- * s and rescaling the result back into the caller's native units.
849
- *
850
- * Non-finite values ⇒ null: an unusable outcome must not be classified as a
851
- * clean pass/fail shape.
852
- */
853
- declare function pairedBinaryScale(before: ArrayLike<number>, after: ArrayLike<number>): number | null;
854
- /** Fraction of paired observations whose delta is an exact tie (|after − before|
855
- * < 1e-9). Throws on unequal sample sizes; 0 pairs ⇒ 0. */
856
- declare function pairedDeltaTieFraction(before: ArrayLike<number>, after: ArrayLike<number>): number;
857
- /**
858
- * The paired-delta statistic a DECISION is computed on, package-wide.
859
- *
860
- * The mean paired delta is the estimator that answers the question a promotion
861
- * gate asks — "by how much did the candidate move the score" — in the caller's
862
- * own units, and it equals the aggregate lift everyone quotes. The MEDIAN
863
- * answers a different question and loses the answer to this one in every regime
864
- * eval data actually lands in:
865
- * - TWO-POINT (pass/fail) outcomes on any encoding: the delta vector lives in
866
- * {−s, 0, +s} dominated by zeros, so the median and its whole bootstrap CI
867
- * are pinned at exactly 0 however large the shift. (Decide these on
868
- * {@link pairedRiskDifferenceExact} instead — same estimand, exact interval.)
869
- * - TIE-DOMINATED outcomes: at half the pairs tied the sample median is 0 by
870
- * construction, and `ci.low > threshold` then answers "no" forever at a
871
- * non-negative threshold and "yes" forever at a negative one.
872
- * - LOW-CARDINALITY outcomes, even well below half ties: judge dimensions on
873
- * integer 0-100, and block scores like {⅔, 1} from averaging pass/fail
874
- * leaves, put the median on a coarse lattice whose bootstrap percentiles
875
- * land on atoms. Measured: 26 blocks of 3 pass/fail leaves carrying a real
876
- * +12.8pp lift, only 23% of pairs tied, gives a median CI of [0, 0.333] —
877
- * lower bound exactly 0, so a gate at threshold 0 refuses a real lift.
878
- * That last case is why there is no tie-fraction threshold here: any cutoff on
879
- * ties leaves the lattice case open on the other side of it.
880
- *
881
- * `heldoutSignificance` has defaulted to the mean since #316 for the same
882
- * reason. The median remains available per call site for callers who
883
- * specifically want outlier robustness and accept the blindness.
884
- */
885
- declare const DECISION_PAIRED_DELTA_STATISTIC: 'mean';
886
- /**
887
- * Unbiased pass@k for code generation (Chen et al. 2021, "Evaluating Large
888
- * Language Models Trained on Code"). Given `n` independent samples for one
889
- * problem of which `c` pass, the probability that at least one of a random k of
890
- * them passes is 1 − C(n−c, k) / C(n, k). Estimating pass@k as "did any of the
891
- * first k pass" is biased high at small n; this is the variance-reduced estimator
892
- * averaged implicitly over all k-subsets. Average the per-problem values across
893
- * the suite for the corpus pass@k. Computed in the numerically stable product
894
- * form. Requires 1 ≤ k ≤ n and 0 ≤ c ≤ n.
895
- */
896
- declare function passAtK(n: number, c: number, k: number): number;
897
- interface EProcessOptions {
898
- /** Type-I error budget. The process decides when wealth ≥ 1/alpha
899
- * (Ville's inequality). Default 0.05. */
900
- alpha?: number;
901
- /** Truncation bound on the predictable bet λ ∈ [0, maxBet]. Must satisfy
902
- * maxBet < 1/nullMean so every wealth factor stays strictly positive.
903
- * Default 0.5. */
904
- maxBet?: number;
905
- /** The null boundary m₀ for H0: E[x] ≤ m₀ on x ∈ [0,1]. Default 0.5
906
- * (the paired-delta encoding x = (d+1)/2 maps "no effect" to 1/2).
907
- * A pre-registered minEffect shifts this — see `sequentialPairedGate`. */
908
- nullMean?: number;
909
- }
910
- interface EProcessStep {
911
- /** Current wealth W_n — the e-value against H0 after n observations. */
912
- wealth: number;
913
- /** Observations consumed so far. */
914
- n: number;
915
- /** True from the first n where W_n ≥ 1/alpha onward (sticky). */
916
- decided: boolean;
917
- }
918
- interface EProcessState extends EProcessStep {
919
- alpha: number;
920
- maxBet: number;
921
- nullMean: number;
922
- /** The decision boundary 1/alpha. */
923
- threshold: number;
924
- /** Observation count at the first threshold crossing; undefined until decided. */
925
- decidedAtN?: number;
926
- }
927
- interface EProcess {
928
- /** Consume one observation x ∈ [0,1]. Throws on non-finite / out-of-range
929
- * input — a silent clamp would corrupt the type-I guarantee. */
930
- update(x: number): EProcessStep;
931
- state(): EProcessState;
932
- }
933
- /**
934
- * Betting test-martingale for bounded observations — the e-process core of
935
- * anytime-valid sequential testing (Waudby-Smith & Ramdas, "Estimating means
936
- * of bounded random variables by betting", JRSS-B 2024).
937
- *
938
- * Observations x_i ∈ [0,1]; H0: E[x] ≤ m₀ (`nullMean`, default 1/2). Wealth
939
- *
940
- * W_t = Π_{i≤t} (1 + λ_i (x_i − m₀)), W_0 = 1
941
- *
942
- * with the truncated GROW-style plug-in bet computed from PRIOR observations:
943
- *
944
- * λ_i = clamp((μ̂_{i−1} − m₀) / (σ̂²_{i−1} + (μ̂_{i−1} − m₀)²), 0, maxBet)
945
- *
946
- * where μ̂/σ̂² are the shrunk running estimates μ̂_t = (1/2 + Σx_i)/(t+1),
947
- * σ̂²_t = (1/4 + Σ(x_i − μ̂_i)²)/(t+1).
948
- *
949
- * PREDICTABILITY INVARIANT (load-bearing): λ_i is a function of x_1..x_{i−1}
950
- * ONLY — it may never see x_i. With λ_i ≥ 0 predictable, each factor has
951
- * E[1 + λ_i(x_i − m₀) | past] ≤ 1 under H0, so W is a nonnegative
952
- * supermartingale and Ville's inequality gives P(∃t: W_t ≥ 1/α) ≤ α — the
953
- * type-I guarantee holds at ANY data-dependent stopping time. λ_1 is always 0
954
- * (no prior evidence), so the first observation never moves wealth.
955
- *
956
- * `decided` latches at the first crossing W_t ≥ 1/α and never un-latches;
957
- * wealth keeps updating after the crossing (the e-process remains valid), but
958
- * the decision time is the first crossing.
959
- */
960
- declare function eProcess(opts?: EProcessOptions): EProcess;
961
- /** Tiny seedable PRNG (mulberry32) — deterministic resampling/shuffling, not
962
- * cryptographic. Exported so e-process shuffles and bootstrap resampling
963
- * share ONE PRNG implementation. Every distinct 32-bit seed gives a distinct
964
- * stream, including 0. */
965
- declare function mulberry32(seed: number): () => number;
966
- //#endregion
967
- export { pairedCohensDz as $, WeightedCompositeInput as A, positionalBias as At, eProcess as B, RankTestMethod as C, GoldenItem as Ct, ScoreRiskDifferenceResult as D, calibrateJudge as Dt, RiskDifferenceResult as E, VerbosityBiasResult as Et, cliffsDelta as F, mannWhitneyU as G, interRaterReliability as H, cohensD as I, mcnemarRequiredN as J, mcnemar as K, confidenceInterval as L, WilcoxonSignedRankResult as M, verbosityBias as Mt, benjaminiHochberg as N, SignTestAlternative as O, calibrateJudgeContinuous as Ot, bonferroni as P, pairedBootstrap as Q, corpusInterRaterAgreement as R, ProportionInterval as S, ContinuousCalibrationResult as St, RankTestOptions as T, SelfPreferenceResult as Tt, interpretCliffs as U, holm as V, isBinaryOutcomeVector as W, normalizeScores as X, mulberry32 as Y, pairedBinaryScale as Z, McNemarResult as _, wilson as _t, CorpusAgreementReport as a, pairedSignTest as at, PairedSignTestResult as b, ContinuousAgreement as bt, DEFAULT_PERMUTATIONS as c, passAtK as ct, EProcessState as d, requiredPairedSampleSize as dt, pairedDeltaTieFraction as et, EProcessStep as f, requiredSampleSize as ft, MannWhitneyResult as g, wilcoxonSignedRank as gt, MANN_WHITNEY_EXACT_MAX_WORK as h, weightedMean as ht, CorpusAgreementPerDimension as i, pairedRiskDifferenceScore as it, WeightedCompositeResult as j, selfPreference as jt, WILCOXON_EXACT_MAX_N as k, continuousAgreement as kt, EProcess as l, pearsonR as lt, MANN_WHITNEY_EXACT_MAX_STATES as m, weightedComposite as mt, CliffsMagnitude as n, pairedRiskDifference as nt, CorpusScoreRecord as o, pairedTTest as ot, ExactRiskDifferenceResult as p, spearmanR as pt, mcnemarPower as q, CorpusAgreementOptions as r, pairedRiskDifferenceExact as rt, DECISION_PAIRED_DELTA_STATISTIC as s, partialCredit as st, BOOTSTRAP_GATE_MIN_N as t, pairedMde as tt, EProcessOptions as u, ranks as ut, PairedBootstrapOptions as v, CalibrationResult as vt, RankTestMethodRequest as w, PositionalBiasResult as wt, PairedTTestResult as x, ContinuousAgreementOptions as xt, PairedBootstrapResult as y, CandidateScore as yt, corpusInterRaterAgreementFromJudgeScores as z };
968
- //# sourceMappingURL=statistics-D6Uebe_4.d.ts.map