@tangle-network/agent-eval 0.144.11 → 0.144.13

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (378) hide show
  1. package/CHANGELOG.md +11 -0
  2. package/dist/agent-profile-B9_GGsG8.d.ts +84 -0
  3. package/dist/agent-profile-B9_GGsG8.d.ts.map +1 -0
  4. package/dist/{agent-profile-cell-BOP-iA9Q.d.ts → agent-profile-cell-BkcRDikH.d.ts} +2 -2
  5. package/dist/{agent-profile-cell-BOP-iA9Q.d.ts.map → agent-profile-cell-BkcRDikH.d.ts.map} +1 -1
  6. package/dist/analyst/index.d.ts +134 -16
  7. package/dist/analyst/index.d.ts.map +1 -1
  8. package/dist/analyst/index.js +364 -10
  9. package/dist/analyst/index.js.map +1 -1
  10. package/dist/backend-integrity-CsVin_Wb.d.ts +280 -0
  11. package/dist/backend-integrity-CsVin_Wb.d.ts.map +1 -0
  12. package/dist/{baseline-C-GocmIW.js → baseline-BhPRQBVn.js} +3 -2
  13. package/dist/{baseline-C-GocmIW.js.map → baseline-BhPRQBVn.js.map} +1 -1
  14. package/dist/{benchmark-CWeqGl7x.js → benchmark-BhT16ep9.js} +2 -2
  15. package/dist/{benchmark-CWeqGl7x.js.map → benchmark-BhT16ep9.js.map} +1 -1
  16. package/dist/{agent-profile-DPi7IZg7.d.ts → benchmark-Ceoan7vk.d.ts} +4 -91
  17. package/dist/benchmark-Ceoan7vk.d.ts.map +1 -0
  18. package/dist/{benchmark-command-BteMFN62.js → benchmark-command-CY6cbhOB.js} +36 -79
  19. package/dist/benchmark-command-CY6cbhOB.js.map +1 -0
  20. package/dist/benchmarks/index.d.ts +244 -2
  21. package/dist/benchmarks/index.d.ts.map +1 -0
  22. package/dist/benchmarks/index.js +733 -1
  23. package/dist/benchmarks/index.js.map +1 -0
  24. package/dist/builder-eval/index.d.ts +23 -2
  25. package/dist/builder-eval/index.d.ts.map +1 -1
  26. package/dist/builder-eval/index.js +227 -3
  27. package/dist/builder-eval/index.js.map +1 -1
  28. package/dist/campaign/index.d.ts +10 -8
  29. package/dist/campaign/index.js +9 -6
  30. package/dist/{campaign-C2TTzQII.js → campaign-BYjBAypg.js} +21 -634
  31. package/dist/campaign-BYjBAypg.js.map +1 -0
  32. package/dist/{canonical-D011XM8r.js → canonical-D-XsTQ6_.js} +2 -2
  33. package/dist/{canonical-D011XM8r.js.map → canonical-D-XsTQ6_.js.map} +1 -1
  34. package/dist/{index-CQTZ-4XN.d.ts → capture-fetch-DDvpjVRU.d.ts} +3 -3
  35. package/dist/capture-fetch-DDvpjVRU.d.ts.map +1 -0
  36. package/dist/{default-registry-BmktKy8r.js → chat-client-Bp4Ebuuc.js} +1276 -1276
  37. package/dist/chat-client-Bp4Ebuuc.js.map +1 -0
  38. package/dist/cli.js +2 -2
  39. package/dist/{client-C9gzZE59.d.ts → client-KUbGulm_.d.ts} +4 -4
  40. package/dist/{client-C9gzZE59.d.ts.map → client-KUbGulm_.d.ts.map} +1 -1
  41. package/dist/contract/index.d.ts +13 -390
  42. package/dist/contract/index.d.ts.map +1 -1
  43. package/dist/contract/index.js +18 -542
  44. package/dist/contract/index.js.map +1 -1
  45. package/dist/{cost-ledger-DMFxsLKr.js → cost-ledger-BSe92yAV.js} +3 -3
  46. package/dist/{cost-ledger-DMFxsLKr.js.map → cost-ledger-BSe92yAV.js.map} +1 -1
  47. package/dist/{cost-ledger-Bv_e8XHY.d.ts → cost-ledger-DbQdN3nO.d.ts} +2 -2
  48. package/dist/{cost-ledger-Bv_e8XHY.d.ts.map → cost-ledger-DbQdN3nO.d.ts.map} +1 -1
  49. package/dist/{counterfactual-CxmxAONP.d.ts → counterfactual--bpysZF0.d.ts} +2 -14
  50. package/dist/{counterfactual-CxmxAONP.d.ts.map → counterfactual--bpysZF0.d.ts.map} +1 -1
  51. package/dist/{counterfactual-CWPTrMH7.js → counterfactual-lDfCx0Uz.js} +3 -31
  52. package/dist/{counterfactual-CWPTrMH7.js.map → counterfactual-lDfCx0Uz.js.map} +1 -1
  53. package/dist/{dataset-C8xaLXdY.d.ts → dataset-CJjKqQfA.d.ts} +2 -8
  54. package/dist/dataset-CJjKqQfA.d.ts.map +1 -0
  55. package/dist/{default-registry-Bf8Woqmq.d.ts → default-registry-Di6HP6pG.d.ts} +7 -12
  56. package/dist/default-registry-Di6HP6pG.d.ts.map +1 -0
  57. package/dist/{analyze-runs-C30yljDJ.js → define-agent-eval-iqjT--ZZ.js} +542 -130
  58. package/dist/define-agent-eval-iqjT--ZZ.js.map +1 -0
  59. package/dist/define-agent-eval-sH24zBfM.d.ts +388 -0
  60. package/dist/define-agent-eval-sH24zBfM.d.ts.map +1 -0
  61. package/dist/descriptive-B5MwKfbf.js +144 -0
  62. package/dist/descriptive-B5MwKfbf.js.map +1 -0
  63. package/dist/{dspy-rlm-engine-DbTk4JdR.js → dspy-rlm-engine-B_qhSc21.js} +4 -4
  64. package/dist/{dspy-rlm-engine-DbTk4JdR.js.map → dspy-rlm-engine-B_qhSc21.js.map} +1 -1
  65. package/dist/effect-sizes-DiH8MGOH.js +82 -0
  66. package/dist/effect-sizes-DiH8MGOH.js.map +1 -0
  67. package/dist/{engine-3hL-XqwJ.d.ts → engine-otFpE2gF.d.ts} +10 -38
  68. package/dist/engine-otFpE2gF.d.ts.map +1 -0
  69. package/dist/{errors-CKPfb2aH.d.ts → errors-DEE6u6ot.d.ts} +2 -14
  70. package/dist/{errors-CKPfb2aH.d.ts.map → errors-DEE6u6ot.d.ts.map} +1 -1
  71. package/dist/{errors-D-LKuDhb.js → errors-Dngq5h35.js} +2 -8
  72. package/dist/{errors-D-LKuDhb.js.map → errors-Dngq5h35.js.map} +1 -1
  73. package/dist/{eval-campaign-B_7wcnav.js → eval-campaign-UB-usSQ2.js} +6 -6
  74. package/dist/{eval-campaign-B_7wcnav.js.map → eval-campaign-UB-usSQ2.js.map} +1 -1
  75. package/dist/{exact-types-DSFFpLLI.d.ts → exact-types-BH1twmAJ.d.ts} +2 -2
  76. package/dist/{exact-types-DSFFpLLI.d.ts.map → exact-types-BH1twmAJ.d.ts.map} +1 -1
  77. package/dist/experiment/index.d.ts +9 -6
  78. package/dist/experiment/index.d.ts.map +1 -1
  79. package/dist/experiment/index.js +11 -7
  80. package/dist/experiment/index.js.map +1 -1
  81. package/dist/experiment-tracker-C29gXM4B.js +269 -0
  82. package/dist/experiment-tracker-C29gXM4B.js.map +1 -0
  83. package/dist/{experiment-tracker-IMntXr6J.d.ts → experiment-tracker-DWHZBAYL.d.ts} +77 -121
  84. package/dist/experiment-tracker-DWHZBAYL.d.ts.map +1 -0
  85. package/dist/{external-optimizer-contracts-CZuJNcT5.d.ts → external-optimizer-contracts-lixrOZdX.d.ts} +3 -3
  86. package/dist/{external-optimizer-contracts-CZuJNcT5.d.ts.map → external-optimizer-contracts-lixrOZdX.d.ts.map} +1 -1
  87. package/dist/external-optimizer-process-BTiNB-RH.js +301 -0
  88. package/dist/external-optimizer-process-BTiNB-RH.js.map +1 -0
  89. package/dist/{single-run-lock-DFWHEB09.js → external-optimizer-subprocess-DrJ9hR8u.js} +144 -438
  90. package/dist/external-optimizer-subprocess-DrJ9hR8u.js.map +1 -0
  91. package/dist/extract-usage-BrQ8mCLX.js +155 -0
  92. package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
  93. package/dist/{failure-cluster-CqcvCcdR.d.ts → failure-cluster-BLURuWG4.d.ts} +2 -3
  94. package/dist/failure-cluster-BLURuWG4.d.ts.map +1 -0
  95. package/dist/{feedback-trajectory-CacHpxsp.d.ts → feedback-trajectory-juOozjAc.d.ts} +5 -36
  96. package/dist/feedback-trajectory-juOozjAc.d.ts.map +1 -0
  97. package/dist/fuzz.d.ts +2 -2
  98. package/dist/fuzz.js +2 -2
  99. package/dist/hosted/index.d.ts +3 -3
  100. package/dist/{index-CvSN3IG1.d.ts → index-B8Ui1mr1.d.ts} +2 -2
  101. package/dist/{index-CvSN3IG1.d.ts.map → index-B8Ui1mr1.d.ts.map} +1 -1
  102. package/dist/{skill-usage-CO9OLRBx.d.ts → index-BWDrSVfw.d.ts} +11 -136
  103. package/dist/index-BWDrSVfw.d.ts.map +1 -0
  104. package/dist/index-Ba3YrbAL.d.ts +1 -0
  105. package/dist/{index-BnEuDAK2.d.ts → index-COQYtuRF.d.ts} +3 -3
  106. package/dist/{index-BnEuDAK2.d.ts.map → index-COQYtuRF.d.ts.map} +1 -1
  107. package/dist/{index-C3ssXVLv.d.ts → index-CvjYbU0D.d.ts} +2 -2
  108. package/dist/{index-C3ssXVLv.d.ts.map → index-CvjYbU0D.d.ts.map} +1 -1
  109. package/dist/{index-DPPGNJ_R.d.ts → index-DSmEylT9.d.ts} +17 -115
  110. package/dist/index-DSmEylT9.d.ts.map +1 -0
  111. package/dist/index.d.ts +2397 -5308
  112. package/dist/index.d.ts.map +1 -1
  113. package/dist/index.js +5914 -10496
  114. package/dist/index.js.map +1 -1
  115. package/dist/{insight-report-CgX_s0Ez.d.ts → insight-report-CRi-Ufrj.d.ts} +4 -4
  116. package/dist/{insight-report-CgX_s0Ez.d.ts.map → insight-report-CRi-Ufrj.d.ts.map} +1 -1
  117. package/dist/{integrity-MLzHOfV9.js → integrity-Cy9WHAtb.js} +2 -2
  118. package/dist/{integrity-MLzHOfV9.js.map → integrity-Cy9WHAtb.js.map} +1 -1
  119. package/dist/{integrity-DY6tIbl0.js → integrity-DysDBWDu.js} +2 -2
  120. package/dist/{integrity-DY6tIbl0.js.map → integrity-DysDBWDu.js.map} +1 -1
  121. package/dist/{integrity-BuqEKu-x.d.ts → integrity-OrcI9Nau.d.ts} +3 -3
  122. package/dist/{integrity-BuqEKu-x.d.ts.map → integrity-OrcI9Nau.d.ts.map} +1 -1
  123. package/dist/internal-BDHPCnjk.js +230 -0
  124. package/dist/internal-BDHPCnjk.js.map +1 -0
  125. package/dist/judge-calibration-C5CbMYce.d.ts +117 -0
  126. package/dist/judge-calibration-C5CbMYce.d.ts.map +1 -0
  127. package/dist/judge-calibration-DZkWrm5H.js +317 -0
  128. package/dist/judge-calibration-DZkWrm5H.js.map +1 -0
  129. package/dist/{kind-factory-B8-r8-y8.js → kind-factory-CPmSd58s.js} +208 -208
  130. package/dist/{kind-factory-B8-r8-y8.js.map → kind-factory-CPmSd58s.js.map} +1 -1
  131. package/dist/ledger-core/index.d.ts +1 -1
  132. package/dist/ledger-core/index.js +2 -2
  133. package/dist/{ledger-core-DXZIqu17.js → ledger-core-BmZt19oQ.js} +110 -110
  134. package/dist/{ledger-core-DXZIqu17.js.map → ledger-core-BmZt19oQ.js.map} +1 -1
  135. package/dist/{llm-client-DzvMUsS_.js → llm-client-d0-2TT1g.js} +4 -64
  136. package/dist/{llm-client-DzvMUsS_.js.map → llm-client-d0-2TT1g.js.map} +1 -1
  137. package/dist/{skillopt-optimization-method-CQdVeM8k.js → llm-judge-dZ8P6nGI.js} +3349 -5401
  138. package/dist/llm-judge-dZ8P6nGI.js.map +1 -0
  139. package/dist/matrix/index.d.ts +2 -2
  140. package/dist/meta-eval/index.d.ts +3 -3
  141. package/dist/meta-eval/index.js +3 -3
  142. package/dist/{metrics-C9YY1OcL.js → metrics-Cl0L1KUy.js} +2 -108
  143. package/dist/{metrics-C9YY1OcL.js.map → metrics-Cl0L1KUy.js.map} +1 -1
  144. package/dist/{mint-B2O60ACG.js → mint-Dj9Ww_3I.js} +4 -4
  145. package/dist/{mint-B2O60ACG.js.map → mint-Dj9Ww_3I.js.map} +1 -1
  146. package/dist/{multi-layer-verifier-DnAqwl0h.d.ts → multi-layer-verifier-DIguZc8Z.d.ts} +3 -3
  147. package/dist/{multi-layer-verifier-DnAqwl0h.d.ts.map → multi-layer-verifier-DIguZc8Z.d.ts.map} +1 -1
  148. package/dist/multiplicity-DIWHvysC.d.ts +43 -0
  149. package/dist/multiplicity-DIWHvysC.d.ts.map +1 -0
  150. package/dist/multishot/index.d.ts +3 -3
  151. package/dist/multishot/index.js +1 -1
  152. package/dist/openapi.json +1 -1
  153. package/dist/{opencode-sqlite-8r6WUfHc.js → opencode-sqlite-DJWAXLms.js} +2 -2
  154. package/dist/{opencode-sqlite-8r6WUfHc.js.map → opencode-sqlite-DJWAXLms.js.map} +1 -1
  155. package/dist/package-version-D7lQHt_-.js +34 -0
  156. package/dist/package-version-D7lQHt_-.js.map +1 -0
  157. package/dist/paired-arms-D-XRF_fy.js +1045 -0
  158. package/dist/paired-arms-D-XRF_fy.js.map +1 -0
  159. package/dist/paired-promotion-decision-CGzg0cI_.d.ts +251 -0
  160. package/dist/paired-promotion-decision-CGzg0cI_.d.ts.map +1 -0
  161. package/dist/paired-tests-BHIhYVdu.js +213 -0
  162. package/dist/paired-tests-BHIhYVdu.js.map +1 -0
  163. package/dist/pareto-BqNW3LJR.d.ts +117 -0
  164. package/dist/pareto-BqNW3LJR.d.ts.map +1 -0
  165. package/dist/pipelines/index.d.ts +3 -64
  166. package/dist/pipelines/index.d.ts.map +1 -1
  167. package/dist/pipelines/index.js +4 -284
  168. package/dist/pipelines/index.js.map +1 -1
  169. package/dist/power-and-mde-CHIrXJll.js +195 -0
  170. package/dist/power-and-mde-CHIrXJll.js.map +1 -0
  171. package/dist/{promotion-policy-CrLrmys8.js → power-preflight-DEw-uC7q.js} +4 -184
  172. package/dist/power-preflight-DEw-uC7q.js.map +1 -0
  173. package/dist/pre-registration-CZwSFQS4.d.ts +577 -0
  174. package/dist/pre-registration-CZwSFQS4.d.ts.map +1 -0
  175. package/dist/{prime-protocol-BfSalTfR.js → prime-protocol-6tZTVsWm.js} +72 -21
  176. package/dist/prime-protocol-6tZTVsWm.js.map +1 -0
  177. package/dist/produced-state-DU79a81m.js +586 -0
  178. package/dist/produced-state-DU79a81m.js.map +1 -0
  179. package/dist/profile-cell.d.ts +1 -1
  180. package/dist/profile-cell.js +1 -1
  181. package/dist/promotion-policy-D0nPhkSy.d.ts +134 -0
  182. package/dist/promotion-policy-D0nPhkSy.d.ts.map +1 -0
  183. package/dist/promotion-policy-xzA40Evo.js +186 -0
  184. package/dist/promotion-policy-xzA40Evo.js.map +1 -0
  185. package/dist/{skillopt-optimization-method-USDKhxSA.d.ts → provenance-CCdxgLDT.d.ts} +57 -478
  186. package/dist/provenance-CCdxgLDT.d.ts.map +1 -0
  187. package/dist/registry-oJeeI4-a.d.ts +178 -0
  188. package/dist/registry-oJeeI4-a.d.ts.map +1 -0
  189. package/dist/{release-report-DA2BCu5a.d.ts → release-confidence-BFRE5WSp.d.ts} +180 -112
  190. package/dist/release-confidence-BFRE5WSp.d.ts.map +1 -0
  191. package/dist/{release-report-BUYmoKo2.js → release-confidence-CxDuiAev.js} +125 -217
  192. package/dist/release-confidence-CxDuiAev.js.map +1 -0
  193. package/dist/reporting.d.ts +6 -5
  194. package/dist/reporting.js +7 -5
  195. package/dist/{researcher-BchpD55R.d.ts → researcher-DaN4GST-.d.ts} +7 -40
  196. package/dist/{researcher-BchpD55R.d.ts.map → researcher-DaN4GST-.d.ts.map} +1 -1
  197. package/dist/{reward-hacking-BDToousL.js → reward-hacking-DNgjilrV.js} +3 -3
  198. package/dist/reward-hacking-DNgjilrV.js.map +1 -0
  199. package/dist/{reward-hacking-BI0OMAlo.d.ts → reward-hacking-DSSTuI9r.d.ts} +5 -5
  200. package/dist/{reward-hacking-BI0OMAlo.d.ts.map → reward-hacking-DSSTuI9r.d.ts.map} +1 -1
  201. package/dist/rl.d.ts +7 -7
  202. package/dist/rl.js +11 -10
  203. package/dist/rl.js.map +1 -1
  204. package/dist/rollout/index.d.ts +2 -2
  205. package/dist/rollout/index.js +4 -4
  206. package/dist/{rollout-DRrksrcV.js → rollout-BWtw0I_6.js} +3 -3
  207. package/dist/{rollout-DRrksrcV.js.map → rollout-BWtw0I_6.js.map} +1 -1
  208. package/dist/{rubric-predictive-validity-DvkjPYCe.d.ts → rubric-predictive-validity-BgxtKe4G.d.ts} +2 -2
  209. package/dist/{rubric-predictive-validity-DvkjPYCe.d.ts.map → rubric-predictive-validity-BgxtKe4G.d.ts.map} +1 -1
  210. package/dist/{rubric-predictive-validity-BRR632r1.js → rubric-predictive-validity-Cwwyd7ah.js} +2 -2
  211. package/dist/{rubric-predictive-validity-BRR632r1.js.map → rubric-predictive-validity-Cwwyd7ah.js.map} +1 -1
  212. package/dist/{run-record-BmSPWXJR.js → run-record-BvHPVS-i.js} +3 -3
  213. package/dist/{run-record-BmSPWXJR.js.map → run-record-BvHPVS-i.js.map} +1 -1
  214. package/dist/{run-record-CF4Dwpxr.d.ts → run-record-CKiihE6f.d.ts} +4 -4
  215. package/dist/{run-record-CF4Dwpxr.d.ts.map → run-record-CKiihE6f.d.ts.map} +1 -1
  216. package/dist/{proposal-findings-2GIUo1et.js → run-score-lDzV0X8j.js} +31 -31
  217. package/dist/run-score-lDzV0X8j.js.map +1 -0
  218. package/dist/sandbox-harness-BlSOu4LX.d.ts +70 -0
  219. package/dist/sandbox-harness-BlSOu4LX.d.ts.map +1 -0
  220. package/dist/{schema-Cef2cFmb2.d.ts → schema-Cef2cFmb.d.ts} +1 -1
  221. package/dist/schema-Cef2cFmb.d.ts.map +1 -0
  222. package/dist/semantic-concept-judge-BiJxScqe.js +406 -0
  223. package/dist/semantic-concept-judge-BiJxScqe.js.map +1 -0
  224. package/dist/{sequential-D-BLJBKU.js → sequential-C458DXNf.js} +4 -3
  225. package/dist/{sequential-D-BLJBKU.js.map → sequential-C458DXNf.js.map} +1 -1
  226. package/dist/sequential-eprocess-CbUt2htw.js +83 -0
  227. package/dist/sequential-eprocess-CbUt2htw.js.map +1 -0
  228. package/dist/series-convergence-D1cL1f-4.d.ts +129 -0
  229. package/dist/series-convergence-D1cL1f-4.d.ts.map +1 -0
  230. package/dist/{server-iu0ede49.js → server-ulsOdrTI.js} +5 -21
  231. package/dist/server-ulsOdrTI.js.map +1 -0
  232. package/dist/skillopt-optimization-method-C5cotF4E.d.ts +427 -0
  233. package/dist/skillopt-optimization-method-C5cotF4E.d.ts.map +1 -0
  234. package/dist/skillopt-optimization-method-jjdnc3YK.js +1999 -0
  235. package/dist/skillopt-optimization-method-jjdnc3YK.js.map +1 -0
  236. package/dist/{statistical-heldout-TQ-4CYiN.d.ts → statistical-heldout-DoFF5KJX.d.ts} +5 -278
  237. package/dist/statistical-heldout-DoFF5KJX.d.ts.map +1 -0
  238. package/dist/{store-otlp-Dw8PPIlL.js → store-otlp-CsptLYpN.js} +3 -3
  239. package/dist/{store-otlp-Dw8PPIlL.js.map → store-otlp-CsptLYpN.js.map} +1 -1
  240. package/dist/store-tool-spans-Cq9mFd-q.js +667 -0
  241. package/dist/store-tool-spans-Cq9mFd-q.js.map +1 -0
  242. package/dist/store-tool-spans-DbDLOBOb.d.ts +342 -0
  243. package/dist/store-tool-spans-DbDLOBOb.d.ts.map +1 -0
  244. package/dist/student-t-CvBq2mve.js +38 -0
  245. package/dist/student-t-CvBq2mve.js.map +1 -0
  246. package/dist/{summary-report-Lf-5I7xh.js → summary-report-Blysd6Z2.js} +6 -3
  247. package/dist/{summary-report-Lf-5I7xh.js.map → summary-report-Blysd6Z2.js.map} +1 -1
  248. package/dist/{summary-report-BNR7DWTj.d.ts → summary-report-CaL-Hnxt.d.ts} +5 -5
  249. package/dist/{summary-report-BNR7DWTj.d.ts.map → summary-report-CaL-Hnxt.d.ts.map} +1 -1
  250. package/dist/supervisor-run/index.d.ts +391 -3
  251. package/dist/supervisor-run/index.d.ts.map +1 -0
  252. package/dist/supervisor-run/index.js +1689 -2
  253. package/dist/{supervisor-run-D_sokXcO.js.map → supervisor-run/index.js.map} +1 -1
  254. package/dist/{extract-usage-CdZdoj1s.js → task-failure-attributes-CpQ4y5RD.js} +5 -157
  255. package/dist/task-failure-attributes-CpQ4y5RD.js.map +1 -0
  256. package/dist/{tool-groups-CZPGGlHf.d.ts → tool-groups-Cteb03Ps.d.ts} +3 -3
  257. package/dist/tool-groups-Cteb03Ps.d.ts.map +1 -0
  258. package/dist/tool-waste-BDdBZG1F.js +803 -0
  259. package/dist/tool-waste-BDdBZG1F.js.map +1 -0
  260. package/dist/tool-waste-DjRDEsuI.d.ts +128 -0
  261. package/dist/tool-waste-DjRDEsuI.d.ts.map +1 -0
  262. package/dist/trace-repair/index.d.ts +14 -5
  263. package/dist/trace-repair/index.d.ts.map +1 -1
  264. package/dist/trace-repair/index.js +35 -7
  265. package/dist/trace-repair/index.js.map +1 -1
  266. package/dist/traces.d.ts +406 -7
  267. package/dist/traces.d.ts.map +1 -0
  268. package/dist/traces.js +1011 -10
  269. package/dist/traces.js.map +1 -0
  270. package/dist/trajectory-replay/index.d.ts +16 -3
  271. package/dist/trajectory-replay/index.d.ts.map +1 -1
  272. package/dist/trajectory-replay/index.js +52 -5
  273. package/dist/trajectory-replay/index.js.map +1 -1
  274. package/dist/types-BEPZc6eo.d.ts +93 -0
  275. package/dist/types-BEPZc6eo.d.ts.map +1 -0
  276. package/dist/{usage-receipt-t7vAzCRQ.js → types-BI4fT3HN.js} +67 -67
  277. package/dist/types-BI4fT3HN.js.map +1 -0
  278. package/dist/{types-BnjdJ70P.d.ts → types-BZ59Ahr8.d.ts} +5 -5
  279. package/dist/{types-BnjdJ70P.d.ts.map → types-BZ59Ahr8.d.ts.map} +1 -1
  280. package/dist/{types-D216SgwM.d.ts → types-Cx3YUh2r.d.ts} +4 -241
  281. package/dist/types-Cx3YUh2r.d.ts.map +1 -0
  282. package/dist/{types-BvDKaULh.d.ts → types-DLQx4mKU.d.ts} +4 -4
  283. package/dist/{types-BvDKaULh.d.ts.map → types-DLQx4mKU.d.ts.map} +1 -1
  284. package/dist/{types-D4mog56g.d.ts → types-yLK8gXE9.d.ts} +2 -2
  285. package/dist/{types-D4mog56g.d.ts.map → types-yLK8gXE9.d.ts.map} +1 -1
  286. package/dist/verdict-BndeTAh_.js +61 -0
  287. package/dist/verdict-BndeTAh_.js.map +1 -0
  288. package/dist/verdict-E4eRNf7-.d.ts +392 -0
  289. package/dist/verdict-E4eRNf7-.d.ts.map +1 -0
  290. package/dist/{verdict-cache-BCcOh0kF.js → verdict-cache-mZf5FEiY.js} +3 -55
  291. package/dist/{verdict-cache-BCcOh0kF.js.map → verdict-cache-mZf5FEiY.js.map} +1 -1
  292. package/dist/wire/index.d.ts +3 -3
  293. package/dist/wire/index.d.ts.map +1 -1
  294. package/dist/wire/index.js +1 -1
  295. package/docs/charter.md +3 -3
  296. package/docs/control-runtime.md +3 -42
  297. package/docs/experiment.md +0 -1
  298. package/docs/feature-guide.md +2 -2
  299. package/docs/trace-repair-grader.md +1 -0
  300. package/docs/trajectory-replay.md +1 -0
  301. package/docs/verdicts.md +43 -0
  302. package/docs/verification-strategies.md +3 -2
  303. package/package.json +6 -11
  304. package/dist/agent-profile-DPi7IZg7.d.ts.map +0 -1
  305. package/dist/analyze-runs-C30yljDJ.js.map +0 -1
  306. package/dist/baseline-CavEbRyH.d.ts +0 -136
  307. package/dist/baseline-CavEbRyH.d.ts.map +0 -1
  308. package/dist/benchmark-command-BteMFN62.js.map +0 -1
  309. package/dist/benchmarks-Dzs8CKb1.js +0 -755
  310. package/dist/benchmarks-Dzs8CKb1.js.map +0 -1
  311. package/dist/campaign-C2TTzQII.js.map +0 -1
  312. package/dist/completion-verifier-DJA5BhPb.d.ts +0 -414
  313. package/dist/completion-verifier-DJA5BhPb.d.ts.map +0 -1
  314. package/dist/control.d.ts +0 -3
  315. package/dist/control.js +0 -2
  316. package/dist/dataset-C8xaLXdY.d.ts.map +0 -1
  317. package/dist/default-registry-Bf8Woqmq.d.ts.map +0 -1
  318. package/dist/default-registry-BmktKy8r.js.map +0 -1
  319. package/dist/engine-3hL-XqwJ.d.ts.map +0 -1
  320. package/dist/experiment-tracker-CnRICnMl.js +0 -500
  321. package/dist/experiment-tracker-CnRICnMl.js.map +0 -1
  322. package/dist/experiment-tracker-IMntXr6J.d.ts.map +0 -1
  323. package/dist/extract-usage-CdZdoj1s.js.map +0 -1
  324. package/dist/failure-cluster-CqcvCcdR.d.ts.map +0 -1
  325. package/dist/feedback-trajectory-CacHpxsp.d.ts.map +0 -1
  326. package/dist/index-BZ3-y4YL.d.ts +0 -391
  327. package/dist/index-BZ3-y4YL.d.ts.map +0 -1
  328. package/dist/index-CQTZ-4XN.d.ts.map +0 -1
  329. package/dist/index-DPPGNJ_R.d.ts.map +0 -1
  330. package/dist/index-YE4KdKbO2.d.ts +0 -335
  331. package/dist/index-YE4KdKbO2.d.ts.map +0 -1
  332. package/dist/paired-arms-iZ08VFMN.js +0 -260
  333. package/dist/paired-arms-iZ08VFMN.js.map +0 -1
  334. package/dist/paired-promotion-decision-B6zJ3gYM.d.ts +0 -114
  335. package/dist/paired-promotion-decision-B6zJ3gYM.d.ts.map +0 -1
  336. package/dist/prime-protocol-BfSalTfR.js.map +0 -1
  337. package/dist/promotion-policy-Ckjhzg_4.d.ts +0 -289
  338. package/dist/promotion-policy-Ckjhzg_4.d.ts.map +0 -1
  339. package/dist/promotion-policy-CrLrmys8.js.map +0 -1
  340. package/dist/proposal-findings-2GIUo1et.js.map +0 -1
  341. package/dist/propose-review-control-dSNPjFUH.js +0 -1458
  342. package/dist/propose-review-control-dSNPjFUH.js.map +0 -1
  343. package/dist/release-report-BUYmoKo2.js.map +0 -1
  344. package/dist/release-report-DA2BCu5a.d.ts.map +0 -1
  345. package/dist/replay-CohS93nE.js +0 -1859
  346. package/dist/replay-CohS93nE.js.map +0 -1
  347. package/dist/replay-DbhZ4Ked.d.ts +0 -834
  348. package/dist/replay-DbhZ4Ked.d.ts.map +0 -1
  349. package/dist/reward-hacking-BDToousL.js.map +0 -1
  350. package/dist/run-evidence-Dhi3C81V.d.ts +0 -225
  351. package/dist/run-evidence-Dhi3C81V.d.ts.map +0 -1
  352. package/dist/schema-Cef2cFmb2.d.ts.map +0 -1
  353. package/dist/semantic-concept-judge-D1z-KepS.js +0 -767
  354. package/dist/semantic-concept-judge-D1z-KepS.js.map +0 -1
  355. package/dist/series-convergence-ofsqPWhs.d.ts +0 -35
  356. package/dist/series-convergence-ofsqPWhs.d.ts.map +0 -1
  357. package/dist/server-iu0ede49.js.map +0 -1
  358. package/dist/single-run-lock-DFWHEB09.js.map +0 -1
  359. package/dist/skill-usage-CO9OLRBx.d.ts.map +0 -1
  360. package/dist/skillopt-optimization-method-CQdVeM8k.js.map +0 -1
  361. package/dist/skillopt-optimization-method-USDKhxSA.d.ts.map +0 -1
  362. package/dist/statistical-heldout-TQ-4CYiN.d.ts.map +0 -1
  363. package/dist/statistics-ByxzSiOM.js +0 -2212
  364. package/dist/statistics-ByxzSiOM.js.map +0 -1
  365. package/dist/statistics-D6Uebe_4.d.ts +0 -968
  366. package/dist/statistics-D6Uebe_4.d.ts.map +0 -1
  367. package/dist/supervisor-run-D_sokXcO.js +0 -1690
  368. package/dist/test-graded-scenario-D1TaI2va.d.ts +0 -141
  369. package/dist/test-graded-scenario-D1TaI2va.d.ts.map +0 -1
  370. package/dist/test-graded-scenario-JHcKQNpq.js +0 -318
  371. package/dist/test-graded-scenario-JHcKQNpq.js.map +0 -1
  372. package/dist/tool-groups-CZPGGlHf.d.ts.map +0 -1
  373. package/dist/tool-use-metrics-DEGMKycK.js +0 -370
  374. package/dist/tool-use-metrics-DEGMKycK.js.map +0 -1
  375. package/dist/types-D216SgwM.d.ts.map +0 -1
  376. package/dist/usage-receipt-t7vAzCRQ.js.map +0 -1
  377. package/dist/verdict-DExhxfgR.d.ts +0 -201
  378. package/dist/verdict-DExhxfgR.d.ts.map +0 -1
@@ -0,0 +1,577 @@
1
+ import { a as RunRecord } from "./run-record-CKiihE6f.js";
2
+ import { d as PairedBootstrapResult, u as PairedBootstrapOptions } from "./paired-promotion-decision-CGzg0cI_.js";
3
+ //#region src/statistics/paired-binary.d.ts
4
+ /** A binomial proportion estimate with a confidence interval. */
5
+ interface ProportionInterval {
6
+ /** Point estimate successes / n (0 when n = 0). */
7
+ estimate: number;
8
+ /** Lower bound, clamped to [0, 1]. */
9
+ lower: number;
10
+ /** Upper bound, clamped to [0, 1]. */
11
+ upper: number;
12
+ }
13
+ /**
14
+ * Wilson score interval for a binomial proportion. Correct at small n and near
15
+ * 0/1, where the normal (Wald) approximation produces bounds outside [0, 1] and
16
+ * understates coverage. Use this for any pass-rate / hit-rate / realness-rate
17
+ * CI — the continuous `confidenceInterval` assumes the wrong distribution for a
18
+ * proportion. `n = 0 ⇒ {0, 0, 0}`.
19
+ */
20
+ declare function wilson(successes: number, n: number, confidence?: number): ProportionInterval;
21
+ /**
22
+ * Are these per-item outcomes binary (every value exactly 0 or 1)?
23
+ *
24
+ * The discriminator a promotion gate needs before choosing a paired statistic.
25
+ * On binary outcomes the paired delta vector lives in {-1, 0, +1} and is
26
+ * normally dominated by zeros (both arms solve, or both arms miss, most items),
27
+ * so its MEDIAN is pinned at exactly 0 no matter how large the real shift in
28
+ * success rate is — and a bootstrap CI on that median collapses to [0, 0].
29
+ * A gate keying on `ci.low > threshold` is then structurally unable to see
30
+ * either a gain or a regression. Detect this shape and switch to the
31
+ * paired-binary estimators ({@link mcnemar}, {@link pairedRiskDifference})
32
+ * instead of silently answering "no" forever.
33
+ *
34
+ * Empty input is NOT binary: there is no evidence of the outcome's shape, and
35
+ * defaulting an empty vector into the binary branch would pick a statistic on
36
+ * no data at all.
37
+ *
38
+ * NOT the right discriminator for a gate. It recognises the literal {0, 1}
39
+ * encoding and nothing else, so a pass/fail dimension emitted on 0-100 — which
40
+ * judges in this codebase do routinely — reads as non-binary, and a single
41
+ * partial-credit score in an otherwise pass/fail vector flips it to false while
42
+ * leaving the median just as blind. Gates want {@link pairedBinaryScale} (any
43
+ * two-point encoding). This predicate remains for callers that specifically
44
+ * mean "literally 0/1".
45
+ */
46
+ declare function isBinaryOutcomeVector(values: ArrayLike<number>): boolean;
47
+ /** Result of a McNemar paired-binary significance test. */
48
+ interface McNemarResult {
49
+ /** Total paired observations. */
50
+ n: number;
51
+ /** Discordant pairs (b + c) — the only ones that carry signal. */
52
+ nDiscordant: number;
53
+ /** Pairs where treatment succeeded and control failed ("newly correct"). */
54
+ b: number;
55
+ /** Pairs where control succeeded and treatment failed ("newly wrong"). */
56
+ c: number;
57
+ /** Continuity-corrected chi-square statistic (reference; exact p drives the call). */
58
+ statistic: number;
59
+ /** Two-sided p-value. Exact (binomial sign test on discordant pairs). */
60
+ pValue: number;
61
+ }
62
+ /**
63
+ * McNemar's test for paired binary outcomes — the correct significance test for
64
+ * "does treatment change the success rate vs control on the SAME items". Only
65
+ * discordant pairs (one arm right, the other wrong) carry information; concordant
66
+ * pairs are uninformative, so a paired t-test / two-proportion z-test on the raw
67
+ * rates is wrong here. The p-value is exact: under H0 the b "treatment-wins" are
68
+ * Binomial(b + c, 0.5), so the two-sided p is the doubled binomial tail — correct
69
+ * at the small discordant counts typical of eval runs (no continuity-corrected
70
+ * chi-square approximation needed, though it is returned as `statistic` for
71
+ * reference). Inputs are paired 0/1 (or boolean) arrays, control first to match
72
+ * the module's (before, after) convention. Throws on unequal lengths.
73
+ */
74
+ declare function mcnemar(control: ArrayLike<number | boolean>, treatment: ArrayLike<number | boolean>): McNemarResult;
75
+ /** A paired binary effect size (treatment rate − control rate) with a CI. */
76
+ interface RiskDifferenceResult {
77
+ /** Total paired observations. */
78
+ n: number;
79
+ /** Discordant pairs: treatment-win count. */
80
+ b: number;
81
+ /** Discordant pairs: control-win count. */
82
+ c: number;
83
+ /** Paired risk difference p(treatment) − p(control) = (b − c) / n. */
84
+ riskDifference: number;
85
+ /** Lower bound of the CI, clamped to [-1, 1]. */
86
+ lower: number;
87
+ /** Upper bound of the CI, clamped to [-1, 1]. */
88
+ upper: number;
89
+ /** Confidence level used. */
90
+ confidence: number;
91
+ }
92
+ /**
93
+ * Paired risk difference (the effect-size companion to {@link mcnemar}): the
94
+ * change in success rate p(treatment) − p(control) on matched items, which for
95
+ * paired binary data equals (b − c) / n. The CI uses the paired variance from
96
+ * the discordant counts, not the independent-samples formula (which overstates
97
+ * the interval by ignoring the pairing). Inputs are paired 0/1 (or boolean)
98
+ * arrays, control first. Throws on unequal lengths.
99
+ *
100
+ * REPORTING ONLY — do NOT decide a promotion on this interval. The CI is a Wald
101
+ * normal approximation, which badly UNDERCOVERS when only a handful of pairs are
102
+ * discordant: at n = 3 with b = 2, c = 0 it returns [0.133, 1.000], excluding 0,
103
+ * while McNemar's exact test on the same data gives p = 0.50. A gate keying on
104
+ * `lower > 0` would promote noise. Use {@link pairedRiskDifferenceExact}, whose
105
+ * interval is dual to the exact test by construction, for any decision.
106
+ */
107
+ declare function pairedRiskDifference(control: ArrayLike<number | boolean>, treatment: ArrayLike<number | boolean>, confidence?: number): RiskDifferenceResult;
108
+ /** A paired binary effect size with an EXACT interval and the exact test that
109
+ * bounds it — one object so a caller cannot read the estimate without the
110
+ * significance it is entitled to. */
111
+ interface ExactRiskDifferenceResult {
112
+ /** Total paired observations. */
113
+ n: number;
114
+ /** Discordant pairs: treatment-win count. */
115
+ b: number;
116
+ /** Discordant pairs: control-win count. */
117
+ c: number;
118
+ /** Discordant pairs (b + c) — the only ones carrying information. */
119
+ nDiscordant: number;
120
+ /** Paired risk difference p(treatment) − p(control) = (b − c) / n. */
121
+ riskDifference: number;
122
+ /** Exact conditional CI lower bound. 0 when there are no discordant pairs. */
123
+ lower: number;
124
+ /** Exact conditional CI upper bound. 0 when there are no discordant pairs. */
125
+ upper: number;
126
+ /** Confidence level used. */
127
+ confidence: number;
128
+ /** McNemar's exact two-sided p-value on the same discordant counts. */
129
+ pValue: number;
130
+ }
131
+ /**
132
+ * Paired risk difference with the EXACT CONDITIONAL interval — the estimator a
133
+ * promotion gate may decide on.
134
+ *
135
+ * Conditional on the number of discordant pairs m = b + c, the treatment-win
136
+ * count b is Binomial(m, π) with π = P(treatment wins | discordant), and the
137
+ * risk difference is an exact reparameterisation: RD = (2π − 1)·m/n. So a
138
+ * Clopper-Pearson exact interval for π maps straight onto RD. This buys the
139
+ * property the Wald interval in {@link pairedRiskDifference} does not have:
140
+ *
141
+ * **`lower > 0` ⟺ McNemar's exact test rejects at α = 1 − confidence.**
142
+ *
143
+ * Clopper-Pearson excludes π = 0.5 exactly when the two-sided exact binomial
144
+ * test of π = 0.5 rejects, and that test IS {@link mcnemar}'s p-value — so the
145
+ * interval and the test can never disagree, and a gate keyed on `lower` cannot
146
+ * promote what the exact test refuses. The exact p is returned in the same
147
+ * object so the two are impossible to compute apart.
148
+ *
149
+ * The interval is conservative (exact intervals over-cover; conditioning on m
150
+ * discards the concordant pairs' information about m itself). That is the
151
+ * correct direction for a promotion gate: it refuses more often, never less.
152
+ *
153
+ * With m = 0 there are no discordant pairs and π is not identified: the result
154
+ * is the degenerate [0, 0] with p = 1. That is NOT evidence of equivalence —
155
+ * callers must treat a zero-width interval as "cannot decide", not as "no
156
+ * difference". Inputs are paired 0/1 (or boolean) arrays, control first.
157
+ * Throws on unequal lengths.
158
+ */
159
+ declare function pairedRiskDifferenceExact(control: ArrayLike<number | boolean>, treatment: ArrayLike<number | boolean>, confidence?: number): ExactRiskDifferenceResult;
160
+ /** A paired binary effect size with an interval that is valid at a NONZERO
161
+ * margin — the estimator a noninferiority decision may be made on. */
162
+ interface ScoreRiskDifferenceResult {
163
+ /** Total paired observations. */
164
+ n: number;
165
+ /** Discordant pairs: treatment-win count. */
166
+ b: number;
167
+ /** Discordant pairs: control-win count. */
168
+ c: number;
169
+ /** Discordant pairs (b + c). */
170
+ nDiscordant: number;
171
+ /** Paired risk difference p(treatment) − p(control) = (b − c) / n. */
172
+ riskDifference: number;
173
+ /** Score-interval lower bound on the population risk difference. */
174
+ lower: number;
175
+ /** Score-interval upper bound on the population risk difference. */
176
+ upper: number;
177
+ /** Confidence level used. */
178
+ confidence: number;
179
+ }
180
+ /**
181
+ * Paired risk difference with TANGO'S (1998) SCORE INTERVAL — the estimator a
182
+ * promotion gate may decide on **at a nonzero margin**.
183
+ *
184
+ * {@link pairedRiskDifferenceExact} conditions on the observed discordant count
185
+ * `m = b + c`, builds a Clopper-Pearson interval for the win share among those
186
+ * `m` pairs, and multiplies by the observed `m/n`. That is exact for testing
187
+ * RD = 0 — it is dual to McNemar — but it is NOT a confidence interval for the
188
+ * population risk difference at a nonzero margin, because the sampling
189
+ * variability of `m/n` itself is discarded. The gap is not academic: with the
190
+ * production caller's `pairedDeltaThreshold: -0.05`, a process whose true risk
191
+ * difference sits exactly on that margin clears a nominal-95 % `lower > margin`
192
+ * check 24.75 % of the time at n = 40 and 43.95 % at n = 76 (2000 replicates
193
+ * each) when the conditional interval decides.
194
+ *
195
+ * Tango's interval inverts the score test of RD = delta, which estimates the
196
+ * nuisance loss rate under each hypothesised delta instead of fixing it at the
197
+ * observed value, so `m` contributes its own uncertainty. It is the method
198
+ * `ratesci::scorepairci` uses for paired risk-difference noninferiority, and it
199
+ * is not conditional, so it stays valid as the margin moves away from zero.
200
+ *
201
+ * The bounds are found by bisecting `tangoScore(delta) = ±z` — the score is
202
+ * monotone decreasing in delta, so each crossing is unique. Inputs are paired
203
+ * 0/1 (or boolean) arrays, control first. Throws on unequal lengths.
204
+ */
205
+ declare function pairedRiskDifferenceScore(control: ArrayLike<number | boolean>, treatment: ArrayLike<number | boolean>, confidence?: number): ScoreRiskDifferenceResult;
206
+ /**
207
+ * The common positive level `s` such that EVERY value across both paired arms is
208
+ * exactly 0 or `s` — i.e. the outcome is pass/fail, whatever encoding it arrived
209
+ * in. Returns null when the outcomes are not two-point, when the two arms use
210
+ * different levels, or when no positive value was observed at all (all-zero
211
+ * arms: the level is not identified, and there is nothing to decide anyway).
212
+ *
213
+ * This is the scale-aware successor to {@link isBinaryOutcomeVector}, which only
214
+ * recognises literal {0, 1}. Judges in this codebase emit dimensions on 0-100 as
215
+ * well as [0,1] (see `detectScale` in `campaign/gates/statistical-heldout.ts`),
216
+ * so a pass/fail dimension routinely arrives as {0, 100} and a {0,1}-only test
217
+ * silently sends it down the median path that cannot see it. Any positive level
218
+ * is accepted, not just 1 and 100: for a two-point {0, s} outcome the mean paired
219
+ * delta is exactly s·(b − c)/n, so the binary estimators apply after dividing by
220
+ * s and rescaling the result back into the caller's native units.
221
+ *
222
+ * Non-finite values ⇒ null: an unusable outcome must not be classified as a
223
+ * clean pass/fail shape.
224
+ */
225
+ declare function pairedBinaryScale(before: ArrayLike<number>, after: ArrayLike<number>): number | null;
226
+ /**
227
+ * Unbiased pass@k for code generation (Chen et al. 2021, "Evaluating Large
228
+ * Language Models Trained on Code"). Given `n` independent samples for one
229
+ * problem of which `c` pass, the probability that at least one of a random k of
230
+ * them passes is 1 − C(n−c, k) / C(n, k). Estimating pass@k as "did any of the
231
+ * first k pass" is biased high at small n; this is the variance-reduced estimator
232
+ * averaged implicitly over all k-subsets. Average the per-problem values across
233
+ * the suite for the corpus pass@k. Computed in the numerically stable product
234
+ * form. Requires 1 ≤ k ≤ n and 0 ≤ c ≤ n.
235
+ */
236
+ declare function passAtK(n: number, c: number, k: number): number;
237
+ //#endregion
238
+ //#region src/statistics/sequential-eprocess.d.ts
239
+ interface EProcessOptions {
240
+ /** Type-I error budget. The process decides when wealth ≥ 1/alpha
241
+ * (Ville's inequality). Default 0.05. */
242
+ alpha?: number;
243
+ /** Truncation bound on the predictable bet λ ∈ [0, maxBet]. Must satisfy
244
+ * maxBet < 1/nullMean so every wealth factor stays strictly positive.
245
+ * Default 0.5. */
246
+ maxBet?: number;
247
+ /** The null boundary m₀ for H0: E[x] ≤ m₀ on x ∈ [0,1]. Default 0.5
248
+ * (the paired-delta encoding x = (d+1)/2 maps "no effect" to 1/2).
249
+ * A pre-registered minEffect shifts this — see `sequentialPairedGate`. */
250
+ nullMean?: number;
251
+ }
252
+ interface EProcessStep {
253
+ /** Current wealth W_n — the e-value against H0 after n observations. */
254
+ wealth: number;
255
+ /** Observations consumed so far. */
256
+ n: number;
257
+ /** True from the first n where W_n ≥ 1/alpha onward (sticky). */
258
+ decided: boolean;
259
+ }
260
+ interface EProcessState extends EProcessStep {
261
+ alpha: number;
262
+ maxBet: number;
263
+ nullMean: number;
264
+ /** The decision boundary 1/alpha. */
265
+ threshold: number;
266
+ /** Observation count at the first threshold crossing; undefined until decided. */
267
+ decidedAtN?: number;
268
+ }
269
+ interface EProcess {
270
+ /** Consume one observation x ∈ [0,1]. Throws on non-finite / out-of-range
271
+ * input — a silent clamp would corrupt the type-I guarantee. */
272
+ update(x: number): EProcessStep;
273
+ state(): EProcessState;
274
+ }
275
+ /**
276
+ * Betting test-martingale for bounded observations — the e-process core of
277
+ * anytime-valid sequential testing (Waudby-Smith & Ramdas, "Estimating means
278
+ * of bounded random variables by betting", JRSS-B 2024).
279
+ *
280
+ * Observations x_i ∈ [0,1]; H0: E[x] ≤ m₀ (`nullMean`, default 1/2). Wealth
281
+ *
282
+ * W_t = Π_{i≤t} (1 + λ_i (x_i − m₀)), W_0 = 1
283
+ *
284
+ * with the truncated GROW-style plug-in bet computed from PRIOR observations:
285
+ *
286
+ * λ_i = clamp((μ̂_{i−1} − m₀) / (σ̂²_{i−1} + (μ̂_{i−1} − m₀)²), 0, maxBet)
287
+ *
288
+ * where μ̂/σ̂² are the shrunk running estimates μ̂_t = (1/2 + Σx_i)/(t+1),
289
+ * σ̂²_t = (1/4 + Σ(x_i − μ̂_i)²)/(t+1).
290
+ *
291
+ * PREDICTABILITY INVARIANT (load-bearing): λ_i is a function of x_1..x_{i−1}
292
+ * ONLY — it may never see x_i. With λ_i ≥ 0 predictable, each factor has
293
+ * E[1 + λ_i(x_i − m₀) | past] ≤ 1 under H0, so W is a nonnegative
294
+ * supermartingale and Ville's inequality gives P(∃t: W_t ≥ 1/α) ≤ α — the
295
+ * type-I guarantee holds at ANY data-dependent stopping time. λ_1 is always 0
296
+ * (no prior evidence), so the first observation never moves wealth.
297
+ *
298
+ * `decided` latches at the first crossing W_t ≥ 1/α and never un-latches;
299
+ * wealth keeps updating after the crossing (the e-process remains valid), but
300
+ * the decision time is the first crossing.
301
+ */
302
+ declare function eProcess(opts?: EProcessOptions): EProcess;
303
+ //#endregion
304
+ //#region src/paired-arms.d.ts
305
+ /** One arm observation of one work item. Structural on purpose: callers
306
+ * project their own record type (e.g. a `RunRecord`) into this shape. */
307
+ interface PairedArmRow {
308
+ /** Matching key — rows sharing a `pairKey` across both arms form pairs
309
+ * (typically the task/scenario/seed identity). */
310
+ pairKey: string;
311
+ /** Rep identity within a `pairKey` (e.g. a seed or rep number). Required on
312
+ * every row of a `pairKey` that has more than one rep in either arm; reps
313
+ * then pair only on exact (`pairKey`, `repKey`) match, never on outcome
314
+ * content. Optional when each arm has at most one rep of the item. */
315
+ repKey?: string;
316
+ /** Arm label this row was produced under. */
317
+ arm: string;
318
+ /** Binary outcome; omit when the comparison has no pass/fail notion. */
319
+ pass?: boolean;
320
+ /** Named numeric measurements (score, cost, latency, …). */
321
+ metrics?: Record<string, number>;
322
+ }
323
+ interface PairArmsOptions {
324
+ /** Arm treated as the control side of every pair. */
325
+ baselineArm: string;
326
+ /** Arm treated as the treatment side of every pair. */
327
+ treatmentArm: string;
328
+ }
329
+ /** One matched (baseline, treatment) observation of the same work item. */
330
+ interface MatchedPair {
331
+ pairKey: string;
332
+ /** 0-based position of this pair within its `pairKey`, ordered by sorted
333
+ * `repKey` (always 0 for a single-rep item). The rep identity itself is on
334
+ * the rows (`baseline.repKey` / `treatment.repKey`). */
335
+ repIndex: number;
336
+ baseline: PairedArmRow;
337
+ treatment: PairedArmRow;
338
+ }
339
+ interface PairArmsResult {
340
+ /** Matched pairs, ordered by (`pairKey`, `repIndex`). */
341
+ pairs: MatchedPair[];
342
+ /** Baseline rows left without a treatment counterpart — reported, never
343
+ * silently dropped. */
344
+ unpairedBaseline: PairedArmRow[];
345
+ /** Treatment rows left without a baseline counterpart. */
346
+ unpairedTreatment: PairedArmRow[];
347
+ }
348
+ /**
349
+ * Match rows across two arms into (baseline, treatment) pairs by `pairKey`.
350
+ *
351
+ * A `pairKey` with at most one row per arm pairs directly, no `repKey`
352
+ * needed. A `pairKey` with multiple reps in either arm requires `repKey` on
353
+ * every one of its rows, and reps pair only on exact (`pairKey`, `repKey`)
354
+ * match — pairing is keyed purely on row identity, never on outcome content
355
+ * (outcome-keyed matching deflates discordant counts and biases McNemar), and
356
+ * is therefore independent of input order. Reps whose `repKey` has no
357
+ * counterpart in the other arm, and items present in only one arm, land in
358
+ * the unpaired lists — reported, never truncated.
359
+ *
360
+ * Fail-loud: throws when either named arm has zero rows (an unknown arm
361
+ * name would otherwise read as "everything unpaired"), when the two arm
362
+ * names are equal, when a multi-rep `pairKey` has a row without `repKey`, or
363
+ * when a (`pairKey`, arm) group repeats a `repKey` (the match would be
364
+ * ambiguous).
365
+ */
366
+ declare function pairArms(rows: readonly PairedArmRow[], opts: PairArmsOptions): PairArmsResult;
367
+ /** Paired pass/fail comparison over the pairs where BOTH sides carry `pass`. */
368
+ interface PairedCorrectness {
369
+ /** Discordant pairs where the treatment passed and the baseline failed. */
370
+ b10: number;
371
+ /** Discordant pairs where the baseline passed and the treatment failed. */
372
+ b01: number;
373
+ /** Exact McNemar significance over the paired outcomes (`b === b10`, `c === b01`). */
374
+ mcnemar: McNemarResult;
375
+ /** Paired effect size: p(treatment) − p(baseline) with a paired-variance CI. */
376
+ riskDifference: RiskDifferenceResult;
377
+ }
378
+ /** Paired delta summary for one named metric (delta = treatment − baseline). */
379
+ interface PairedMetricDelta {
380
+ name: string;
381
+ /** Pairs where BOTH sides carry a finite value for this metric. */
382
+ n: number;
383
+ /** Pairs where at least one side does not carry the metric. */
384
+ nMissing: number;
385
+ /** Median paired delta, or null when `n === 0`. */
386
+ medianDelta: number | null;
387
+ /** Mean paired delta, or null when `n === 0`. */
388
+ meanDelta: number | null;
389
+ /** Bootstrap CI on the paired delta (`pairedBootstrap`); null when
390
+ * `n === 0` — a zero-width [0, 0] interval on no data would read as a
391
+ * measured tight null. */
392
+ bootstrapCi: PairedBootstrapResult | null;
393
+ /** Wilcoxon signed-rank test on the paired deltas; null when `n === 0`. */
394
+ wilcoxon: {
395
+ w: number;
396
+ p: number;
397
+ } | null;
398
+ }
399
+ interface ComparePairedArmsOptions extends PairArmsOptions {
400
+ /** Metrics to compare. Default: every metric name observed on any matched
401
+ * pair, sorted. A name that appears on no pair is still reported (with
402
+ * `n = 0`) so a misspelled metric is visible instead of vanishing. */
403
+ metricNames?: string[];
404
+ /** Passed through to `pairedBootstrap` — set `seed` for reproducible CIs. */
405
+ bootstrap?: PairedBootstrapOptions;
406
+ }
407
+ interface PairedArmsComparison {
408
+ nPairs: number;
409
+ nUnpairedBaseline: number;
410
+ nUnpairedTreatment: number;
411
+ /** null when no matched pair carries `pass` on both sides — a pass/fail
412
+ * verdict over rows that never measured pass/fail would be fabricated. */
413
+ correctness: PairedCorrectness | null;
414
+ metricDeltas: PairedMetricDelta[];
415
+ }
416
+ /**
417
+ * Full matched-pair arm comparison: pair via {@link pairArms}, then compose
418
+ * the paired estimators from `statistics` over the matched pairs.
419
+ *
420
+ * Correctness uses only the pairs where both sides carry `pass` (`mcnemar.n`
421
+ * is that subset's size); each metric uses only the pairs where both sides
422
+ * carry a finite value for it, with the remainder counted in `nMissing`.
423
+ * Deltas are treatment − baseline throughout.
424
+ *
425
+ * Fail-loud: inherits {@link pairArms}'s unknown-arm throw, and throws on a
426
+ * non-finite metric value — silently treating corrupt telemetry as "metric
427
+ * absent" would misreport it as missing coverage.
428
+ */
429
+ declare function comparePairedArms(rows: readonly PairedArmRow[], opts: ComparePairedArmsOptions): PairedArmsComparison;
430
+ interface MatchedRunRecordPair {
431
+ pairKey: string;
432
+ repKey: string;
433
+ baseline: RunRecord;
434
+ treatment: RunRecord;
435
+ }
436
+ interface PairRunRecordsResult {
437
+ pairs: MatchedRunRecordPair[];
438
+ unpairedBaseline: RunRecord[];
439
+ unpairedTreatment: RunRecord[];
440
+ }
441
+ /**
442
+ * Pair two RunRecord arms by the identity of the evaluated work:
443
+ * `(experimentId, scenarioId, seed)`.
444
+ *
445
+ * Falling back to array order, candidate id, or experiment id can compare
446
+ * different tasks and fabricate lift. Duplicate identities throw.
447
+ */
448
+ declare function pairRunRecords(baselineRuns: readonly RunRecord[], treatmentRuns: readonly RunRecord[]): PairRunRecordsResult;
449
+ //#endregion
450
+ //#region src/pre-registration.d.ts
451
+ /**
452
+ * Pre-registered hypotheses — declare what you're testing BEFORE the
453
+ * run, check it AFTER. Prevents p-hacking, optional stopping, and the
454
+ * "we ran until it looked good" failure mode.
455
+ *
456
+ * Manifest is a plain JSON-friendly object. Sign it with a content hash
457
+ * + timestamp; the registered record becomes immutable. Post-run,
458
+ * evaluate the manifest against observed results — the library refuses
459
+ * to let you re-interpret a different metric as the declared one.
460
+ */
461
+ interface HypothesisManifest {
462
+ id: string;
463
+ /** Human prose — goes into the audit trail. */
464
+ hypothesis: string;
465
+ /** Metric the hypothesis claims to move. */
466
+ metric: string;
467
+ /** 'increase' = candidate should score higher than baseline; 'decrease' = lower. */
468
+ direction: 'increase' | 'decrease';
469
+ /** Minimum effect size to count (same units as the metric). */
470
+ minEffect: number;
471
+ /** Alpha threshold. */
472
+ alpha: number;
473
+ /** Target statistical power at which sample size was pre-computed. */
474
+ power: number;
475
+ /** Declared N per arm before running. */
476
+ preRegisteredN: number;
477
+ /** ISO8601 timestamp the manifest was registered. */
478
+ registeredAt: string;
479
+ /** Optional identifiers to tie into the trace corpus. */
480
+ baselineLabel?: string;
481
+ candidateLabel?: string;
482
+ }
483
+ /**
484
+ * Identifier for the hashing scheme used to produce `contentHash`.
485
+ *
486
+ * `'sha256-content'` — sha256 hex over the canonicalized manifest with
487
+ * the `contentHash` and `algo` fields stripped. Held as a string union
488
+ * so future schemes can be added without breaking parsers; SignedManifest
489
+ * values without `algo` deserialize cleanly because the field is optional.
490
+ */
491
+ type SignedManifestAlgo = 'sha256-content';
492
+ interface SignedManifest extends HypothesisManifest {
493
+ /** sha256 hex of canonicalized manifest (everything except contentHash and algo). */
494
+ contentHash: string;
495
+ /**
496
+ * Algorithm string describing how `contentHash` was produced.
497
+ *
498
+ * Optional on the type so serialized manifests without it still parse,
499
+ * but ALWAYS populated by {@link signManifest}. Consumers that want to
500
+ * enforce a known algorithm should reject manifests where this field
501
+ * is missing or unrecognized.
502
+ */
503
+ algo?: SignedManifestAlgo;
504
+ }
505
+ interface HypothesisResult {
506
+ manifest: SignedManifest;
507
+ observedN: number;
508
+ observedEffect: number;
509
+ observedPValue: number;
510
+ /** True iff the observed effect hits the pre-declared direction with
511
+ * magnitude ≥ minEffect AND p < alpha. */
512
+ confirmed: boolean;
513
+ /** Enumerated reasons the hypothesis was rejected (each a machine-tag). */
514
+ rejectionReasons: Array<'wrong_direction' | 'effect_too_small' | 'not_significant' | 'undersampled'>;
515
+ notes?: string;
516
+ }
517
+ /**
518
+ * Deterministic JSON canonicalization — sort object keys recursively.
519
+ *
520
+ * Two semantically-equal objects produce byte-identical canonicalized output;
521
+ * this is what makes a content-hash stable across encoders, key insertion
522
+ * orders, and runtime versions. Exported for any consumer that needs the same
523
+ * canonicalization guarantee outside the manifest-signing path (e.g., signing
524
+ * an artifact bundle, hashing a dataset version, etc.).
525
+ */
526
+ declare function canonicalize(v: unknown): unknown;
527
+ /**
528
+ * SHA-256 hex (full 64 chars) over the canonicalized JSON encoding of `obj`.
529
+ *
530
+ * The same primitive `signManifest` and `verifyManifest` are built on, exposed
531
+ * directly so consumers signing arbitrary structured content (artifact bundles,
532
+ * production packets, dataset manifests, etc.) don't have to re-derive
533
+ * canonicalize+sha256 from scratch.
534
+ *
535
+ * Stable across:
536
+ * - object key insertion order (canonicalization sorts keys recursively)
537
+ * - encoder choice (UTF-8 via TextEncoder, fixed)
538
+ * - runtime (uses the Web Crypto subtle digest, present in Node ≥18 and browsers)
539
+ *
540
+ * Named `hashJson` to disambiguate from `prompt-registry.ts`'s `hashContent`,
541
+ * which takes a string input and returns a truncated 12-char prompt id.
542
+ * Use `hashJson` when you mean "canonicalize then hash."
543
+ *
544
+ * @example
545
+ * const hash = await hashJson({ id: '1', kind: 'spec' })
546
+ * // 'a3f1...' (64 hex chars)
547
+ */
548
+ declare function hashJson<T>(obj: T): Promise<string>;
549
+ /**
550
+ * Sign a manifest with a SHA-256 content hash.
551
+ *
552
+ * The hash covers the canonicalized manifest with the `contentHash`
553
+ * and `algo` fields stripped; this lets verifiers re-sign the rest and
554
+ * compare. Returned manifest always carries `algo: 'sha256-content'`
555
+ * so downstream consumers can identify the scheme; manifests without
556
+ * `algo` still verify because it is stripped before hashing on both sides.
557
+ */
558
+ declare function signManifest(m: HypothesisManifest): Promise<SignedManifest>;
559
+ /**
560
+ * Verify that a signed manifest has not been tampered with.
561
+ *
562
+ * Strips `contentHash` and `algo` before re-signing so manifests without
563
+ * `algo` verify identically to ones that carry it.
564
+ */
565
+ declare function verifyManifest(m: SignedManifest): Promise<boolean>;
566
+ /**
567
+ * Evaluate a pre-registered hypothesis against observed results.
568
+ * Mechanical — no re-interpretation permitted.
569
+ */
570
+ declare function evaluateHypothesis(manifest: SignedManifest, observed: {
571
+ n: number;
572
+ effect: number;
573
+ pValue: number;
574
+ }): Promise<HypothesisResult>;
575
+ //#endregion
576
+ export { isBinaryOutcomeVector as A, EProcessStep as C, ProportionInterval as D, McNemarResult as E, pairedRiskDifferenceScore as F, passAtK as I, wilson as L, pairedBinaryScale as M, pairedRiskDifference as N, RiskDifferenceResult as O, pairedRiskDifferenceExact as P, EProcessState as S, ExactRiskDifferenceResult as T, comparePairedArms as _, evaluateHypothesis as a, EProcess as b, verifyManifest as c, PairArmsOptions as d, PairArmsResult as f, PairedMetricDelta as g, PairedCorrectness as h, canonicalize as i, mcnemar as j, ScoreRiskDifferenceResult as k, ComparePairedArmsOptions as l, PairedArmsComparison as m, HypothesisResult as n, hashJson as o, PairedArmRow as p, SignedManifest as r, signManifest as s, HypothesisManifest as t, MatchedPair as u, pairArms as v, eProcess as w, EProcessOptions as x, pairRunRecords as y };
577
+ //# sourceMappingURL=pre-registration-CZwSFQS4.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"pre-registration-CZwSFQS4.d.ts","names":[],"sources":["../src/statistics/paired-binary.ts","../src/statistics/sequential-eprocess.ts","../src/paired-arms.ts","../src/pre-registration.ts"],"mappings":";;;;UAgBiB;;EAEf;;EAEA;;EAEA;;;;;;;;;iBAUc,OAAO,mBAAmB,WAAW,sBAAoB;;;;;;;;;;;;;;;;;;;;;;;;;;iBA2CzD,sBAAsB,QAAQ;;UAU7B;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;;;;;;;iBAec,QACd,SAAS,6BACT,WAAW,8BACV;;UAmBc;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;;;;;;;;;;iBAkBc,qBACd,SAAS,6BACT,WAAW,6BACX,sBACC;;;;UAkCc;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;iBA+Bc,0BACd,SAAS,6BACT,WAAW,6BACX,sBACC;;;UAyEc;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;;;;;;;;;;;;;;;;;;;;iBA8Dc,0BACd,SAAS,6BACT,WAAW,6BACX,sBACC;;;;;;;;;;;;;;;;;;;;iBA6Ea,kBACd,QAAQ,mBACR,OAAO;;;;;;;;;;;iBA0BO,QAAQ,WAAW,WAAW;;;UCjgB7B;;;EAGf;;;;EAIA;;;;EAIA;;UAGe;;EAEf;;EAEA;;EAEA;;UAGe,sBAAsB;EACrC;EACA;EACA;;EAEA;;EAEA;;UAGe;;;EAGf,OAAO,YAAY;EACnB,SAAS;;;;;;;;;;;;;;;;;;;;;;;;;;;;;iBA8BK,SAAS,OAAM,kBAAuB;;;;;UC/BrC;;;EAGf;;;;;EAKA;;EAEA;;EAEA;;EAEA,UAAU;;UAGK;;EAEf;;EAEA;;;UAIe;EACf;;;;EAIA;EACA,UAAU;EACV,WAAW;;UAGI;;EAEf,OAAO;;;EAGP,kBAAkB;;EAElB,mBAAmB;;;;;;;;;;;;;;;;;;;;iBAqBL,SAAS,eAAe,gBAAgB,MAAM,kBAAkB;;UA8G/D;;EAEf;;EAEA;;EAEA,SAAS;;EAET,gBAAgB;;;UAID;EACf;;EAEA;;EAEA;;EAEA;;EAEA;;;;EAIA,aAAa;;EAEb;IAAY;IAAW;;;UAGR,iCAAiC;;;;EAIhD;;EAEA,YAAY;;UAGG;EACf;EACA;EACA;;;EAGA,aAAa;EACb,cAAc;;;;;;;;;;;;;;;iBAgBA,kBACd,eAAe,gBACf,MAAM,2BACL;UAmEc;EACf;EACA;EACA,UAAU;EACV,WAAW;;UAGI;EACf,OAAO;EACP,kBAAkB;EAClB,mBAAmB;;;;;;;;;iBAeL,eACd,uBAAuB,aACvB,wBAAwB,cACvB;;;;;;;;;;;;;UC1Wc;EACf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;EACA;;;;;;;;;;KAWU;UAEK,uBAAuB;;EAEtC;;;;;;;;;EASA,OAAO;;UAGQ;EACf,UAAU;EACV;EACA;EACA;;;EAGA;;EAEA,kBAAkB;EAGlB;;;;;;;;;;;iBAYc,aAAa;;;;;;;;;;;;;;;;;;;;;;iBA8BP,SAAS,GAAG,KAAK,IAAI;;;;;;;;;;iBAkBrB,aAAa,GAAG,qBAAqB,QAAQ;;;;;;;iBAW7C,eAAe,GAAG,iBAAiB;;;;;iBAWnC,mBACpB,UAAU,gBACV;EAAY;EAAW;EAAgB;IACtC,QAAQ"}