@tangle-network/agent-eval 0.144.11 → 0.144.13
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +11 -0
- package/dist/agent-profile-B9_GGsG8.d.ts +84 -0
- package/dist/agent-profile-B9_GGsG8.d.ts.map +1 -0
- package/dist/{agent-profile-cell-BOP-iA9Q.d.ts → agent-profile-cell-BkcRDikH.d.ts} +2 -2
- package/dist/{agent-profile-cell-BOP-iA9Q.d.ts.map → agent-profile-cell-BkcRDikH.d.ts.map} +1 -1
- package/dist/analyst/index.d.ts +134 -16
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +364 -10
- package/dist/analyst/index.js.map +1 -1
- package/dist/backend-integrity-CsVin_Wb.d.ts +280 -0
- package/dist/backend-integrity-CsVin_Wb.d.ts.map +1 -0
- package/dist/{baseline-C-GocmIW.js → baseline-BhPRQBVn.js} +3 -2
- package/dist/{baseline-C-GocmIW.js.map → baseline-BhPRQBVn.js.map} +1 -1
- package/dist/{benchmark-CWeqGl7x.js → benchmark-BhT16ep9.js} +2 -2
- package/dist/{benchmark-CWeqGl7x.js.map → benchmark-BhT16ep9.js.map} +1 -1
- package/dist/{agent-profile-DPi7IZg7.d.ts → benchmark-Ceoan7vk.d.ts} +4 -91
- package/dist/benchmark-Ceoan7vk.d.ts.map +1 -0
- package/dist/{benchmark-command-BteMFN62.js → benchmark-command-CY6cbhOB.js} +36 -79
- package/dist/benchmark-command-CY6cbhOB.js.map +1 -0
- package/dist/benchmarks/index.d.ts +244 -2
- package/dist/benchmarks/index.d.ts.map +1 -0
- package/dist/benchmarks/index.js +733 -1
- package/dist/benchmarks/index.js.map +1 -0
- package/dist/builder-eval/index.d.ts +23 -2
- package/dist/builder-eval/index.d.ts.map +1 -1
- package/dist/builder-eval/index.js +227 -3
- package/dist/builder-eval/index.js.map +1 -1
- package/dist/campaign/index.d.ts +10 -8
- package/dist/campaign/index.js +9 -6
- package/dist/{campaign-C2TTzQII.js → campaign-BYjBAypg.js} +21 -634
- package/dist/campaign-BYjBAypg.js.map +1 -0
- package/dist/{canonical-D011XM8r.js → canonical-D-XsTQ6_.js} +2 -2
- package/dist/{canonical-D011XM8r.js.map → canonical-D-XsTQ6_.js.map} +1 -1
- package/dist/{index-CQTZ-4XN.d.ts → capture-fetch-DDvpjVRU.d.ts} +3 -3
- package/dist/capture-fetch-DDvpjVRU.d.ts.map +1 -0
- package/dist/{default-registry-BmktKy8r.js → chat-client-Bp4Ebuuc.js} +1276 -1276
- package/dist/chat-client-Bp4Ebuuc.js.map +1 -0
- package/dist/cli.js +2 -2
- package/dist/{client-C9gzZE59.d.ts → client-KUbGulm_.d.ts} +4 -4
- package/dist/{client-C9gzZE59.d.ts.map → client-KUbGulm_.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +13 -390
- package/dist/contract/index.d.ts.map +1 -1
- package/dist/contract/index.js +18 -542
- package/dist/contract/index.js.map +1 -1
- package/dist/{cost-ledger-DMFxsLKr.js → cost-ledger-BSe92yAV.js} +3 -3
- package/dist/{cost-ledger-DMFxsLKr.js.map → cost-ledger-BSe92yAV.js.map} +1 -1
- package/dist/{cost-ledger-Bv_e8XHY.d.ts → cost-ledger-DbQdN3nO.d.ts} +2 -2
- package/dist/{cost-ledger-Bv_e8XHY.d.ts.map → cost-ledger-DbQdN3nO.d.ts.map} +1 -1
- package/dist/{counterfactual-CxmxAONP.d.ts → counterfactual--bpysZF0.d.ts} +2 -14
- package/dist/{counterfactual-CxmxAONP.d.ts.map → counterfactual--bpysZF0.d.ts.map} +1 -1
- package/dist/{counterfactual-CWPTrMH7.js → counterfactual-lDfCx0Uz.js} +3 -31
- package/dist/{counterfactual-CWPTrMH7.js.map → counterfactual-lDfCx0Uz.js.map} +1 -1
- package/dist/{dataset-C8xaLXdY.d.ts → dataset-CJjKqQfA.d.ts} +2 -8
- package/dist/dataset-CJjKqQfA.d.ts.map +1 -0
- package/dist/{default-registry-Bf8Woqmq.d.ts → default-registry-Di6HP6pG.d.ts} +7 -12
- package/dist/default-registry-Di6HP6pG.d.ts.map +1 -0
- package/dist/{analyze-runs-C30yljDJ.js → define-agent-eval-iqjT--ZZ.js} +542 -130
- package/dist/define-agent-eval-iqjT--ZZ.js.map +1 -0
- package/dist/define-agent-eval-sH24zBfM.d.ts +388 -0
- package/dist/define-agent-eval-sH24zBfM.d.ts.map +1 -0
- package/dist/descriptive-B5MwKfbf.js +144 -0
- package/dist/descriptive-B5MwKfbf.js.map +1 -0
- package/dist/{dspy-rlm-engine-DbTk4JdR.js → dspy-rlm-engine-B_qhSc21.js} +4 -4
- package/dist/{dspy-rlm-engine-DbTk4JdR.js.map → dspy-rlm-engine-B_qhSc21.js.map} +1 -1
- package/dist/effect-sizes-DiH8MGOH.js +82 -0
- package/dist/effect-sizes-DiH8MGOH.js.map +1 -0
- package/dist/{engine-3hL-XqwJ.d.ts → engine-otFpE2gF.d.ts} +10 -38
- package/dist/engine-otFpE2gF.d.ts.map +1 -0
- package/dist/{errors-CKPfb2aH.d.ts → errors-DEE6u6ot.d.ts} +2 -14
- package/dist/{errors-CKPfb2aH.d.ts.map → errors-DEE6u6ot.d.ts.map} +1 -1
- package/dist/{errors-D-LKuDhb.js → errors-Dngq5h35.js} +2 -8
- package/dist/{errors-D-LKuDhb.js.map → errors-Dngq5h35.js.map} +1 -1
- package/dist/{eval-campaign-B_7wcnav.js → eval-campaign-UB-usSQ2.js} +6 -6
- package/dist/{eval-campaign-B_7wcnav.js.map → eval-campaign-UB-usSQ2.js.map} +1 -1
- package/dist/{exact-types-DSFFpLLI.d.ts → exact-types-BH1twmAJ.d.ts} +2 -2
- package/dist/{exact-types-DSFFpLLI.d.ts.map → exact-types-BH1twmAJ.d.ts.map} +1 -1
- package/dist/experiment/index.d.ts +9 -6
- package/dist/experiment/index.d.ts.map +1 -1
- package/dist/experiment/index.js +11 -7
- package/dist/experiment/index.js.map +1 -1
- package/dist/experiment-tracker-C29gXM4B.js +269 -0
- package/dist/experiment-tracker-C29gXM4B.js.map +1 -0
- package/dist/{experiment-tracker-IMntXr6J.d.ts → experiment-tracker-DWHZBAYL.d.ts} +77 -121
- package/dist/experiment-tracker-DWHZBAYL.d.ts.map +1 -0
- package/dist/{external-optimizer-contracts-CZuJNcT5.d.ts → external-optimizer-contracts-lixrOZdX.d.ts} +3 -3
- package/dist/{external-optimizer-contracts-CZuJNcT5.d.ts.map → external-optimizer-contracts-lixrOZdX.d.ts.map} +1 -1
- package/dist/external-optimizer-process-BTiNB-RH.js +301 -0
- package/dist/external-optimizer-process-BTiNB-RH.js.map +1 -0
- package/dist/{single-run-lock-DFWHEB09.js → external-optimizer-subprocess-DrJ9hR8u.js} +144 -438
- package/dist/external-optimizer-subprocess-DrJ9hR8u.js.map +1 -0
- package/dist/extract-usage-BrQ8mCLX.js +155 -0
- package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
- package/dist/{failure-cluster-CqcvCcdR.d.ts → failure-cluster-BLURuWG4.d.ts} +2 -3
- package/dist/failure-cluster-BLURuWG4.d.ts.map +1 -0
- package/dist/{feedback-trajectory-CacHpxsp.d.ts → feedback-trajectory-juOozjAc.d.ts} +5 -36
- package/dist/feedback-trajectory-juOozjAc.d.ts.map +1 -0
- package/dist/fuzz.d.ts +2 -2
- package/dist/fuzz.js +2 -2
- package/dist/hosted/index.d.ts +3 -3
- package/dist/{index-CvSN3IG1.d.ts → index-B8Ui1mr1.d.ts} +2 -2
- package/dist/{index-CvSN3IG1.d.ts.map → index-B8Ui1mr1.d.ts.map} +1 -1
- package/dist/{skill-usage-CO9OLRBx.d.ts → index-BWDrSVfw.d.ts} +11 -136
- package/dist/index-BWDrSVfw.d.ts.map +1 -0
- package/dist/index-Ba3YrbAL.d.ts +1 -0
- package/dist/{index-BnEuDAK2.d.ts → index-COQYtuRF.d.ts} +3 -3
- package/dist/{index-BnEuDAK2.d.ts.map → index-COQYtuRF.d.ts.map} +1 -1
- package/dist/{index-C3ssXVLv.d.ts → index-CvjYbU0D.d.ts} +2 -2
- package/dist/{index-C3ssXVLv.d.ts.map → index-CvjYbU0D.d.ts.map} +1 -1
- package/dist/{index-DPPGNJ_R.d.ts → index-DSmEylT9.d.ts} +17 -115
- package/dist/index-DSmEylT9.d.ts.map +1 -0
- package/dist/index.d.ts +2397 -5308
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +5914 -10496
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-CgX_s0Ez.d.ts → insight-report-CRi-Ufrj.d.ts} +4 -4
- package/dist/{insight-report-CgX_s0Ez.d.ts.map → insight-report-CRi-Ufrj.d.ts.map} +1 -1
- package/dist/{integrity-MLzHOfV9.js → integrity-Cy9WHAtb.js} +2 -2
- package/dist/{integrity-MLzHOfV9.js.map → integrity-Cy9WHAtb.js.map} +1 -1
- package/dist/{integrity-DY6tIbl0.js → integrity-DysDBWDu.js} +2 -2
- package/dist/{integrity-DY6tIbl0.js.map → integrity-DysDBWDu.js.map} +1 -1
- package/dist/{integrity-BuqEKu-x.d.ts → integrity-OrcI9Nau.d.ts} +3 -3
- package/dist/{integrity-BuqEKu-x.d.ts.map → integrity-OrcI9Nau.d.ts.map} +1 -1
- package/dist/internal-BDHPCnjk.js +230 -0
- package/dist/internal-BDHPCnjk.js.map +1 -0
- package/dist/judge-calibration-C5CbMYce.d.ts +117 -0
- package/dist/judge-calibration-C5CbMYce.d.ts.map +1 -0
- package/dist/judge-calibration-DZkWrm5H.js +317 -0
- package/dist/judge-calibration-DZkWrm5H.js.map +1 -0
- package/dist/{kind-factory-B8-r8-y8.js → kind-factory-CPmSd58s.js} +208 -208
- package/dist/{kind-factory-B8-r8-y8.js.map → kind-factory-CPmSd58s.js.map} +1 -1
- package/dist/ledger-core/index.d.ts +1 -1
- package/dist/ledger-core/index.js +2 -2
- package/dist/{ledger-core-DXZIqu17.js → ledger-core-BmZt19oQ.js} +110 -110
- package/dist/{ledger-core-DXZIqu17.js.map → ledger-core-BmZt19oQ.js.map} +1 -1
- package/dist/{llm-client-DzvMUsS_.js → llm-client-d0-2TT1g.js} +4 -64
- package/dist/{llm-client-DzvMUsS_.js.map → llm-client-d0-2TT1g.js.map} +1 -1
- package/dist/{skillopt-optimization-method-CQdVeM8k.js → llm-judge-dZ8P6nGI.js} +3349 -5401
- package/dist/llm-judge-dZ8P6nGI.js.map +1 -0
- package/dist/matrix/index.d.ts +2 -2
- package/dist/meta-eval/index.d.ts +3 -3
- package/dist/meta-eval/index.js +3 -3
- package/dist/{metrics-C9YY1OcL.js → metrics-Cl0L1KUy.js} +2 -108
- package/dist/{metrics-C9YY1OcL.js.map → metrics-Cl0L1KUy.js.map} +1 -1
- package/dist/{mint-B2O60ACG.js → mint-Dj9Ww_3I.js} +4 -4
- package/dist/{mint-B2O60ACG.js.map → mint-Dj9Ww_3I.js.map} +1 -1
- package/dist/{multi-layer-verifier-DnAqwl0h.d.ts → multi-layer-verifier-DIguZc8Z.d.ts} +3 -3
- package/dist/{multi-layer-verifier-DnAqwl0h.d.ts.map → multi-layer-verifier-DIguZc8Z.d.ts.map} +1 -1
- package/dist/multiplicity-DIWHvysC.d.ts +43 -0
- package/dist/multiplicity-DIWHvysC.d.ts.map +1 -0
- package/dist/multishot/index.d.ts +3 -3
- package/dist/multishot/index.js +1 -1
- package/dist/openapi.json +1 -1
- package/dist/{opencode-sqlite-8r6WUfHc.js → opencode-sqlite-DJWAXLms.js} +2 -2
- package/dist/{opencode-sqlite-8r6WUfHc.js.map → opencode-sqlite-DJWAXLms.js.map} +1 -1
- package/dist/package-version-D7lQHt_-.js +34 -0
- package/dist/package-version-D7lQHt_-.js.map +1 -0
- package/dist/paired-arms-D-XRF_fy.js +1045 -0
- package/dist/paired-arms-D-XRF_fy.js.map +1 -0
- package/dist/paired-promotion-decision-CGzg0cI_.d.ts +251 -0
- package/dist/paired-promotion-decision-CGzg0cI_.d.ts.map +1 -0
- package/dist/paired-tests-BHIhYVdu.js +213 -0
- package/dist/paired-tests-BHIhYVdu.js.map +1 -0
- package/dist/pareto-BqNW3LJR.d.ts +117 -0
- package/dist/pareto-BqNW3LJR.d.ts.map +1 -0
- package/dist/pipelines/index.d.ts +3 -64
- package/dist/pipelines/index.d.ts.map +1 -1
- package/dist/pipelines/index.js +4 -284
- package/dist/pipelines/index.js.map +1 -1
- package/dist/power-and-mde-CHIrXJll.js +195 -0
- package/dist/power-and-mde-CHIrXJll.js.map +1 -0
- package/dist/{promotion-policy-CrLrmys8.js → power-preflight-DEw-uC7q.js} +4 -184
- package/dist/power-preflight-DEw-uC7q.js.map +1 -0
- package/dist/pre-registration-CZwSFQS4.d.ts +577 -0
- package/dist/pre-registration-CZwSFQS4.d.ts.map +1 -0
- package/dist/{prime-protocol-BfSalTfR.js → prime-protocol-6tZTVsWm.js} +72 -21
- package/dist/prime-protocol-6tZTVsWm.js.map +1 -0
- package/dist/produced-state-DU79a81m.js +586 -0
- package/dist/produced-state-DU79a81m.js.map +1 -0
- package/dist/profile-cell.d.ts +1 -1
- package/dist/profile-cell.js +1 -1
- package/dist/promotion-policy-D0nPhkSy.d.ts +134 -0
- package/dist/promotion-policy-D0nPhkSy.d.ts.map +1 -0
- package/dist/promotion-policy-xzA40Evo.js +186 -0
- package/dist/promotion-policy-xzA40Evo.js.map +1 -0
- package/dist/{skillopt-optimization-method-USDKhxSA.d.ts → provenance-CCdxgLDT.d.ts} +57 -478
- package/dist/provenance-CCdxgLDT.d.ts.map +1 -0
- package/dist/registry-oJeeI4-a.d.ts +178 -0
- package/dist/registry-oJeeI4-a.d.ts.map +1 -0
- package/dist/{release-report-DA2BCu5a.d.ts → release-confidence-BFRE5WSp.d.ts} +180 -112
- package/dist/release-confidence-BFRE5WSp.d.ts.map +1 -0
- package/dist/{release-report-BUYmoKo2.js → release-confidence-CxDuiAev.js} +125 -217
- package/dist/release-confidence-CxDuiAev.js.map +1 -0
- package/dist/reporting.d.ts +6 -5
- package/dist/reporting.js +7 -5
- package/dist/{researcher-BchpD55R.d.ts → researcher-DaN4GST-.d.ts} +7 -40
- package/dist/{researcher-BchpD55R.d.ts.map → researcher-DaN4GST-.d.ts.map} +1 -1
- package/dist/{reward-hacking-BDToousL.js → reward-hacking-DNgjilrV.js} +3 -3
- package/dist/reward-hacking-DNgjilrV.js.map +1 -0
- package/dist/{reward-hacking-BI0OMAlo.d.ts → reward-hacking-DSSTuI9r.d.ts} +5 -5
- package/dist/{reward-hacking-BI0OMAlo.d.ts.map → reward-hacking-DSSTuI9r.d.ts.map} +1 -1
- package/dist/rl.d.ts +7 -7
- package/dist/rl.js +11 -10
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +2 -2
- package/dist/rollout/index.js +4 -4
- package/dist/{rollout-DRrksrcV.js → rollout-BWtw0I_6.js} +3 -3
- package/dist/{rollout-DRrksrcV.js.map → rollout-BWtw0I_6.js.map} +1 -1
- package/dist/{rubric-predictive-validity-DvkjPYCe.d.ts → rubric-predictive-validity-BgxtKe4G.d.ts} +2 -2
- package/dist/{rubric-predictive-validity-DvkjPYCe.d.ts.map → rubric-predictive-validity-BgxtKe4G.d.ts.map} +1 -1
- package/dist/{rubric-predictive-validity-BRR632r1.js → rubric-predictive-validity-Cwwyd7ah.js} +2 -2
- package/dist/{rubric-predictive-validity-BRR632r1.js.map → rubric-predictive-validity-Cwwyd7ah.js.map} +1 -1
- package/dist/{run-record-BmSPWXJR.js → run-record-BvHPVS-i.js} +3 -3
- package/dist/{run-record-BmSPWXJR.js.map → run-record-BvHPVS-i.js.map} +1 -1
- package/dist/{run-record-CF4Dwpxr.d.ts → run-record-CKiihE6f.d.ts} +4 -4
- package/dist/{run-record-CF4Dwpxr.d.ts.map → run-record-CKiihE6f.d.ts.map} +1 -1
- package/dist/{proposal-findings-2GIUo1et.js → run-score-lDzV0X8j.js} +31 -31
- package/dist/run-score-lDzV0X8j.js.map +1 -0
- package/dist/sandbox-harness-BlSOu4LX.d.ts +70 -0
- package/dist/sandbox-harness-BlSOu4LX.d.ts.map +1 -0
- package/dist/{schema-Cef2cFmb2.d.ts → schema-Cef2cFmb.d.ts} +1 -1
- package/dist/schema-Cef2cFmb.d.ts.map +1 -0
- package/dist/semantic-concept-judge-BiJxScqe.js +406 -0
- package/dist/semantic-concept-judge-BiJxScqe.js.map +1 -0
- package/dist/{sequential-D-BLJBKU.js → sequential-C458DXNf.js} +4 -3
- package/dist/{sequential-D-BLJBKU.js.map → sequential-C458DXNf.js.map} +1 -1
- package/dist/sequential-eprocess-CbUt2htw.js +83 -0
- package/dist/sequential-eprocess-CbUt2htw.js.map +1 -0
- package/dist/series-convergence-D1cL1f-4.d.ts +129 -0
- package/dist/series-convergence-D1cL1f-4.d.ts.map +1 -0
- package/dist/{server-iu0ede49.js → server-ulsOdrTI.js} +5 -21
- package/dist/server-ulsOdrTI.js.map +1 -0
- package/dist/skillopt-optimization-method-C5cotF4E.d.ts +427 -0
- package/dist/skillopt-optimization-method-C5cotF4E.d.ts.map +1 -0
- package/dist/skillopt-optimization-method-jjdnc3YK.js +1999 -0
- package/dist/skillopt-optimization-method-jjdnc3YK.js.map +1 -0
- package/dist/{statistical-heldout-TQ-4CYiN.d.ts → statistical-heldout-DoFF5KJX.d.ts} +5 -278
- package/dist/statistical-heldout-DoFF5KJX.d.ts.map +1 -0
- package/dist/{store-otlp-Dw8PPIlL.js → store-otlp-CsptLYpN.js} +3 -3
- package/dist/{store-otlp-Dw8PPIlL.js.map → store-otlp-CsptLYpN.js.map} +1 -1
- package/dist/store-tool-spans-Cq9mFd-q.js +667 -0
- package/dist/store-tool-spans-Cq9mFd-q.js.map +1 -0
- package/dist/store-tool-spans-DbDLOBOb.d.ts +342 -0
- package/dist/store-tool-spans-DbDLOBOb.d.ts.map +1 -0
- package/dist/student-t-CvBq2mve.js +38 -0
- package/dist/student-t-CvBq2mve.js.map +1 -0
- package/dist/{summary-report-Lf-5I7xh.js → summary-report-Blysd6Z2.js} +6 -3
- package/dist/{summary-report-Lf-5I7xh.js.map → summary-report-Blysd6Z2.js.map} +1 -1
- package/dist/{summary-report-BNR7DWTj.d.ts → summary-report-CaL-Hnxt.d.ts} +5 -5
- package/dist/{summary-report-BNR7DWTj.d.ts.map → summary-report-CaL-Hnxt.d.ts.map} +1 -1
- package/dist/supervisor-run/index.d.ts +391 -3
- package/dist/supervisor-run/index.d.ts.map +1 -0
- package/dist/supervisor-run/index.js +1689 -2
- package/dist/{supervisor-run-D_sokXcO.js.map → supervisor-run/index.js.map} +1 -1
- package/dist/{extract-usage-CdZdoj1s.js → task-failure-attributes-CpQ4y5RD.js} +5 -157
- package/dist/task-failure-attributes-CpQ4y5RD.js.map +1 -0
- package/dist/{tool-groups-CZPGGlHf.d.ts → tool-groups-Cteb03Ps.d.ts} +3 -3
- package/dist/tool-groups-Cteb03Ps.d.ts.map +1 -0
- package/dist/tool-waste-BDdBZG1F.js +803 -0
- package/dist/tool-waste-BDdBZG1F.js.map +1 -0
- package/dist/tool-waste-DjRDEsuI.d.ts +128 -0
- package/dist/tool-waste-DjRDEsuI.d.ts.map +1 -0
- package/dist/trace-repair/index.d.ts +14 -5
- package/dist/trace-repair/index.d.ts.map +1 -1
- package/dist/trace-repair/index.js +35 -7
- package/dist/trace-repair/index.js.map +1 -1
- package/dist/traces.d.ts +406 -7
- package/dist/traces.d.ts.map +1 -0
- package/dist/traces.js +1011 -10
- package/dist/traces.js.map +1 -0
- package/dist/trajectory-replay/index.d.ts +16 -3
- package/dist/trajectory-replay/index.d.ts.map +1 -1
- package/dist/trajectory-replay/index.js +52 -5
- package/dist/trajectory-replay/index.js.map +1 -1
- package/dist/types-BEPZc6eo.d.ts +93 -0
- package/dist/types-BEPZc6eo.d.ts.map +1 -0
- package/dist/{usage-receipt-t7vAzCRQ.js → types-BI4fT3HN.js} +67 -67
- package/dist/types-BI4fT3HN.js.map +1 -0
- package/dist/{types-BnjdJ70P.d.ts → types-BZ59Ahr8.d.ts} +5 -5
- package/dist/{types-BnjdJ70P.d.ts.map → types-BZ59Ahr8.d.ts.map} +1 -1
- package/dist/{types-D216SgwM.d.ts → types-Cx3YUh2r.d.ts} +4 -241
- package/dist/types-Cx3YUh2r.d.ts.map +1 -0
- package/dist/{types-BvDKaULh.d.ts → types-DLQx4mKU.d.ts} +4 -4
- package/dist/{types-BvDKaULh.d.ts.map → types-DLQx4mKU.d.ts.map} +1 -1
- package/dist/{types-D4mog56g.d.ts → types-yLK8gXE9.d.ts} +2 -2
- package/dist/{types-D4mog56g.d.ts.map → types-yLK8gXE9.d.ts.map} +1 -1
- package/dist/verdict-BndeTAh_.js +61 -0
- package/dist/verdict-BndeTAh_.js.map +1 -0
- package/dist/verdict-E4eRNf7-.d.ts +392 -0
- package/dist/verdict-E4eRNf7-.d.ts.map +1 -0
- package/dist/{verdict-cache-BCcOh0kF.js → verdict-cache-mZf5FEiY.js} +3 -55
- package/dist/{verdict-cache-BCcOh0kF.js.map → verdict-cache-mZf5FEiY.js.map} +1 -1
- package/dist/wire/index.d.ts +3 -3
- package/dist/wire/index.d.ts.map +1 -1
- package/dist/wire/index.js +1 -1
- package/docs/charter.md +3 -3
- package/docs/control-runtime.md +3 -42
- package/docs/experiment.md +0 -1
- package/docs/feature-guide.md +2 -2
- package/docs/trace-repair-grader.md +1 -0
- package/docs/trajectory-replay.md +1 -0
- package/docs/verdicts.md +43 -0
- package/docs/verification-strategies.md +3 -2
- package/package.json +6 -11
- package/dist/agent-profile-DPi7IZg7.d.ts.map +0 -1
- package/dist/analyze-runs-C30yljDJ.js.map +0 -1
- package/dist/baseline-CavEbRyH.d.ts +0 -136
- package/dist/baseline-CavEbRyH.d.ts.map +0 -1
- package/dist/benchmark-command-BteMFN62.js.map +0 -1
- package/dist/benchmarks-Dzs8CKb1.js +0 -755
- package/dist/benchmarks-Dzs8CKb1.js.map +0 -1
- package/dist/campaign-C2TTzQII.js.map +0 -1
- package/dist/completion-verifier-DJA5BhPb.d.ts +0 -414
- package/dist/completion-verifier-DJA5BhPb.d.ts.map +0 -1
- package/dist/control.d.ts +0 -3
- package/dist/control.js +0 -2
- package/dist/dataset-C8xaLXdY.d.ts.map +0 -1
- package/dist/default-registry-Bf8Woqmq.d.ts.map +0 -1
- package/dist/default-registry-BmktKy8r.js.map +0 -1
- package/dist/engine-3hL-XqwJ.d.ts.map +0 -1
- package/dist/experiment-tracker-CnRICnMl.js +0 -500
- package/dist/experiment-tracker-CnRICnMl.js.map +0 -1
- package/dist/experiment-tracker-IMntXr6J.d.ts.map +0 -1
- package/dist/extract-usage-CdZdoj1s.js.map +0 -1
- package/dist/failure-cluster-CqcvCcdR.d.ts.map +0 -1
- package/dist/feedback-trajectory-CacHpxsp.d.ts.map +0 -1
- package/dist/index-BZ3-y4YL.d.ts +0 -391
- package/dist/index-BZ3-y4YL.d.ts.map +0 -1
- package/dist/index-CQTZ-4XN.d.ts.map +0 -1
- package/dist/index-DPPGNJ_R.d.ts.map +0 -1
- package/dist/index-YE4KdKbO2.d.ts +0 -335
- package/dist/index-YE4KdKbO2.d.ts.map +0 -1
- package/dist/paired-arms-iZ08VFMN.js +0 -260
- package/dist/paired-arms-iZ08VFMN.js.map +0 -1
- package/dist/paired-promotion-decision-B6zJ3gYM.d.ts +0 -114
- package/dist/paired-promotion-decision-B6zJ3gYM.d.ts.map +0 -1
- package/dist/prime-protocol-BfSalTfR.js.map +0 -1
- package/dist/promotion-policy-Ckjhzg_4.d.ts +0 -289
- package/dist/promotion-policy-Ckjhzg_4.d.ts.map +0 -1
- package/dist/promotion-policy-CrLrmys8.js.map +0 -1
- package/dist/proposal-findings-2GIUo1et.js.map +0 -1
- package/dist/propose-review-control-dSNPjFUH.js +0 -1458
- package/dist/propose-review-control-dSNPjFUH.js.map +0 -1
- package/dist/release-report-BUYmoKo2.js.map +0 -1
- package/dist/release-report-DA2BCu5a.d.ts.map +0 -1
- package/dist/replay-CohS93nE.js +0 -1859
- package/dist/replay-CohS93nE.js.map +0 -1
- package/dist/replay-DbhZ4Ked.d.ts +0 -834
- package/dist/replay-DbhZ4Ked.d.ts.map +0 -1
- package/dist/reward-hacking-BDToousL.js.map +0 -1
- package/dist/run-evidence-Dhi3C81V.d.ts +0 -225
- package/dist/run-evidence-Dhi3C81V.d.ts.map +0 -1
- package/dist/schema-Cef2cFmb2.d.ts.map +0 -1
- package/dist/semantic-concept-judge-D1z-KepS.js +0 -767
- package/dist/semantic-concept-judge-D1z-KepS.js.map +0 -1
- package/dist/series-convergence-ofsqPWhs.d.ts +0 -35
- package/dist/series-convergence-ofsqPWhs.d.ts.map +0 -1
- package/dist/server-iu0ede49.js.map +0 -1
- package/dist/single-run-lock-DFWHEB09.js.map +0 -1
- package/dist/skill-usage-CO9OLRBx.d.ts.map +0 -1
- package/dist/skillopt-optimization-method-CQdVeM8k.js.map +0 -1
- package/dist/skillopt-optimization-method-USDKhxSA.d.ts.map +0 -1
- package/dist/statistical-heldout-TQ-4CYiN.d.ts.map +0 -1
- package/dist/statistics-ByxzSiOM.js +0 -2212
- package/dist/statistics-ByxzSiOM.js.map +0 -1
- package/dist/statistics-D6Uebe_4.d.ts +0 -968
- package/dist/statistics-D6Uebe_4.d.ts.map +0 -1
- package/dist/supervisor-run-D_sokXcO.js +0 -1690
- package/dist/test-graded-scenario-D1TaI2va.d.ts +0 -141
- package/dist/test-graded-scenario-D1TaI2va.d.ts.map +0 -1
- package/dist/test-graded-scenario-JHcKQNpq.js +0 -318
- package/dist/test-graded-scenario-JHcKQNpq.js.map +0 -1
- package/dist/tool-groups-CZPGGlHf.d.ts.map +0 -1
- package/dist/tool-use-metrics-DEGMKycK.js +0 -370
- package/dist/tool-use-metrics-DEGMKycK.js.map +0 -1
- package/dist/types-D216SgwM.d.ts.map +0 -1
- package/dist/usage-receipt-t7vAzCRQ.js.map +0 -1
- package/dist/verdict-DExhxfgR.d.ts +0 -201
- package/dist/verdict-DExhxfgR.d.ts.map +0 -1
|
@@ -1,968 +0,0 @@
|
|
|
1
|
-
import { g as JudgeScore } from "./types-D216SgwM.js";
|
|
2
|
-
//#region src/judge-calibration.d.ts
|
|
3
|
-
/**
|
|
4
|
-
* Judge calibration — measure judge quality against human gold + bias.
|
|
5
|
-
*
|
|
6
|
-
* Workflow:
|
|
7
|
-
* 1. Build a golden set: {itemId, humanScore}[].
|
|
8
|
-
* 2. Run candidate judges; each produces {itemId, score}.
|
|
9
|
-
* 3. `calibrateJudge(golden, candidate)` reports κ + Pearson + MAE.
|
|
10
|
-
* 4. `calibrateJudgeContinuous(golden, candidate)` adds quadratic-weighted
|
|
11
|
-
* κ over the un-rounded [0,1] scores plus ICC(2,1), Pearson, Spearman,
|
|
12
|
-
* and bootstrap CIs — use this for fine-grained judges where rounding
|
|
13
|
-
* to int discards information (e.g. 0.78 vs 0.81 both round to 1 and
|
|
14
|
-
* look "perfectly agreed" to integer κ).
|
|
15
|
-
* 5. Run bias probes (positional, verbosity, self-preference) to
|
|
16
|
-
* detect systematic score inflation.
|
|
17
|
-
* 6. For N≥2 judges on the same items, `continuousAgreement(scores)`
|
|
18
|
-
* reports ICC(2,1) + κ_w + Pearson + Spearman with bootstrap CIs.
|
|
19
|
-
*
|
|
20
|
-
* Returns actionable diagnostics, not a single number. Consumers then
|
|
21
|
-
* decide whether to trust the judge, retrain it, or add a tie-breaker.
|
|
22
|
-
*/
|
|
23
|
-
interface GoldenItem {
|
|
24
|
-
itemId: string;
|
|
25
|
-
humanScore: number;
|
|
26
|
-
/** Optional group used for per-group bias audits (e.g. model-of-output family). */
|
|
27
|
-
group?: string;
|
|
28
|
-
}
|
|
29
|
-
interface CandidateScore {
|
|
30
|
-
itemId: string;
|
|
31
|
-
score: number;
|
|
32
|
-
/** Optional — enables positional-bias analysis (did order matter?). */
|
|
33
|
-
positionOfAInput?: 'first' | 'second';
|
|
34
|
-
}
|
|
35
|
-
interface CalibrationResult {
|
|
36
|
-
n: number;
|
|
37
|
-
pearson: number;
|
|
38
|
-
/** Cohen's κ with quadratic weights over integer-rounded scores. */
|
|
39
|
-
kappa: number;
|
|
40
|
-
/** Mean absolute error vs human. */
|
|
41
|
-
mae: number;
|
|
42
|
-
/** Worst-5 miscalibrations (largest |judge - human|). */
|
|
43
|
-
worstItems: Array<{
|
|
44
|
-
itemId: string;
|
|
45
|
-
judge: number;
|
|
46
|
-
human: number;
|
|
47
|
-
delta: number;
|
|
48
|
-
}>;
|
|
49
|
-
}
|
|
50
|
-
/**
|
|
51
|
-
* Measure judge quality against human gold labels: computes Cohen's κ, Pearson correlation, and MAE over matched item ids.
|
|
52
|
-
*/
|
|
53
|
-
declare function calibrateJudge(golden: GoldenItem[], candidate: CandidateScore[]): CalibrationResult;
|
|
54
|
-
interface PositionalBiasResult {
|
|
55
|
-
/**
|
|
56
|
-
* Score delta (first-position - second-position) averaged across items
|
|
57
|
-
* presented in both positions. Non-zero = positional bias.
|
|
58
|
-
*/
|
|
59
|
-
avgDelta: number;
|
|
60
|
-
n: number;
|
|
61
|
-
}
|
|
62
|
-
/**
|
|
63
|
-
* Feed the same items to the judge twice with A/B swapped and pass all
|
|
64
|
-
* results here. Items that don't appear in both positions are ignored.
|
|
65
|
-
*/
|
|
66
|
-
declare function positionalBias(scores: CandidateScore[]): PositionalBiasResult;
|
|
67
|
-
interface VerbosityBiasResult {
|
|
68
|
-
/** Pearson correlation between output length and score. Strong positive = verbosity bias. */
|
|
69
|
-
pearson: number;
|
|
70
|
-
n: number;
|
|
71
|
-
}
|
|
72
|
-
declare function verbosityBias(samples: Array<{
|
|
73
|
-
outputLen: number;
|
|
74
|
-
score: number;
|
|
75
|
-
}>): VerbosityBiasResult;
|
|
76
|
-
interface SelfPreferenceResult {
|
|
77
|
-
/** Mean judge score when judge's family matches output's family. */
|
|
78
|
-
inFamilyMean: number;
|
|
79
|
-
outOfFamilyMean: number;
|
|
80
|
-
deltaMean: number;
|
|
81
|
-
n: number;
|
|
82
|
-
}
|
|
83
|
-
/**
|
|
84
|
-
* Pass the same scenarios scored with judge-model X grading outputs from
|
|
85
|
-
* model X (in-family) and model Y (out-of-family). Non-zero delta
|
|
86
|
-
* indicates self-preference.
|
|
87
|
-
*/
|
|
88
|
-
declare function selfPreference(samples: Array<{
|
|
89
|
-
score: number;
|
|
90
|
-
inFamily: boolean;
|
|
91
|
-
}>): SelfPreferenceResult;
|
|
92
|
-
interface ContinuousAgreement {
|
|
93
|
-
/** Cohen's κ_w with quadratic weights, computed on raw [0,1] scores. */
|
|
94
|
-
weightedKappa: number;
|
|
95
|
-
/** ICC(2,1): two-way random effects, absolute agreement, single rater. */
|
|
96
|
-
icc: number;
|
|
97
|
-
/** Pearson product-moment correlation (averaged over rater pairs if N>2). */
|
|
98
|
-
pearson: number;
|
|
99
|
-
/** Spearman rank correlation (averaged over rater pairs if N>2). */
|
|
100
|
-
spearman: number;
|
|
101
|
-
/** 95% bootstrap percentile CIs over items. */
|
|
102
|
-
ci: {
|
|
103
|
-
icc: [number, number];
|
|
104
|
-
weightedKappa: [number, number];
|
|
105
|
-
};
|
|
106
|
-
/** Number of complete items (no NaN across raters). */
|
|
107
|
-
n: number;
|
|
108
|
-
/** Number of raters. */
|
|
109
|
-
raters: number;
|
|
110
|
-
}
|
|
111
|
-
interface ContinuousAgreementOptions {
|
|
112
|
-
/** Bootstrap iterations. Default 1000. Set to 0 to skip CIs (CI = [NaN, NaN]). */
|
|
113
|
-
bootstrap?: number;
|
|
114
|
-
/** κ weighting scheme. Default 'quadratic'. */
|
|
115
|
-
weights?: 'linear' | 'quadratic';
|
|
116
|
-
/** PRNG seed for reproducible bootstrap. Default 0xC0FFEE. */
|
|
117
|
-
seed?: number;
|
|
118
|
-
/** Confidence level for percentile CI. Default 0.95. */
|
|
119
|
-
ciLevel?: number;
|
|
120
|
-
}
|
|
121
|
-
/**
|
|
122
|
-
* Inter-rater agreement on continuous (typically [0,1]) scores.
|
|
123
|
-
*
|
|
124
|
-
* `scores` has shape [n_items][n_raters]. Rows with any non-finite entry
|
|
125
|
-
* are dropped. Returns NaN metrics if fewer than 2 raters or 2 complete
|
|
126
|
-
* items remain.
|
|
127
|
-
*/
|
|
128
|
-
declare function continuousAgreement(scores: number[][], opts?: ContinuousAgreementOptions): ContinuousAgreement;
|
|
129
|
-
interface ContinuousCalibrationResult extends CalibrationResult {
|
|
130
|
-
/** Cohen's κ_w computed on raw (un-rounded) scores. */
|
|
131
|
-
weightedKappaContinuous: number;
|
|
132
|
-
/** ICC(2,1) treating golden + candidate as two raters. */
|
|
133
|
-
icc: number;
|
|
134
|
-
spearman: number;
|
|
135
|
-
ci: {
|
|
136
|
-
icc: [number, number];
|
|
137
|
-
weightedKappa: [number, number];
|
|
138
|
-
};
|
|
139
|
-
}
|
|
140
|
-
/**
|
|
141
|
-
* Extends `calibrateJudge` with continuous-value agreement metrics while
|
|
142
|
-
* retaining its base calibration summary.
|
|
143
|
-
*/
|
|
144
|
-
declare function calibrateJudgeContinuous(golden: GoldenItem[], candidate: CandidateScore[], opts?: ContinuousAgreementOptions): ContinuousCalibrationResult;
|
|
145
|
-
//#endregion
|
|
146
|
-
//#region src/statistics.d.ts
|
|
147
|
-
/** Identity: dimensions already follow "higher = better" by prompt convention
|
|
148
|
-
* (inverted dims like hallucination are scored 10 = best at the source). */
|
|
149
|
-
declare const normalizeScores: (scores: JudgeScore[]) => JudgeScore[];
|
|
150
|
-
/** Weighted mean — falls back to uniform weights when omitted */
|
|
151
|
-
declare function weightedMean(scores: {
|
|
152
|
-
score: number;
|
|
153
|
-
weight?: number;
|
|
154
|
-
}[]): number;
|
|
155
|
-
/**
|
|
156
|
-
* Percentile bootstrap confidence interval on the mean of `scores`.
|
|
157
|
-
*
|
|
158
|
-
* Descriptive spread. It is not a significance test, and at small n its bounds
|
|
159
|
-
* are anti-conservative in the same way {@link pairedBootstrap}'s are — see
|
|
160
|
-
* {@link BOOTSTRAP_GATE_MIN_N}. With no `seed` the resampling is seeded from
|
|
161
|
-
* the scores themselves, so the interval is reproducible either way.
|
|
162
|
-
*/
|
|
163
|
-
declare function confidenceInterval(scores: number[], confidence?: number, opts?: {
|
|
164
|
-
seed?: number;
|
|
165
|
-
resamples?: number;
|
|
166
|
-
}): {
|
|
167
|
-
mean: number;
|
|
168
|
-
lower: number;
|
|
169
|
-
upper: number;
|
|
170
|
-
};
|
|
171
|
-
/**
|
|
172
|
-
* Inter-rater reliability — Krippendorff's α under the squared-difference
|
|
173
|
-
* metric, pooled across dimensions.
|
|
174
|
-
*
|
|
175
|
-
* Each inner array is one judge's scores. Items are matched by position
|
|
176
|
-
* WITHIN a dimension: the k-th score a judge supplies carrying dimension
|
|
177
|
-
* `d` is item k of `d`, and the ratings compared against each other are
|
|
178
|
-
* the ones different judges gave to the same item. Every judge that scores
|
|
179
|
-
* a dimension at all must supply the same number of scores for it —
|
|
180
|
-
* ragged input cannot be aligned into items and throws rather than
|
|
181
|
-
* comparing mismatched items.
|
|
182
|
-
*
|
|
183
|
-
* α = 1 − D_observed / D_expected: D_observed averages the squared
|
|
184
|
-
* difference over within-item judge pairs, D_expected over every pair of
|
|
185
|
-
* ratings irrespective of item. α = 1 is perfect agreement, 0 is chance,
|
|
186
|
-
* negative is systematic disagreement.
|
|
187
|
-
*/
|
|
188
|
-
declare function interRaterReliability(judgeScores: JudgeScore[][]): number;
|
|
189
|
-
/** How a rank test's p-value was actually computed. */
|
|
190
|
-
type RankTestMethod = 'exact' | 'permutation' | 'asymptotic';
|
|
191
|
-
/**
|
|
192
|
-
* What the caller asks for. `'auto'` selects `'exact'` inside the enumeration
|
|
193
|
-
* threshold and `'permutation'` above it, and never selects `'asymptotic'`.
|
|
194
|
-
*/
|
|
195
|
-
type RankTestMethodRequest = 'auto' | 'exact' | 'asymptotic';
|
|
196
|
-
interface RankTestOptions {
|
|
197
|
-
/** Default `'auto'`. `'asymptotic'` inside the exact-feasible range throws. */
|
|
198
|
-
method?: RankTestMethodRequest;
|
|
199
|
-
/** Resamples on the Monte Carlo permutation path. Default 100000. */
|
|
200
|
-
permutations?: number;
|
|
201
|
-
/** Seed for the permutation path. Omitted ⇒ derived from the data itself, so
|
|
202
|
-
* the result is reproducible either way. */
|
|
203
|
-
seed?: number;
|
|
204
|
-
}
|
|
205
|
-
/** Maximum dynamic-programming cells used by an exact two-sample rank test. */
|
|
206
|
-
declare const MANN_WHITNEY_EXACT_MAX_STATES = 8192;
|
|
207
|
-
/** Maximum inner-loop transitions used by an exact two-sample rank test. */
|
|
208
|
-
declare const MANN_WHITNEY_EXACT_MAX_WORK = 250000;
|
|
209
|
-
/** Non-zero differences up to which the signed-rank null is enumerated exactly. */
|
|
210
|
-
declare const WILCOXON_EXACT_MAX_N = 20;
|
|
211
|
-
/** Resamples used when a rank test falls back to Monte Carlo permutation. */
|
|
212
|
-
declare const DEFAULT_PERMUTATIONS = 100000;
|
|
213
|
-
interface MannWhitneyResult {
|
|
214
|
-
/** `min(U_a, U_b)` — the conventional reported statistic. */
|
|
215
|
-
u: number;
|
|
216
|
-
/** U for sample `a`. Carries the direction of the effect, which `u` discards. */
|
|
217
|
-
uA: number;
|
|
218
|
-
/** Two-sided p-value. */
|
|
219
|
-
p: number;
|
|
220
|
-
/** How `p` was computed. */
|
|
221
|
-
method: RankTestMethod;
|
|
222
|
-
/** Smallest two-sided p this design can produce. `p` can never be below it. */
|
|
223
|
-
pFloor: number;
|
|
224
|
-
}
|
|
225
|
-
/**
|
|
226
|
-
* Mann-Whitney U — two independent samples, no distributional assumption.
|
|
227
|
-
*
|
|
228
|
-
* Exact conditional (permutation) p by default when the dynamic program fits
|
|
229
|
-
* {@link MANN_WHITNEY_EXACT_MAX_STATES} cells and
|
|
230
|
-
* {@link MANN_WHITNEY_EXACT_MAX_WORK} transitions, seeded Monte Carlo
|
|
231
|
-
* permutation above those limits. This keeps imbalanced designs such as 1+24
|
|
232
|
-
* exact without admitting expensive balanced designs merely because they have
|
|
233
|
-
* the same total size. Throws on non-finite input and on `method:
|
|
234
|
-
* 'asymptotic'` where an exact answer is available. Empty input yields `p = 1,
|
|
235
|
-
* pFloor = 1` — no design, no attainable evidence.
|
|
236
|
-
*/
|
|
237
|
-
declare function mannWhitneyU(a: number[], b: number[], opts?: RankTestOptions): MannWhitneyResult;
|
|
238
|
-
/** Partial credit: returns 0-1 ratio of current toward target */
|
|
239
|
-
declare function partialCredit(current: number, target: number): number;
|
|
240
|
-
interface PairedTTestResult {
|
|
241
|
-
/** Null when the statistic is undefined — see {@link pairedTTest}. */
|
|
242
|
-
t: number | null;
|
|
243
|
-
df: number;
|
|
244
|
-
/** Null exactly when `t` is null. */
|
|
245
|
-
p: number | null;
|
|
246
|
-
}
|
|
247
|
-
/**
|
|
248
|
-
* Paired t-test — before/after measurements on the SAME items.
|
|
249
|
-
* Pairing removes inter-item variance, giving tighter significance than
|
|
250
|
-
* an unpaired test when comparing prompt v1 vs prompt v2 on identical
|
|
251
|
-
* scenarios.
|
|
252
|
-
*
|
|
253
|
-
* Returns `t = p = null` where the statistic is undefined: fewer than two
|
|
254
|
-
* pairs, or a non-zero constant delta whose observed variance is zero. A
|
|
255
|
-
* constant shift carries no information about the variance it would have to
|
|
256
|
-
* be compared against, so the honest answer is "undefined", not `p = 0` —
|
|
257
|
-
* three observations cannot buy absolute certainty. This is the same contract
|
|
258
|
-
* {@link pairedCohensDz} states for the same condition. An all-zero delta is
|
|
259
|
-
* different: it is a measured null, and returns `t = 0, p = 1`.
|
|
260
|
-
*/
|
|
261
|
-
declare function pairedTTest(before: number[], after: number[]): PairedTTestResult;
|
|
262
|
-
interface WilcoxonSignedRankResult {
|
|
263
|
-
/** W⁺, the rank sum of the positive differences. (scipy reports
|
|
264
|
-
* `min(W⁺, W⁻)`; compare statistics only after converting.) */
|
|
265
|
-
w: number;
|
|
266
|
-
/** Two-sided p-value. */
|
|
267
|
-
p: number;
|
|
268
|
-
/** How `p` was computed. */
|
|
269
|
-
method: RankTestMethod;
|
|
270
|
-
/** Smallest two-sided p this design can produce. */
|
|
271
|
-
pFloor: number;
|
|
272
|
-
/** Non-zero differences — zero differences are dropped and carry no rank. */
|
|
273
|
-
nNonZero: number;
|
|
274
|
-
}
|
|
275
|
-
/**
|
|
276
|
-
* Wilcoxon signed-rank — paired, no distributional assumption on the deltas.
|
|
277
|
-
*
|
|
278
|
-
* Exact conditional (sign-flip) p by default at `n ≤
|
|
279
|
-
* {@link WILCOXON_EXACT_MAX_N}` non-zero differences, seeded Monte Carlo
|
|
280
|
-
* permutation above it. Throws on non-finite input and on `method:
|
|
281
|
-
* 'asymptotic'` where an exact answer is available.
|
|
282
|
-
*
|
|
283
|
-
* `n` is the count of NON-ZERO differences: exact ties are dropped before
|
|
284
|
-
* ranking, so a run of tied pairs shrinks the design and raises `pFloor`.
|
|
285
|
-
* All-tied input yields `p = 1, pFloor = 1` — no attainable evidence, which
|
|
286
|
-
* `pFloor` states rather than leaving `p = 1` to be read as a measured null.
|
|
287
|
-
*/
|
|
288
|
-
declare function wilcoxonSignedRank(before: number[], after: number[], opts?: RankTestOptions): WilcoxonSignedRankResult;
|
|
289
|
-
/**
|
|
290
|
-
* Cohen's d — standardized effect size for two independent groups.
|
|
291
|
-
* Positive d means group b has higher mean than group a.
|
|
292
|
-
* Rule of thumb: |d| < 0.2 negligible, 0.2–0.5 small, 0.5–0.8 medium, > 0.8 large.
|
|
293
|
-
*
|
|
294
|
-
* Returns null where the standardized effect is undefined: fewer than two
|
|
295
|
-
* observations in either group, or a zero pooled standard deviation with
|
|
296
|
-
* unequal means. Null is NOT "no effect" — zero within-group spread across a
|
|
297
|
-
* real mean gap is an unbounded effect, the opposite of negligible. Equal
|
|
298
|
-
* means with zero spread is a genuine 0. Same contract as
|
|
299
|
-
* {@link pairedCohensDz}.
|
|
300
|
-
*/
|
|
301
|
-
declare function cohensD(a: number[], b: number[]): number | null;
|
|
302
|
-
/**
|
|
303
|
-
* Cohen's dz for paired observations: mean(after - before) divided by the
|
|
304
|
-
* sample standard deviation of those within-pair deltas.
|
|
305
|
-
*
|
|
306
|
-
* Returns null when fewer than two pairs exist or a non-zero constant delta
|
|
307
|
-
* has zero observed variance. In that case the standardized effect is
|
|
308
|
-
* undefined, not an arbitrarily large finite number.
|
|
309
|
-
*/
|
|
310
|
-
declare function pairedCohensDz(before: number[], after: number[]): number | null;
|
|
311
|
-
type CliffsMagnitude = 'negligible' | 'small' | 'medium' | 'large';
|
|
312
|
-
/**
|
|
313
|
-
* Cliff's delta — a non-parametric effect size for two independent samples.
|
|
314
|
-
* `δ = (#(after > before) − #(after < before)) / (n_before · n_after)`,
|
|
315
|
-
* ranging [-1, 1]. Positive ⇒ `after` tends to exceed `before` (improvement).
|
|
316
|
-
*
|
|
317
|
-
* Distribution-free counterpart to Cohen's d: no normality assumption, robust
|
|
318
|
-
* to the bounded/skewed score distributions judges produce. Pairs with
|
|
319
|
-
* `pairedBootstrap` / `wilcoxonSignedRank` for the non-parametric reporting
|
|
320
|
-
* path. Returns 0 when either sample is empty.
|
|
321
|
-
*/
|
|
322
|
-
declare function cliffsDelta(before: number[], after: number[]): number;
|
|
323
|
-
/**
|
|
324
|
-
* Map a Cliff's delta to a qualitative magnitude using the standard
|
|
325
|
-
* Romano et al. thresholds (|δ|): <0.147 negligible, <0.33 small,
|
|
326
|
-
* <0.474 medium, else large.
|
|
327
|
-
*/
|
|
328
|
-
declare function interpretCliffs(delta: number): CliffsMagnitude;
|
|
329
|
-
/**
|
|
330
|
-
* Average-rank-with-ties transform (1-indexed). Tied values receive the mean
|
|
331
|
-
* of the ranks they span, the standard correction for Spearman's ρ.
|
|
332
|
-
*/
|
|
333
|
-
declare function ranks(xs: number[]): number[];
|
|
334
|
-
/**
|
|
335
|
-
* Pearson product-moment correlation coefficient r ∈ [-1, 1] between two
|
|
336
|
-
* equal-length series. See the edge-case contract above: NaN for n < 2 or
|
|
337
|
-
* unequal lengths, 1 when both series are constant, 0 when exactly one is.
|
|
338
|
-
*/
|
|
339
|
-
declare function pearsonR(a: number[], b: number[]): number;
|
|
340
|
-
/**
|
|
341
|
-
* Spearman's rank correlation ρ — Pearson over the average-rank-with-ties
|
|
342
|
-
* transform of each series. Same edge-case contract as {@link pearsonR}.
|
|
343
|
-
*/
|
|
344
|
-
declare function spearmanR(a: number[], b: number[]): number;
|
|
345
|
-
interface WeightedCompositeInput {
|
|
346
|
-
/** Per-dimension scores (typically 0..1). */
|
|
347
|
-
dims: Record<string, number>;
|
|
348
|
-
/** Weight per dimension. Every weighted dimension MUST be present in
|
|
349
|
-
* `dims` — a weight for an absent dimension is a config error and throws,
|
|
350
|
-
* because silently dropping it would renormalise the composite onto a
|
|
351
|
-
* different denominator than intended. */
|
|
352
|
-
weights: Record<string, number>;
|
|
353
|
-
/** Optional pass threshold; when set, the result reports `pass`. */
|
|
354
|
-
threshold?: number;
|
|
355
|
-
}
|
|
356
|
-
interface WeightedCompositeResult {
|
|
357
|
-
composite: number;
|
|
358
|
-
pass?: boolean;
|
|
359
|
-
}
|
|
360
|
-
/**
|
|
361
|
-
* Weighted composite over judge dimensions: `Σ(score_d · w_d) / Σ(w_d)` across
|
|
362
|
-
* the weighted dimensions. The canonical replacement for the per-consumer
|
|
363
|
-
* hand-rolled composite math (tax/legal/creative/gtm each ship a copy).
|
|
364
|
-
*
|
|
365
|
-
* Fail-loud: throws if a weighted dimension is missing from `dims`, if any
|
|
366
|
-
* weight is negative, or if the weights sum to 0 — none of which can produce
|
|
367
|
-
* a meaningful composite.
|
|
368
|
-
*/
|
|
369
|
-
declare function weightedComposite(input: WeightedCompositeInput): WeightedCompositeResult;
|
|
370
|
-
interface CorpusScoreRecord {
|
|
371
|
-
/** Stable identifier for the rated item (scenario, span, turn, …). */
|
|
372
|
-
itemId: string;
|
|
373
|
-
/** Identifier for the judge that produced this score. */
|
|
374
|
-
judgeName: string;
|
|
375
|
-
/** Dimension name (matches `JudgeScore.dimension`). */
|
|
376
|
-
dimension: string;
|
|
377
|
-
/** Numeric score; must be finite. */
|
|
378
|
-
score: number;
|
|
379
|
-
}
|
|
380
|
-
interface CorpusAgreementPerDimension extends ContinuousAgreement {
|
|
381
|
-
dimension: string;
|
|
382
|
-
/** Item IDs that contributed to this dimension's matrix (every judge scored them). */
|
|
383
|
-
itemIds: string[];
|
|
384
|
-
/** Judge IDs that contributed to this dimension's matrix. */
|
|
385
|
-
judgeIds: string[];
|
|
386
|
-
}
|
|
387
|
-
interface CorpusAgreementReport {
|
|
388
|
-
/** Per-dimension ICC(2,1) + κ_w + Pearson + Spearman + bootstrap CIs. */
|
|
389
|
-
perDimension: CorpusAgreementPerDimension[];
|
|
390
|
-
/** Mean ICC across dimensions (NaN if no dimension yielded a finite ICC). */
|
|
391
|
-
overallIcc: number;
|
|
392
|
-
/** Mean weighted κ across dimensions (NaN if none finite). */
|
|
393
|
-
overallWeightedKappa: number;
|
|
394
|
-
/** Dimensions evaluated (sorted). */
|
|
395
|
-
dimensions: string[];
|
|
396
|
-
/** Judges seen across the corpus (sorted). */
|
|
397
|
-
judgeIds: string[];
|
|
398
|
-
}
|
|
399
|
-
interface CorpusAgreementOptions extends ContinuousAgreementOptions {
|
|
400
|
-
/**
|
|
401
|
-
* Restrict the audit to these dimensions. Default = every dimension
|
|
402
|
-
* that appears in the input. A dimension named here but absent from
|
|
403
|
-
* the input throws — silent omission would corrupt the overall metric.
|
|
404
|
-
*/
|
|
405
|
-
dimensions?: string[];
|
|
406
|
-
/**
|
|
407
|
-
* Restrict the audit to these judges. Default = every judge that
|
|
408
|
-
* appears in the input. A judge named here but absent from a
|
|
409
|
-
* dimension throws (see "fail loud" below).
|
|
410
|
-
*/
|
|
411
|
-
judges?: string[];
|
|
412
|
-
}
|
|
413
|
-
/**
|
|
414
|
-
* Corpus-wide inter-rater agreement across N items × M judges × D dimensions.
|
|
415
|
-
*
|
|
416
|
-
* For each dimension, builds the [n_items][n_judges] matrix of scores
|
|
417
|
-
* (keeping only items every judge rated on that dimension), then runs
|
|
418
|
-
* `continuousAgreement` to get ICC(2,1), κ_w, Pearson, Spearman, and
|
|
419
|
-
* bootstrap CIs. Reports a pooled mean across dimensions as a single
|
|
420
|
-
* "is this judge panel reliable on this corpus?" number.
|
|
421
|
-
*
|
|
422
|
-
* Fail-loud contract:
|
|
423
|
-
* - Empty input throws.
|
|
424
|
-
* - Fewer than 2 judges or fewer than 2 items per dimension throws.
|
|
425
|
-
* - A judge present in some dimensions but with zero scored items on
|
|
426
|
-
* another dimension throws (would silently shrink the matrix).
|
|
427
|
-
* - Duplicate (itemId, judgeName, dimension) records throw.
|
|
428
|
-
*/
|
|
429
|
-
declare function corpusInterRaterAgreement(records: CorpusScoreRecord[], opts?: CorpusAgreementOptions): CorpusAgreementReport;
|
|
430
|
-
/**
|
|
431
|
-
* Convenience adapter for `JudgeScore[]` data keyed externally by item.
|
|
432
|
-
*
|
|
433
|
-
* Use when you have per-item arrays of `JudgeScore[]` (e.g. one
|
|
434
|
-
* `ScenarioResult.judgeScores` per scenario) and want corpus-wide
|
|
435
|
-
* agreement without manually flattening. `itemId` must be unique per
|
|
436
|
-
* row of `itemsScores`.
|
|
437
|
-
*/
|
|
438
|
-
declare function corpusInterRaterAgreementFromJudgeScores(itemsScores: Array<{
|
|
439
|
-
itemId: string;
|
|
440
|
-
scores: JudgeScore[];
|
|
441
|
-
}>, opts?: CorpusAgreementOptions): CorpusAgreementReport;
|
|
442
|
-
/**
|
|
443
|
-
* Required N per arm for a two-sample comparison at target effect size,
|
|
444
|
-
* alpha, and power. Normal-approximation formula:
|
|
445
|
-
* n = 2 * ( (z_{1-α/2} + z_{1-β}) / d )^2
|
|
446
|
-
* where d is Cohen's d. Returns Infinity for effect ≤ 0.
|
|
447
|
-
*/
|
|
448
|
-
declare function requiredSampleSize(opts: {
|
|
449
|
-
effect: number;
|
|
450
|
-
alpha?: number;
|
|
451
|
-
power?: number;
|
|
452
|
-
twoSided?: boolean;
|
|
453
|
-
}): number;
|
|
454
|
-
/**
|
|
455
|
-
* Required number of paired observations for a target Cohen's dz.
|
|
456
|
-
* Unlike the independent-groups formula, this has no two-arm factor of two.
|
|
457
|
-
*
|
|
458
|
-
* Normal quantiles with no t correction, so treat the result as a LOWER bound:
|
|
459
|
-
* it returns 32 where the exact t-based answer is 34 at dz = 0.5, and 13 where
|
|
460
|
-
* it is 15 at dz = 0.8 — a 6–13 % shortfall precisely in the range a caller
|
|
461
|
-
* consults to decide whether 3–10 repetitions suffice.
|
|
462
|
-
*/
|
|
463
|
-
declare function requiredPairedSampleSize(opts: {
|
|
464
|
-
effect: number;
|
|
465
|
-
alpha?: number;
|
|
466
|
-
power?: number;
|
|
467
|
-
twoSided?: boolean;
|
|
468
|
-
}): number;
|
|
469
|
-
/**
|
|
470
|
-
* Minimum detectable paired effect (standardised units) for a target paired
|
|
471
|
-
* sample size: d_min = (z_{1-α/2} + z_β) / sqrt(n_paired). Multiply by
|
|
472
|
-
* sd(deltas) for score units; treat as a lower bound — Wilcoxon and bootstrap
|
|
473
|
-
* have asymptotic relative efficiency below 1 vs the t-test on heavy tails.
|
|
474
|
-
*/
|
|
475
|
-
declare function pairedMde(opts: {
|
|
476
|
-
nPaired: number;
|
|
477
|
-
alpha?: number;
|
|
478
|
-
power?: number;
|
|
479
|
-
twoSided?: boolean;
|
|
480
|
-
}): number;
|
|
481
|
-
/**
|
|
482
|
-
* Number of paired observations needed for a McNemar test to reach a target
|
|
483
|
-
* power — the pre-registration companion to {@link mcnemar}. Parametrised by the
|
|
484
|
-
* expected discordant-cell probabilities `p10` (P[treatment wins on a pair]) and
|
|
485
|
-
* `p01` (P[control wins]); concordant pairs carry no information, so the count
|
|
486
|
-
* is driven entirely by the discordant rate. Lachin's (1992) asymptotic normal
|
|
487
|
-
* approximation: with discordant rate `pDisc = p10 + p01` and marginal effect
|
|
488
|
-
* `δ = p10 − p01`,
|
|
489
|
-
* n = ( z_{1-α/2}·√pDisc + z_{1-β}·√(pDisc − δ²) )² / δ².
|
|
490
|
-
* Returns Infinity when there is no effect (p10 === p01). Asymptotic — at the
|
|
491
|
-
* tiny discordant counts where the exact {@link mcnemar} differs from the normal
|
|
492
|
-
* approximation, treat the result as a lower bound and prefer the discordant-pair
|
|
493
|
-
* floor.
|
|
494
|
-
*/
|
|
495
|
-
declare function mcnemarRequiredN(opts: {
|
|
496
|
-
p10: number;
|
|
497
|
-
p01: number;
|
|
498
|
-
alpha?: number;
|
|
499
|
-
power?: number;
|
|
500
|
-
twoSided?: boolean;
|
|
501
|
-
}): number;
|
|
502
|
-
/**
|
|
503
|
-
* Power of a McNemar test at a given number of paired observations, the inverse
|
|
504
|
-
* of {@link mcnemarRequiredN} (same Lachin asymptotic model, same parameters).
|
|
505
|
-
* Returns a value in [0, 1]; equals `alpha` when there is no effect.
|
|
506
|
-
*/
|
|
507
|
-
declare function mcnemarPower(opts: {
|
|
508
|
-
p10: number;
|
|
509
|
-
p01: number;
|
|
510
|
-
nPairs: number;
|
|
511
|
-
alpha?: number;
|
|
512
|
-
twoSided?: boolean;
|
|
513
|
-
}): number;
|
|
514
|
-
/**
|
|
515
|
-
* Bonferroni adjustment: multiply every p-value by the test count, clamp at 1.
|
|
516
|
-
*
|
|
517
|
-
* Rejects at `p_adjusted ≤ alpha` — the boundary is inclusive, matching
|
|
518
|
-
* {@link holm}, which uniformly dominates this correction and must therefore
|
|
519
|
-
* never reject less. Validates its inputs on the same terms.
|
|
520
|
-
*/
|
|
521
|
-
declare function bonferroni(pValues: readonly number[], alpha?: number): {
|
|
522
|
-
adjusted: number[];
|
|
523
|
-
significant: boolean[];
|
|
524
|
-
};
|
|
525
|
-
/**
|
|
526
|
-
* Holm step-down family-wise error adjustment.
|
|
527
|
-
*
|
|
528
|
-
* P-values are sorted from smallest to largest, multiplied by their remaining
|
|
529
|
-
* hypothesis count, and made monotonically non-decreasing before being mapped
|
|
530
|
-
* back to input order. This uniformly dominates plain Bonferroni while keeping
|
|
531
|
-
* strong family-wise error control under arbitrary dependence.
|
|
532
|
-
*/
|
|
533
|
-
declare function holm(pValues: readonly number[], alpha?: number): {
|
|
534
|
-
adjusted: number[];
|
|
535
|
-
significant: boolean[];
|
|
536
|
-
};
|
|
537
|
-
/**
|
|
538
|
-
* Benjamini–Hochberg false discovery rate. Returns adjusted q-values and
|
|
539
|
-
* significance at the target FDR; handles ties and preserves q monotonicity.
|
|
540
|
-
*
|
|
541
|
-
* Rejects at `q ≤ fdr` — the BH rule is inclusive at the boundary, so an
|
|
542
|
-
* exactly-`fdr` q-value is a discovery.
|
|
543
|
-
*/
|
|
544
|
-
declare function benjaminiHochberg(pValues: readonly number[], fdr?: number): {
|
|
545
|
-
qValues: number[];
|
|
546
|
-
significant: boolean[];
|
|
547
|
-
};
|
|
548
|
-
interface PairedBootstrapResult {
|
|
549
|
-
/** Number of paired observations. */
|
|
550
|
-
n: number;
|
|
551
|
-
/** Median of paired deltas (after − before). */
|
|
552
|
-
median: number;
|
|
553
|
-
/** Mean of paired deltas. */
|
|
554
|
-
mean: number;
|
|
555
|
-
/** Lower bound of the bootstrap CI on the chosen statistic. */
|
|
556
|
-
low: number;
|
|
557
|
-
/** Upper bound of the bootstrap CI on the chosen statistic. */
|
|
558
|
-
high: number;
|
|
559
|
-
/** Confidence level used (e.g. 0.95). */
|
|
560
|
-
confidence: number;
|
|
561
|
-
/** Number of bootstrap resamples used. */
|
|
562
|
-
resamples: number;
|
|
563
|
-
/** False below {@link BOOTSTRAP_GATE_MIN_N}. See {@link pairedBootstrap}. */
|
|
564
|
-
gateEligible: boolean;
|
|
565
|
-
}
|
|
566
|
-
/**
|
|
567
|
-
* Pairs below which a percentile bootstrap interval is descriptive spread only.
|
|
568
|
-
*
|
|
569
|
-
* `P(low > 0)` under a true null, against a nominal 2.5 %, measured over 4000
|
|
570
|
-
* seeded trials: 13.53 % at n = 3, 3.52 % at n = 10, 3.10 % at n = 20 on the
|
|
571
|
-
* median; 13.85 %, 4.90 %, 3.80 % on the mean. This is intrinsic to resampling
|
|
572
|
-
* three points, not an implementation error — scipy's BCa gives 16.0 % on the
|
|
573
|
-
* same n = 3 data — so no change to the estimator moves it. Below this floor
|
|
574
|
-
* the decision belongs to the exact sign test or exact signed-rank test.
|
|
575
|
-
*/
|
|
576
|
-
declare const BOOTSTRAP_GATE_MIN_N = 20;
|
|
577
|
-
interface PairedBootstrapOptions {
|
|
578
|
-
/** Confidence level. Default 0.95. */
|
|
579
|
-
confidence?: number;
|
|
580
|
-
/** Bootstrap resample count. Default 2000. */
|
|
581
|
-
resamples?: number;
|
|
582
|
-
/** Statistic to bootstrap. Default 'median'. */
|
|
583
|
-
statistic?: 'median' | 'mean';
|
|
584
|
-
/** Deterministic seed. If omitted, derived from the deltas so the interval
|
|
585
|
-
* is reproducible regardless. */
|
|
586
|
-
seed?: number;
|
|
587
|
-
}
|
|
588
|
-
/**
|
|
589
|
-
* Paired bootstrap on (after − before) deltas. Returns a CI on the chosen
|
|
590
|
-
* statistic (median by default); pairs are resampled with replacement. Throws
|
|
591
|
-
* on unequal sample sizes.
|
|
592
|
-
*
|
|
593
|
-
* `low > threshold` carries the stated confidence ONLY at `n ≥
|
|
594
|
-
* {@link BOOTSTRAP_GATE_MIN_N}`, which `gateEligible` reports. Below it the
|
|
595
|
-
* check fires under a true null several times more often than nominal, so the
|
|
596
|
-
* interval is descriptive spread and a promotion must not turn on it.
|
|
597
|
-
*/
|
|
598
|
-
declare function pairedBootstrap(before: number[], after: number[], opts?: PairedBootstrapOptions): PairedBootstrapResult;
|
|
599
|
-
/** Pre-registered direction for a one-sided paired sign test. */
|
|
600
|
-
type SignTestAlternative = 'greater' | 'less';
|
|
601
|
-
/** Exact one-sided sign-test result for paired numeric differences. */
|
|
602
|
-
interface PairedSignTestResult {
|
|
603
|
-
/** Total supplied differences, including zero ties. */
|
|
604
|
-
n: number;
|
|
605
|
-
/** Strictly positive differences. */
|
|
606
|
-
positive: number;
|
|
607
|
-
/** Strictly negative differences. */
|
|
608
|
-
negative: number;
|
|
609
|
-
/** Zero differences excluded from the binomial test. */
|
|
610
|
-
ties: number;
|
|
611
|
-
/** Non-zero differences used by the binomial test. */
|
|
612
|
-
nNonTies: number;
|
|
613
|
-
/** Direction of the pre-registered alternative hypothesis. */
|
|
614
|
-
alternative: SignTestAlternative;
|
|
615
|
-
/** Exact one-sided p-value under P(positive) = P(negative) = 0.5. */
|
|
616
|
-
pValue: number;
|
|
617
|
-
}
|
|
618
|
-
/**
|
|
619
|
-
* Exact one-sided sign test over paired differences.
|
|
620
|
-
*
|
|
621
|
-
* Pass `after[i] - before[i]` for each matched item. `alternative = 'greater'`
|
|
622
|
-
* tests whether positive signs are more likely than negative signs and returns
|
|
623
|
-
* `P(Binomial(nNonTies, 0.5) >= positive)`. `alternative = 'less'` treats
|
|
624
|
-
* negative signs as successes instead. With a continuous difference
|
|
625
|
-
* distribution this is the usual directional median test. Exact zero
|
|
626
|
-
* differences are ties and do not enter the binomial denominator. All-tie and
|
|
627
|
-
* empty inputs return p = 1. Every input difference must be finite, and the
|
|
628
|
-
* direction must be chosen explicitly so a caller cannot select it after
|
|
629
|
-
* seeing the signs.
|
|
630
|
-
*/
|
|
631
|
-
declare function pairedSignTest(differences: readonly number[], alternative: SignTestAlternative): PairedSignTestResult;
|
|
632
|
-
/** A binomial proportion estimate with a confidence interval. */
|
|
633
|
-
interface ProportionInterval {
|
|
634
|
-
/** Point estimate successes / n (0 when n = 0). */
|
|
635
|
-
estimate: number;
|
|
636
|
-
/** Lower bound, clamped to [0, 1]. */
|
|
637
|
-
lower: number;
|
|
638
|
-
/** Upper bound, clamped to [0, 1]. */
|
|
639
|
-
upper: number;
|
|
640
|
-
}
|
|
641
|
-
/**
|
|
642
|
-
* Wilson score interval for a binomial proportion. Correct at small n and near
|
|
643
|
-
* 0/1, where the normal (Wald) approximation produces bounds outside [0, 1] and
|
|
644
|
-
* understates coverage. Use this for any pass-rate / hit-rate / realness-rate
|
|
645
|
-
* CI — the continuous `confidenceInterval` assumes the wrong distribution for a
|
|
646
|
-
* proportion. `n = 0 ⇒ {0, 0, 0}`.
|
|
647
|
-
*/
|
|
648
|
-
declare function wilson(successes: number, n: number, confidence?: number): ProportionInterval;
|
|
649
|
-
/**
|
|
650
|
-
* Are these per-item outcomes binary (every value exactly 0 or 1)?
|
|
651
|
-
*
|
|
652
|
-
* The discriminator a promotion gate needs before choosing a paired statistic.
|
|
653
|
-
* On binary outcomes the paired delta vector lives in {-1, 0, +1} and is
|
|
654
|
-
* normally dominated by zeros (both arms solve, or both arms miss, most items),
|
|
655
|
-
* so its MEDIAN is pinned at exactly 0 no matter how large the real shift in
|
|
656
|
-
* success rate is — and a bootstrap CI on that median collapses to [0, 0].
|
|
657
|
-
* A gate keying on `ci.low > threshold` is then structurally unable to see
|
|
658
|
-
* either a gain or a regression. Detect this shape and switch to the
|
|
659
|
-
* paired-binary estimators ({@link mcnemar}, {@link pairedRiskDifference})
|
|
660
|
-
* instead of silently answering "no" forever.
|
|
661
|
-
*
|
|
662
|
-
* Empty input is NOT binary: there is no evidence of the outcome's shape, and
|
|
663
|
-
* defaulting an empty vector into the binary branch would pick a statistic on
|
|
664
|
-
* no data at all.
|
|
665
|
-
*
|
|
666
|
-
* NOT the right discriminator for a gate. It recognises the literal {0, 1}
|
|
667
|
-
* encoding and nothing else, so a pass/fail dimension emitted on 0-100 — which
|
|
668
|
-
* judges in this codebase do routinely — reads as non-binary, and a single
|
|
669
|
-
* partial-credit score in an otherwise pass/fail vector flips it to false while
|
|
670
|
-
* leaving the median just as blind. Gates want {@link pairedBinaryScale} (any
|
|
671
|
-
* two-point encoding). This predicate remains for callers that specifically
|
|
672
|
-
* mean "literally 0/1".
|
|
673
|
-
*/
|
|
674
|
-
declare function isBinaryOutcomeVector(values: ArrayLike<number>): boolean;
|
|
675
|
-
/** Result of a McNemar paired-binary significance test. */
|
|
676
|
-
interface McNemarResult {
|
|
677
|
-
/** Total paired observations. */
|
|
678
|
-
n: number;
|
|
679
|
-
/** Discordant pairs (b + c) — the only ones that carry signal. */
|
|
680
|
-
nDiscordant: number;
|
|
681
|
-
/** Pairs where treatment succeeded and control failed ("newly correct"). */
|
|
682
|
-
b: number;
|
|
683
|
-
/** Pairs where control succeeded and treatment failed ("newly wrong"). */
|
|
684
|
-
c: number;
|
|
685
|
-
/** Continuity-corrected chi-square statistic (reference; exact p drives the call). */
|
|
686
|
-
statistic: number;
|
|
687
|
-
/** Two-sided p-value. Exact (binomial sign test on discordant pairs). */
|
|
688
|
-
pValue: number;
|
|
689
|
-
}
|
|
690
|
-
/**
|
|
691
|
-
* McNemar's test for paired binary outcomes — the correct significance test for
|
|
692
|
-
* "does treatment change the success rate vs control on the SAME items". Only
|
|
693
|
-
* discordant pairs (one arm right, the other wrong) carry information; concordant
|
|
694
|
-
* pairs are uninformative, so a paired t-test / two-proportion z-test on the raw
|
|
695
|
-
* rates is wrong here. The p-value is exact: under H0 the b "treatment-wins" are
|
|
696
|
-
* Binomial(b + c, 0.5), so the two-sided p is the doubled binomial tail — correct
|
|
697
|
-
* at the small discordant counts typical of eval runs (no continuity-corrected
|
|
698
|
-
* chi-square approximation needed, though it is returned as `statistic` for
|
|
699
|
-
* reference). Inputs are paired 0/1 (or boolean) arrays, control first to match
|
|
700
|
-
* the module's (before, after) convention. Throws on unequal lengths.
|
|
701
|
-
*/
|
|
702
|
-
declare function mcnemar(control: ArrayLike<number | boolean>, treatment: ArrayLike<number | boolean>): McNemarResult;
|
|
703
|
-
/** A paired binary effect size (treatment rate − control rate) with a CI. */
|
|
704
|
-
interface RiskDifferenceResult {
|
|
705
|
-
/** Total paired observations. */
|
|
706
|
-
n: number;
|
|
707
|
-
/** Discordant pairs: treatment-win count. */
|
|
708
|
-
b: number;
|
|
709
|
-
/** Discordant pairs: control-win count. */
|
|
710
|
-
c: number;
|
|
711
|
-
/** Paired risk difference p(treatment) − p(control) = (b − c) / n. */
|
|
712
|
-
riskDifference: number;
|
|
713
|
-
/** Lower bound of the CI, clamped to [-1, 1]. */
|
|
714
|
-
lower: number;
|
|
715
|
-
/** Upper bound of the CI, clamped to [-1, 1]. */
|
|
716
|
-
upper: number;
|
|
717
|
-
/** Confidence level used. */
|
|
718
|
-
confidence: number;
|
|
719
|
-
}
|
|
720
|
-
/**
|
|
721
|
-
* Paired risk difference (the effect-size companion to {@link mcnemar}): the
|
|
722
|
-
* change in success rate p(treatment) − p(control) on matched items, which for
|
|
723
|
-
* paired binary data equals (b − c) / n. The CI uses the paired variance from
|
|
724
|
-
* the discordant counts, not the independent-samples formula (which overstates
|
|
725
|
-
* the interval by ignoring the pairing). Inputs are paired 0/1 (or boolean)
|
|
726
|
-
* arrays, control first. Throws on unequal lengths.
|
|
727
|
-
*
|
|
728
|
-
* REPORTING ONLY — do NOT decide a promotion on this interval. The CI is a Wald
|
|
729
|
-
* normal approximation, which badly UNDERCOVERS when only a handful of pairs are
|
|
730
|
-
* discordant: at n = 3 with b = 2, c = 0 it returns [0.133, 1.000], excluding 0,
|
|
731
|
-
* while McNemar's exact test on the same data gives p = 0.50. A gate keying on
|
|
732
|
-
* `lower > 0` would promote noise. Use {@link pairedRiskDifferenceExact}, whose
|
|
733
|
-
* interval is dual to the exact test by construction, for any decision.
|
|
734
|
-
*/
|
|
735
|
-
declare function pairedRiskDifference(control: ArrayLike<number | boolean>, treatment: ArrayLike<number | boolean>, confidence?: number): RiskDifferenceResult;
|
|
736
|
-
/** A paired binary effect size with an EXACT interval and the exact test that
|
|
737
|
-
* bounds it — one object so a caller cannot read the estimate without the
|
|
738
|
-
* significance it is entitled to. */
|
|
739
|
-
interface ExactRiskDifferenceResult {
|
|
740
|
-
/** Total paired observations. */
|
|
741
|
-
n: number;
|
|
742
|
-
/** Discordant pairs: treatment-win count. */
|
|
743
|
-
b: number;
|
|
744
|
-
/** Discordant pairs: control-win count. */
|
|
745
|
-
c: number;
|
|
746
|
-
/** Discordant pairs (b + c) — the only ones carrying information. */
|
|
747
|
-
nDiscordant: number;
|
|
748
|
-
/** Paired risk difference p(treatment) − p(control) = (b − c) / n. */
|
|
749
|
-
riskDifference: number;
|
|
750
|
-
/** Exact conditional CI lower bound. 0 when there are no discordant pairs. */
|
|
751
|
-
lower: number;
|
|
752
|
-
/** Exact conditional CI upper bound. 0 when there are no discordant pairs. */
|
|
753
|
-
upper: number;
|
|
754
|
-
/** Confidence level used. */
|
|
755
|
-
confidence: number;
|
|
756
|
-
/** McNemar's exact two-sided p-value on the same discordant counts. */
|
|
757
|
-
pValue: number;
|
|
758
|
-
}
|
|
759
|
-
/**
|
|
760
|
-
* Paired risk difference with the EXACT CONDITIONAL interval — the estimator a
|
|
761
|
-
* promotion gate may decide on.
|
|
762
|
-
*
|
|
763
|
-
* Conditional on the number of discordant pairs m = b + c, the treatment-win
|
|
764
|
-
* count b is Binomial(m, π) with π = P(treatment wins | discordant), and the
|
|
765
|
-
* risk difference is an exact reparameterisation: RD = (2π − 1)·m/n. So a
|
|
766
|
-
* Clopper-Pearson exact interval for π maps straight onto RD. This buys the
|
|
767
|
-
* property the Wald interval in {@link pairedRiskDifference} does not have:
|
|
768
|
-
*
|
|
769
|
-
* **`lower > 0` ⟺ McNemar's exact test rejects at α = 1 − confidence.**
|
|
770
|
-
*
|
|
771
|
-
* Clopper-Pearson excludes π = 0.5 exactly when the two-sided exact binomial
|
|
772
|
-
* test of π = 0.5 rejects, and that test IS {@link mcnemar}'s p-value — so the
|
|
773
|
-
* interval and the test can never disagree, and a gate keyed on `lower` cannot
|
|
774
|
-
* promote what the exact test refuses. The exact p is returned in the same
|
|
775
|
-
* object so the two are impossible to compute apart.
|
|
776
|
-
*
|
|
777
|
-
* The interval is conservative (exact intervals over-cover; conditioning on m
|
|
778
|
-
* discards the concordant pairs' information about m itself). That is the
|
|
779
|
-
* correct direction for a promotion gate: it refuses more often, never less.
|
|
780
|
-
*
|
|
781
|
-
* With m = 0 there are no discordant pairs and π is not identified: the result
|
|
782
|
-
* is the degenerate [0, 0] with p = 1. That is NOT evidence of equivalence —
|
|
783
|
-
* callers must treat a zero-width interval as "cannot decide", not as "no
|
|
784
|
-
* difference". Inputs are paired 0/1 (or boolean) arrays, control first.
|
|
785
|
-
* Throws on unequal lengths.
|
|
786
|
-
*/
|
|
787
|
-
declare function pairedRiskDifferenceExact(control: ArrayLike<number | boolean>, treatment: ArrayLike<number | boolean>, confidence?: number): ExactRiskDifferenceResult;
|
|
788
|
-
/** A paired binary effect size with an interval that is valid at a NONZERO
|
|
789
|
-
* margin — the estimator a noninferiority decision may be made on. */
|
|
790
|
-
interface ScoreRiskDifferenceResult {
|
|
791
|
-
/** Total paired observations. */
|
|
792
|
-
n: number;
|
|
793
|
-
/** Discordant pairs: treatment-win count. */
|
|
794
|
-
b: number;
|
|
795
|
-
/** Discordant pairs: control-win count. */
|
|
796
|
-
c: number;
|
|
797
|
-
/** Discordant pairs (b + c). */
|
|
798
|
-
nDiscordant: number;
|
|
799
|
-
/** Paired risk difference p(treatment) − p(control) = (b − c) / n. */
|
|
800
|
-
riskDifference: number;
|
|
801
|
-
/** Score-interval lower bound on the population risk difference. */
|
|
802
|
-
lower: number;
|
|
803
|
-
/** Score-interval upper bound on the population risk difference. */
|
|
804
|
-
upper: number;
|
|
805
|
-
/** Confidence level used. */
|
|
806
|
-
confidence: number;
|
|
807
|
-
}
|
|
808
|
-
/**
|
|
809
|
-
* Paired risk difference with TANGO'S (1998) SCORE INTERVAL — the estimator a
|
|
810
|
-
* promotion gate may decide on **at a nonzero margin**.
|
|
811
|
-
*
|
|
812
|
-
* {@link pairedRiskDifferenceExact} conditions on the observed discordant count
|
|
813
|
-
* `m = b + c`, builds a Clopper-Pearson interval for the win share among those
|
|
814
|
-
* `m` pairs, and multiplies by the observed `m/n`. That is exact for testing
|
|
815
|
-
* RD = 0 — it is dual to McNemar — but it is NOT a confidence interval for the
|
|
816
|
-
* population risk difference at a nonzero margin, because the sampling
|
|
817
|
-
* variability of `m/n` itself is discarded. The gap is not academic: with the
|
|
818
|
-
* production caller's `pairedDeltaThreshold: -0.05`, a process whose true risk
|
|
819
|
-
* difference sits exactly on that margin clears a nominal-95 % `lower > margin`
|
|
820
|
-
* check 24.75 % of the time at n = 40 and 43.95 % at n = 76 (2000 replicates
|
|
821
|
-
* each) when the conditional interval decides.
|
|
822
|
-
*
|
|
823
|
-
* Tango's interval inverts the score test of RD = delta, which estimates the
|
|
824
|
-
* nuisance loss rate under each hypothesised delta instead of fixing it at the
|
|
825
|
-
* observed value, so `m` contributes its own uncertainty. It is the method
|
|
826
|
-
* `ratesci::scorepairci` uses for paired risk-difference noninferiority, and it
|
|
827
|
-
* is not conditional, so it stays valid as the margin moves away from zero.
|
|
828
|
-
*
|
|
829
|
-
* The bounds are found by bisecting `tangoScore(delta) = ±z` — the score is
|
|
830
|
-
* monotone decreasing in delta, so each crossing is unique. Inputs are paired
|
|
831
|
-
* 0/1 (or boolean) arrays, control first. Throws on unequal lengths.
|
|
832
|
-
*/
|
|
833
|
-
declare function pairedRiskDifferenceScore(control: ArrayLike<number | boolean>, treatment: ArrayLike<number | boolean>, confidence?: number): ScoreRiskDifferenceResult;
|
|
834
|
-
/**
|
|
835
|
-
* The common positive level `s` such that EVERY value across both paired arms is
|
|
836
|
-
* exactly 0 or `s` — i.e. the outcome is pass/fail, whatever encoding it arrived
|
|
837
|
-
* in. Returns null when the outcomes are not two-point, when the two arms use
|
|
838
|
-
* different levels, or when no positive value was observed at all (all-zero
|
|
839
|
-
* arms: the level is not identified, and there is nothing to decide anyway).
|
|
840
|
-
*
|
|
841
|
-
* This is the scale-aware successor to {@link isBinaryOutcomeVector}, which only
|
|
842
|
-
* recognises literal {0, 1}. Judges in this codebase emit dimensions on 0-100 as
|
|
843
|
-
* well as [0,1] (see `detectScale` in `campaign/gates/statistical-heldout.ts`),
|
|
844
|
-
* so a pass/fail dimension routinely arrives as {0, 100} and a {0,1}-only test
|
|
845
|
-
* silently sends it down the median path that cannot see it. Any positive level
|
|
846
|
-
* is accepted, not just 1 and 100: for a two-point {0, s} outcome the mean paired
|
|
847
|
-
* delta is exactly s·(b − c)/n, so the binary estimators apply after dividing by
|
|
848
|
-
* s and rescaling the result back into the caller's native units.
|
|
849
|
-
*
|
|
850
|
-
* Non-finite values ⇒ null: an unusable outcome must not be classified as a
|
|
851
|
-
* clean pass/fail shape.
|
|
852
|
-
*/
|
|
853
|
-
declare function pairedBinaryScale(before: ArrayLike<number>, after: ArrayLike<number>): number | null;
|
|
854
|
-
/** Fraction of paired observations whose delta is an exact tie (|after − before|
|
|
855
|
-
* < 1e-9). Throws on unequal sample sizes; 0 pairs ⇒ 0. */
|
|
856
|
-
declare function pairedDeltaTieFraction(before: ArrayLike<number>, after: ArrayLike<number>): number;
|
|
857
|
-
/**
|
|
858
|
-
* The paired-delta statistic a DECISION is computed on, package-wide.
|
|
859
|
-
*
|
|
860
|
-
* The mean paired delta is the estimator that answers the question a promotion
|
|
861
|
-
* gate asks — "by how much did the candidate move the score" — in the caller's
|
|
862
|
-
* own units, and it equals the aggregate lift everyone quotes. The MEDIAN
|
|
863
|
-
* answers a different question and loses the answer to this one in every regime
|
|
864
|
-
* eval data actually lands in:
|
|
865
|
-
* - TWO-POINT (pass/fail) outcomes on any encoding: the delta vector lives in
|
|
866
|
-
* {−s, 0, +s} dominated by zeros, so the median and its whole bootstrap CI
|
|
867
|
-
* are pinned at exactly 0 however large the shift. (Decide these on
|
|
868
|
-
* {@link pairedRiskDifferenceExact} instead — same estimand, exact interval.)
|
|
869
|
-
* - TIE-DOMINATED outcomes: at half the pairs tied the sample median is 0 by
|
|
870
|
-
* construction, and `ci.low > threshold` then answers "no" forever at a
|
|
871
|
-
* non-negative threshold and "yes" forever at a negative one.
|
|
872
|
-
* - LOW-CARDINALITY outcomes, even well below half ties: judge dimensions on
|
|
873
|
-
* integer 0-100, and block scores like {⅔, 1} from averaging pass/fail
|
|
874
|
-
* leaves, put the median on a coarse lattice whose bootstrap percentiles
|
|
875
|
-
* land on atoms. Measured: 26 blocks of 3 pass/fail leaves carrying a real
|
|
876
|
-
* +12.8pp lift, only 23% of pairs tied, gives a median CI of [0, 0.333] —
|
|
877
|
-
* lower bound exactly 0, so a gate at threshold 0 refuses a real lift.
|
|
878
|
-
* That last case is why there is no tie-fraction threshold here: any cutoff on
|
|
879
|
-
* ties leaves the lattice case open on the other side of it.
|
|
880
|
-
*
|
|
881
|
-
* `heldoutSignificance` has defaulted to the mean since #316 for the same
|
|
882
|
-
* reason. The median remains available per call site for callers who
|
|
883
|
-
* specifically want outlier robustness and accept the blindness.
|
|
884
|
-
*/
|
|
885
|
-
declare const DECISION_PAIRED_DELTA_STATISTIC: 'mean';
|
|
886
|
-
/**
|
|
887
|
-
* Unbiased pass@k for code generation (Chen et al. 2021, "Evaluating Large
|
|
888
|
-
* Language Models Trained on Code"). Given `n` independent samples for one
|
|
889
|
-
* problem of which `c` pass, the probability that at least one of a random k of
|
|
890
|
-
* them passes is 1 − C(n−c, k) / C(n, k). Estimating pass@k as "did any of the
|
|
891
|
-
* first k pass" is biased high at small n; this is the variance-reduced estimator
|
|
892
|
-
* averaged implicitly over all k-subsets. Average the per-problem values across
|
|
893
|
-
* the suite for the corpus pass@k. Computed in the numerically stable product
|
|
894
|
-
* form. Requires 1 ≤ k ≤ n and 0 ≤ c ≤ n.
|
|
895
|
-
*/
|
|
896
|
-
declare function passAtK(n: number, c: number, k: number): number;
|
|
897
|
-
interface EProcessOptions {
|
|
898
|
-
/** Type-I error budget. The process decides when wealth ≥ 1/alpha
|
|
899
|
-
* (Ville's inequality). Default 0.05. */
|
|
900
|
-
alpha?: number;
|
|
901
|
-
/** Truncation bound on the predictable bet λ ∈ [0, maxBet]. Must satisfy
|
|
902
|
-
* maxBet < 1/nullMean so every wealth factor stays strictly positive.
|
|
903
|
-
* Default 0.5. */
|
|
904
|
-
maxBet?: number;
|
|
905
|
-
/** The null boundary m₀ for H0: E[x] ≤ m₀ on x ∈ [0,1]. Default 0.5
|
|
906
|
-
* (the paired-delta encoding x = (d+1)/2 maps "no effect" to 1/2).
|
|
907
|
-
* A pre-registered minEffect shifts this — see `sequentialPairedGate`. */
|
|
908
|
-
nullMean?: number;
|
|
909
|
-
}
|
|
910
|
-
interface EProcessStep {
|
|
911
|
-
/** Current wealth W_n — the e-value against H0 after n observations. */
|
|
912
|
-
wealth: number;
|
|
913
|
-
/** Observations consumed so far. */
|
|
914
|
-
n: number;
|
|
915
|
-
/** True from the first n where W_n ≥ 1/alpha onward (sticky). */
|
|
916
|
-
decided: boolean;
|
|
917
|
-
}
|
|
918
|
-
interface EProcessState extends EProcessStep {
|
|
919
|
-
alpha: number;
|
|
920
|
-
maxBet: number;
|
|
921
|
-
nullMean: number;
|
|
922
|
-
/** The decision boundary 1/alpha. */
|
|
923
|
-
threshold: number;
|
|
924
|
-
/** Observation count at the first threshold crossing; undefined until decided. */
|
|
925
|
-
decidedAtN?: number;
|
|
926
|
-
}
|
|
927
|
-
interface EProcess {
|
|
928
|
-
/** Consume one observation x ∈ [0,1]. Throws on non-finite / out-of-range
|
|
929
|
-
* input — a silent clamp would corrupt the type-I guarantee. */
|
|
930
|
-
update(x: number): EProcessStep;
|
|
931
|
-
state(): EProcessState;
|
|
932
|
-
}
|
|
933
|
-
/**
|
|
934
|
-
* Betting test-martingale for bounded observations — the e-process core of
|
|
935
|
-
* anytime-valid sequential testing (Waudby-Smith & Ramdas, "Estimating means
|
|
936
|
-
* of bounded random variables by betting", JRSS-B 2024).
|
|
937
|
-
*
|
|
938
|
-
* Observations x_i ∈ [0,1]; H0: E[x] ≤ m₀ (`nullMean`, default 1/2). Wealth
|
|
939
|
-
*
|
|
940
|
-
* W_t = Π_{i≤t} (1 + λ_i (x_i − m₀)), W_0 = 1
|
|
941
|
-
*
|
|
942
|
-
* with the truncated GROW-style plug-in bet computed from PRIOR observations:
|
|
943
|
-
*
|
|
944
|
-
* λ_i = clamp((μ̂_{i−1} − m₀) / (σ̂²_{i−1} + (μ̂_{i−1} − m₀)²), 0, maxBet)
|
|
945
|
-
*
|
|
946
|
-
* where μ̂/σ̂² are the shrunk running estimates μ̂_t = (1/2 + Σx_i)/(t+1),
|
|
947
|
-
* σ̂²_t = (1/4 + Σ(x_i − μ̂_i)²)/(t+1).
|
|
948
|
-
*
|
|
949
|
-
* PREDICTABILITY INVARIANT (load-bearing): λ_i is a function of x_1..x_{i−1}
|
|
950
|
-
* ONLY — it may never see x_i. With λ_i ≥ 0 predictable, each factor has
|
|
951
|
-
* E[1 + λ_i(x_i − m₀) | past] ≤ 1 under H0, so W is a nonnegative
|
|
952
|
-
* supermartingale and Ville's inequality gives P(∃t: W_t ≥ 1/α) ≤ α — the
|
|
953
|
-
* type-I guarantee holds at ANY data-dependent stopping time. λ_1 is always 0
|
|
954
|
-
* (no prior evidence), so the first observation never moves wealth.
|
|
955
|
-
*
|
|
956
|
-
* `decided` latches at the first crossing W_t ≥ 1/α and never un-latches;
|
|
957
|
-
* wealth keeps updating after the crossing (the e-process remains valid), but
|
|
958
|
-
* the decision time is the first crossing.
|
|
959
|
-
*/
|
|
960
|
-
declare function eProcess(opts?: EProcessOptions): EProcess;
|
|
961
|
-
/** Tiny seedable PRNG (mulberry32) — deterministic resampling/shuffling, not
|
|
962
|
-
* cryptographic. Exported so e-process shuffles and bootstrap resampling
|
|
963
|
-
* share ONE PRNG implementation. Every distinct 32-bit seed gives a distinct
|
|
964
|
-
* stream, including 0. */
|
|
965
|
-
declare function mulberry32(seed: number): () => number;
|
|
966
|
-
//#endregion
|
|
967
|
-
export { pairedCohensDz as $, WeightedCompositeInput as A, positionalBias as At, eProcess as B, RankTestMethod as C, GoldenItem as Ct, ScoreRiskDifferenceResult as D, calibrateJudge as Dt, RiskDifferenceResult as E, VerbosityBiasResult as Et, cliffsDelta as F, mannWhitneyU as G, interRaterReliability as H, cohensD as I, mcnemarRequiredN as J, mcnemar as K, confidenceInterval as L, WilcoxonSignedRankResult as M, verbosityBias as Mt, benjaminiHochberg as N, SignTestAlternative as O, calibrateJudgeContinuous as Ot, bonferroni as P, pairedBootstrap as Q, corpusInterRaterAgreement as R, ProportionInterval as S, ContinuousCalibrationResult as St, RankTestOptions as T, SelfPreferenceResult as Tt, interpretCliffs as U, holm as V, isBinaryOutcomeVector as W, normalizeScores as X, mulberry32 as Y, pairedBinaryScale as Z, McNemarResult as _, wilson as _t, CorpusAgreementReport as a, pairedSignTest as at, PairedSignTestResult as b, ContinuousAgreement as bt, DEFAULT_PERMUTATIONS as c, passAtK as ct, EProcessState as d, requiredPairedSampleSize as dt, pairedDeltaTieFraction as et, EProcessStep as f, requiredSampleSize as ft, MannWhitneyResult as g, wilcoxonSignedRank as gt, MANN_WHITNEY_EXACT_MAX_WORK as h, weightedMean as ht, CorpusAgreementPerDimension as i, pairedRiskDifferenceScore as it, WeightedCompositeResult as j, selfPreference as jt, WILCOXON_EXACT_MAX_N as k, continuousAgreement as kt, EProcess as l, pearsonR as lt, MANN_WHITNEY_EXACT_MAX_STATES as m, weightedComposite as mt, CliffsMagnitude as n, pairedRiskDifference as nt, CorpusScoreRecord as o, pairedTTest as ot, ExactRiskDifferenceResult as p, spearmanR as pt, mcnemarPower as q, CorpusAgreementOptions as r, pairedRiskDifferenceExact as rt, DECISION_PAIRED_DELTA_STATISTIC as s, partialCredit as st, BOOTSTRAP_GATE_MIN_N as t, pairedMde as tt, EProcessOptions as u, ranks as ut, PairedBootstrapOptions as v, CalibrationResult as vt, RankTestMethodRequest as w, PositionalBiasResult as wt, PairedTTestResult as x, ContinuousAgreementOptions as xt, PairedBootstrapResult as y, CandidateScore as yt, corpusInterRaterAgreementFromJudgeScores as z };
|
|
968
|
-
//# sourceMappingURL=statistics-D6Uebe_4.d.ts.map
|