@tangle-network/agent-eval 0.128.2 → 0.130.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +279 -0
- package/README.md +19 -0
- package/dist/active-curriculum-C4mk67HP.js +214 -0
- package/dist/active-curriculum-C4mk67HP.js.map +1 -0
- package/dist/adversarial-smnADNFS.d.ts +21 -0
- package/dist/adversarial-smnADNFS.d.ts.map +1 -0
- package/dist/analyst/index.d.ts +83 -2932
- package/dist/analyst/index.d.ts.map +1 -0
- package/dist/analyst/index.js +319 -364
- package/dist/analyst/index.js.map +1 -1
- package/dist/analyst-BkTS3C58.d.ts +89 -0
- package/dist/analyst-BkTS3C58.d.ts.map +1 -0
- package/dist/analyst-LsnNpSkm.js +152 -0
- package/dist/analyst-LsnNpSkm.js.map +1 -0
- package/dist/analyze-runs-C1CavBMk.js +1067 -0
- package/dist/analyze-runs-C1CavBMk.js.map +1 -0
- package/dist/analyze-runs-FsgYCinh.d.ts +72 -0
- package/dist/analyze-runs-FsgYCinh.d.ts.map +1 -0
- package/dist/attribute-vocabulary-DLJ6303h.d.ts +54 -0
- package/dist/attribute-vocabulary-DLJ6303h.d.ts.map +1 -0
- package/dist/authenticity/index.d.ts +81 -79
- package/dist/authenticity/index.d.ts.map +1 -0
- package/dist/authenticity/index.js +209 -193
- package/dist/authenticity/index.js.map +1 -1
- package/dist/baseline-HsBvw_dk.js +550 -0
- package/dist/baseline-HsBvw_dk.js.map +1 -0
- package/dist/baseline-hG3K85h4.d.ts +125 -0
- package/dist/baseline-hG3K85h4.d.ts.map +1 -0
- package/dist/belief-state/index.d.ts +448 -1205
- package/dist/belief-state/index.d.ts.map +1 -0
- package/dist/belief-state/index.js +1617 -1710
- package/dist/belief-state/index.js.map +1 -1
- package/dist/benchmarks/index.d.ts +2 -894
- package/dist/benchmarks/index.js +2 -59
- package/dist/benchmarks-DviOvUNr.js +754 -0
- package/dist/benchmarks-DviOvUNr.js.map +1 -0
- package/dist/builder-eval/index.d.ts +150 -662
- package/dist/builder-eval/index.d.ts.map +1 -0
- package/dist/builder-eval/index.js +356 -345
- package/dist/builder-eval/index.js.map +1 -1
- package/dist/calibration-CNWWA6K8.js +94 -0
- package/dist/calibration-CNWWA6K8.js.map +1 -0
- package/dist/campaign/index.d.ts +5 -6390
- package/dist/campaign/index.js +3 -212
- package/dist/campaign-CBKZvQ1H.js +3885 -0
- package/dist/campaign-CBKZvQ1H.js.map +1 -0
- package/dist/cli.d.ts +1 -1
- package/dist/cli.js +164 -174
- package/dist/cli.js.map +1 -1
- package/dist/client-C97NMzqi.d.ts +581 -0
- package/dist/client-C97NMzqi.d.ts.map +1 -0
- package/dist/client-CYzbdJOZ.js +637 -0
- package/dist/client-CYzbdJOZ.js.map +1 -0
- package/dist/code-agent-session-BjkMTQ7H.js +1390 -0
- package/dist/code-agent-session-BjkMTQ7H.js.map +1 -0
- package/dist/code-agent-session-DqqgOJaz.d.ts +143 -0
- package/dist/code-agent-session-DqqgOJaz.d.ts.map +1 -0
- package/dist/concurrency-DIxRZF_J.js +85 -0
- package/dist/concurrency-DIxRZF_J.js.map +1 -0
- package/dist/contract/index.d.ts +645 -5605
- package/dist/contract/index.d.ts.map +1 -0
- package/dist/contract/index.js +1730 -1937
- package/dist/contract/index.js.map +1 -1
- package/dist/control.d.ts +3 -1030
- package/dist/control.js +2 -32
- package/dist/cost-ledger-DIgQUFZZ.js +801 -0
- package/dist/cost-ledger-DIgQUFZZ.js.map +1 -0
- package/dist/cost-ledger-Dye6jCgg.d.ts +236 -0
- package/dist/cost-ledger-Dye6jCgg.d.ts.map +1 -0
- package/dist/dataset-BvtnC8Dc.d.ts +115 -0
- package/dist/dataset-BvtnC8Dc.d.ts.map +1 -0
- package/dist/default-registry-C-vFCSEc.js +2579 -0
- package/dist/default-registry-C-vFCSEc.js.map +1 -0
- package/dist/default-registry-CNPo-Vsb.d.ts +540 -0
- package/dist/default-registry-CNPo-Vsb.d.ts.map +1 -0
- package/dist/emitter-CPBAhxum.js +266 -0
- package/dist/emitter-CPBAhxum.js.map +1 -0
- package/dist/emitter-DGQGoLyj.d.ts +113 -0
- package/dist/emitter-DGQGoLyj.d.ts.map +1 -0
- package/dist/errors-8YnH8WlF.js +64 -0
- package/dist/errors-8YnH8WlF.js.map +1 -0
- package/dist/errors-CEk209JS.d.ts +76 -0
- package/dist/errors-CEk209JS.d.ts.map +1 -0
- package/dist/eval-campaign-DEm6c8ru.js +349 -0
- package/dist/eval-campaign-DEm6c8ru.js.map +1 -0
- package/dist/execution-tracks-CpgFPpS5.js +93 -0
- package/dist/execution-tracks-CpgFPpS5.js.map +1 -0
- package/dist/exporters-q9iL-2Jf.js +148 -0
- package/dist/exporters-q9iL-2Jf.js.map +1 -0
- package/dist/extract-usage-BrQ8mCLX.js +155 -0
- package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
- package/dist/failure-cluster-CqcvCcdR.d.ts +59 -0
- package/dist/failure-cluster-CqcvCcdR.d.ts.map +1 -0
- package/dist/feedback-trajectory-CVaeREXV.d.ts +340 -0
- package/dist/feedback-trajectory-CVaeREXV.d.ts.map +1 -0
- package/dist/fuzz.d.ts +320 -646
- package/dist/fuzz.d.ts.map +1 -0
- package/dist/fuzz.js +670 -617
- package/dist/fuzz.js.map +1 -1
- package/dist/hf-dataset-DBJXXoY1.js +763 -0
- package/dist/hf-dataset-DBJXXoY1.js.map +1 -0
- package/dist/hosted/index.d.ts +11 -831
- package/dist/hosted/index.d.ts.map +1 -0
- package/dist/hosted/index.js +2 -37
- package/dist/index-2JJSA6-r2.d.ts +926 -0
- package/dist/index-2JJSA6-r2.d.ts.map +1 -0
- package/dist/index-6N0aYmpW.d.ts +217 -0
- package/dist/index-6N0aYmpW.d.ts.map +1 -0
- package/dist/index-BAvgST_9.d.ts +131 -0
- package/dist/index-BAvgST_9.d.ts.map +1 -0
- package/dist/index-BvnJuTGD.d.ts +68 -0
- package/dist/index-BvnJuTGD.d.ts.map +1 -0
- package/dist/index-C61Wi7yg.d.ts +547 -0
- package/dist/index-C61Wi7yg.d.ts.map +1 -0
- package/dist/index-CAPUUKaM.d.ts +335 -0
- package/dist/index-CAPUUKaM.d.ts.map +1 -0
- package/dist/index-DE5fb3EC.d.ts +2244 -0
- package/dist/index-DE5fb3EC.d.ts.map +1 -0
- package/dist/index-DSC51roc.d.ts +102 -0
- package/dist/index-DSC51roc.d.ts.map +1 -0
- package/dist/index.d.ts +3776 -15120
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +11185 -11191
- package/dist/index.js.map +1 -1
- package/dist/integrity-BzRbCHzi.js +424 -0
- package/dist/integrity-BzRbCHzi.js.map +1 -0
- package/dist/integrity-rmVhXWA7.d.ts +61 -0
- package/dist/integrity-rmVhXWA7.d.ts.map +1 -0
- package/dist/judge-calibration-DFtEMlde.d.ts +146 -0
- package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
- package/dist/ledger-core/index.d.ts +2 -0
- package/dist/ledger-core/index.js +2 -0
- package/dist/ledger-core-DtZz1RG0.js +388 -0
- package/dist/ledger-core-DtZz1RG0.js.map +1 -0
- package/dist/llm-client--GR4JbZE.js +687 -0
- package/dist/llm-client--GR4JbZE.js.map +1 -0
- package/dist/llm-client-B_nIBlYo.d.ts +290 -0
- package/dist/llm-client-B_nIBlYo.d.ts.map +1 -0
- package/dist/matrix/index.d.ts +3 -155
- package/dist/matrix/index.js +2 -8
- package/dist/matrix-BzQnu2S6.js +270 -0
- package/dist/matrix-BzQnu2S6.js.map +1 -0
- package/dist/meta-eval/index.d.ts +4 -1027
- package/dist/meta-eval/index.js +393 -390
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/metrics-C9YY1OcL.js +239 -0
- package/dist/metrics-C9YY1OcL.js.map +1 -0
- package/dist/mint-yN2M2eh0.js +201 -0
- package/dist/mint-yN2M2eh0.js.map +1 -0
- package/dist/multi-layer-verifier-BHY1gWAc.d.ts +138 -0
- package/dist/multi-layer-verifier-BHY1gWAc.d.ts.map +1 -0
- package/dist/multishot/index.d.ts +273 -481
- package/dist/multishot/index.d.ts.map +1 -0
- package/dist/multishot/index.js +595 -548
- package/dist/multishot/index.js.map +1 -1
- package/dist/off-policy-DvgzvtIx.js +220 -0
- package/dist/off-policy-DvgzvtIx.js.map +1 -0
- package/dist/off-policy-mskQw8Mb.d.ts +153 -0
- package/dist/off-policy-mskQw8Mb.d.ts.map +1 -0
- package/dist/openapi.json +1 -1
- package/dist/opencode-sqlite-BGrHeDu3.js +318 -0
- package/dist/opencode-sqlite-BGrHeDu3.js.map +1 -0
- package/dist/outcome-store-BYHIuO0e.d.ts +65 -0
- package/dist/outcome-store-BYHIuO0e.d.ts.map +1 -0
- package/dist/outcome-store-ChBKlTd_.js +75 -0
- package/dist/outcome-store-ChBKlTd_.js.map +1 -0
- package/dist/paired-arms-D9D0wXj2.js +260 -0
- package/dist/paired-arms-D9D0wXj2.js.map +1 -0
- package/dist/pipelines/index.d.ts +95 -532
- package/dist/pipelines/index.d.ts.map +1 -0
- package/dist/pipelines/index.js +497 -478
- package/dist/pipelines/index.js.map +1 -1
- package/dist/pre-registration-DakwTRXk.js +96 -0
- package/dist/pre-registration-DakwTRXk.js.map +1 -0
- package/dist/propose-review-control-LhIWGzHl.js +1458 -0
- package/dist/propose-review-control-LhIWGzHl.js.map +1 -0
- package/dist/query-CJ_DX8vl.d.ts +28 -0
- package/dist/query-CJ_DX8vl.d.ts.map +1 -0
- package/dist/query-Di7eEQ79.js +83 -0
- package/dist/query-Di7eEQ79.js.map +1 -0
- package/dist/raw-provider-sink-BQd7mzyT.js +201 -0
- package/dist/raw-provider-sink-BQd7mzyT.js.map +1 -0
- package/dist/raw-provider-sink-BU29Sh8h.d.ts +134 -0
- package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +1 -0
- package/dist/redact-7Aq1ukl-.js +107 -0
- package/dist/redact-7Aq1ukl-.js.map +1 -0
- package/dist/release-report-Cz9NKH39.js +603 -0
- package/dist/release-report-Cz9NKH39.js.map +1 -0
- package/dist/release-report-mvB2G4_J.d.ts +244 -0
- package/dist/release-report-mvB2G4_J.d.ts.map +1 -0
- package/dist/replay-GnyotH0J.js +1741 -0
- package/dist/replay-GnyotH0J.js.map +1 -0
- package/dist/replay-OoidtG1E.d.ts +749 -0
- package/dist/replay-OoidtG1E.d.ts.map +1 -0
- package/dist/reporting.d.ts +6 -1298
- package/dist/reporting.js +6 -50
- package/dist/researcher-CwTdwXG1.d.ts +314 -0
- package/dist/researcher-CwTdwXG1.d.ts.map +1 -0
- package/dist/reward-hacking-eAnOsynk.d.ts +224 -0
- package/dist/reward-hacking-eAnOsynk.d.ts.map +1 -0
- package/dist/reward-hacking-qipEpKvY.js +596 -0
- package/dist/reward-hacking-qipEpKvY.js.map +1 -0
- package/dist/reward-nw2xZGZG.js +137 -0
- package/dist/reward-nw2xZGZG.js.map +1 -0
- package/dist/rl.d.ts +916 -3596
- package/dist/rl.d.ts.map +1 -0
- package/dist/rl.js +2362 -1751
- package/dist/rl.js.map +1 -1
- package/dist/rolldown-runtime-8H4AJuhK.js +14 -0
- package/dist/rollout/index.d.ts +3 -1048
- package/dist/rollout/index.js +8 -110
- package/dist/rollout-BOYjemfR.js +624 -0
- package/dist/rollout-BOYjemfR.js.map +1 -0
- package/dist/rubric-predictive-validity-B3xmbmS1.js +141 -0
- package/dist/rubric-predictive-validity-B3xmbmS1.js.map +1 -0
- package/dist/rubric-predictive-validity-Ku_clp_1.d.ts +74 -0
- package/dist/rubric-predictive-validity-Ku_clp_1.d.ts.map +1 -0
- package/dist/run-evidence-ClRX_8A9.d.ts +225 -0
- package/dist/run-evidence-ClRX_8A9.d.ts.map +1 -0
- package/dist/run-record-BuoE80Dq.js +467 -0
- package/dist/run-record-BuoE80Dq.js.map +1 -0
- package/dist/run-record-CnZu_gjl.d.ts +357 -0
- package/dist/run-record-CnZu_gjl.d.ts.map +1 -0
- package/dist/run-score-iEEAWiBY.js +41 -0
- package/dist/run-score-iEEAWiBY.js.map +1 -0
- package/dist/runtime-trajectory-1gyaTOoC.js +93 -0
- package/dist/runtime-trajectory-1gyaTOoC.js.map +1 -0
- package/dist/runtime-trajectory-BvSZcCHD.d.ts +50 -0
- package/dist/runtime-trajectory-BvSZcCHD.d.ts.map +1 -0
- package/dist/schema-BtVldJ3T.d.ts +206 -0
- package/dist/schema-BtVldJ3T.d.ts.map +1 -0
- package/dist/schema-C6DW4ZHR.js +821 -0
- package/dist/schema-C6DW4ZHR.js.map +1 -0
- package/dist/schema-CRhEY1SO.js +69 -0
- package/dist/schema-CRhEY1SO.js.map +1 -0
- package/dist/schema-Cef2cFmb.d.ts +408 -0
- package/dist/schema-Cef2cFmb.d.ts.map +1 -0
- package/dist/semantic-concept-judge-B6cWNJ2K.js +725 -0
- package/dist/semantic-concept-judge-B6cWNJ2K.js.map +1 -0
- package/dist/sequential-Br0mAPHA.js +148 -0
- package/dist/sequential-Br0mAPHA.js.map +1 -0
- package/dist/sequential-CYwq6Ff_.d.ts +141 -0
- package/dist/sequential-CYwq6Ff_.d.ts.map +1 -0
- package/dist/series-convergence-CjO2QdRW.js +43 -0
- package/dist/series-convergence-CjO2QdRW.js.map +1 -0
- package/dist/series-convergence-ofsqPWhs.d.ts +35 -0
- package/dist/series-convergence-ofsqPWhs.d.ts.map +1 -0
- package/dist/server-m5D9cvnG.js +1040 -0
- package/dist/server-m5D9cvnG.js.map +1 -0
- package/dist/skill-usage-C_pXm7lP.d.ts +534 -0
- package/dist/skill-usage-C_pXm7lP.d.ts.map +1 -0
- package/dist/skillopt-optimization-method-B7wX7XkF.d.ts +1739 -0
- package/dist/skillopt-optimization-method-B7wX7XkF.d.ts.map +1 -0
- package/dist/skillopt-optimization-method-D4ODwFVV.js +7712 -0
- package/dist/skillopt-optimization-method-D4ODwFVV.js.map +1 -0
- package/dist/statistics-Cmj6nynr.d.ts +514 -0
- package/dist/statistics-Cmj6nynr.d.ts.map +1 -0
- package/dist/statistics-CnnxdpOg.js +1437 -0
- package/dist/statistics-CnnxdpOg.js.map +1 -0
- package/dist/store-CT9YIIve.d.ts +117 -0
- package/dist/store-CT9YIIve.d.ts.map +1 -0
- package/dist/store-CxJry_cs.d.ts +229 -0
- package/dist/store-CxJry_cs.d.ts.map +1 -0
- package/dist/storyboard/index.d.ts +84 -203
- package/dist/storyboard/index.d.ts.map +1 -0
- package/dist/storyboard/index.js +609 -542
- package/dist/storyboard/index.js.map +1 -1
- package/dist/summary-report-BNs5nmXI.js +862 -0
- package/dist/summary-report-BNs5nmXI.js.map +1 -0
- package/dist/summary-report-Cj9gdw4i.d.ts +407 -0
- package/dist/summary-report-Cj9gdw4i.d.ts.map +1 -0
- package/dist/supervisor-run/index.d.ts +2 -849
- package/dist/supervisor-run/index.js +2 -64
- package/dist/supervisor-run-_lnTLM3z.js +1679 -0
- package/dist/supervisor-run-_lnTLM3z.js.map +1 -0
- package/dist/task-failure-attributes-CQZlB3et.js +311 -0
- package/dist/task-failure-attributes-CQZlB3et.js.map +1 -0
- package/dist/test-graded-scenario-BsqWLmPt.js +318 -0
- package/dist/test-graded-scenario-BsqWLmPt.js.map +1 -0
- package/dist/test-graded-scenario-D1TaI2va.d.ts +141 -0
- package/dist/test-graded-scenario-D1TaI2va.d.ts.map +1 -0
- package/dist/tools-BmuN627J.js +1085 -0
- package/dist/tools-BmuN627J.js.map +1 -0
- package/dist/trace-attributes.d.ts +2 -52
- package/dist/trace-attributes.js +131 -61
- package/dist/trace-attributes.js.map +1 -1
- package/dist/traces.d.ts +12 -2365
- package/dist/traces.js +12 -251
- package/dist/trajectory-D_7rLrvE.js +56 -0
- package/dist/trajectory-D_7rLrvE.js.map +1 -0
- package/dist/types-DGsxbAEd.d.ts +387 -0
- package/dist/types-DGsxbAEd.d.ts.map +1 -0
- package/dist/types-k9tZGKUg.d.ts +640 -0
- package/dist/types-k9tZGKUg.d.ts.map +1 -0
- package/dist/verdict-Dps8_okt.d.ts +37 -0
- package/dist/verdict-Dps8_okt.d.ts.map +1 -0
- package/dist/wire/index.d.ts +702 -1174
- package/dist/wire/index.d.ts.map +1 -0
- package/dist/wire/index.js +2 -81
- package/docs/feature-guide.md +1 -1
- package/docs/rollout.md +116 -2
- package/package.json +18 -10
- package/dist/benchmarks/index.js.map +0 -1
- package/dist/campaign/index.js.map +0 -1
- package/dist/chunk-2JX3CFMB.js +0 -695
- package/dist/chunk-2JX3CFMB.js.map +0 -1
- package/dist/chunk-2MKQIFS4.js +0 -183
- package/dist/chunk-2MKQIFS4.js.map +0 -1
- package/dist/chunk-3RF76KTD.js +0 -84
- package/dist/chunk-3RF76KTD.js.map +0 -1
- package/dist/chunk-5DTSBUL2.js +0 -159
- package/dist/chunk-5DTSBUL2.js.map +0 -1
- package/dist/chunk-7ZZMD7UK.js +0 -386
- package/dist/chunk-7ZZMD7UK.js.map +0 -1
- package/dist/chunk-BOD4O7OF.js +0 -40
- package/dist/chunk-BOD4O7OF.js.map +0 -1
- package/dist/chunk-BYT7ELPS.js +0 -1553
- package/dist/chunk-BYT7ELPS.js.map +0 -1
- package/dist/chunk-DJKY2TSY.js +0 -2428
- package/dist/chunk-DJKY2TSY.js.map +0 -1
- package/dist/chunk-DPUHNQLN.js +0 -232
- package/dist/chunk-DPUHNQLN.js.map +0 -1
- package/dist/chunk-DRYIUNWY.js +0 -622
- package/dist/chunk-DRYIUNWY.js.map +0 -1
- package/dist/chunk-EJGRPCO3.js +0 -617
- package/dist/chunk-EJGRPCO3.js.map +0 -1
- package/dist/chunk-EOSZT7PL.js +0 -2001
- package/dist/chunk-EOSZT7PL.js.map +0 -1
- package/dist/chunk-EZJEIH2R.js +0 -1559
- package/dist/chunk-EZJEIH2R.js.map +0 -1
- package/dist/chunk-GGE4NNQT.js +0 -65
- package/dist/chunk-GGE4NNQT.js.map +0 -1
- package/dist/chunk-HHWE3POT.js +0 -94
- package/dist/chunk-HHWE3POT.js.map +0 -1
- package/dist/chunk-IHQDPH7D.js +0 -171
- package/dist/chunk-IHQDPH7D.js.map +0 -1
- package/dist/chunk-JHCHEVET.js +0 -274
- package/dist/chunk-JHCHEVET.js.map +0 -1
- package/dist/chunk-K4DBDHLK.js +0 -158
- package/dist/chunk-K4DBDHLK.js.map +0 -1
- package/dist/chunk-K6N6XJJX.js +0 -306
- package/dist/chunk-K6N6XJJX.js.map +0 -1
- package/dist/chunk-MA6HLL3S.js +0 -65
- package/dist/chunk-MA6HLL3S.js.map +0 -1
- package/dist/chunk-MAZ26DC7.js +0 -99
- package/dist/chunk-MAZ26DC7.js.map +0 -1
- package/dist/chunk-MHELPNRP.js +0 -1212
- package/dist/chunk-MHELPNRP.js.map +0 -1
- package/dist/chunk-NACAGYSY.js +0 -1040
- package/dist/chunk-NACAGYSY.js.map +0 -1
- package/dist/chunk-NKAGIDE2.js +0 -7633
- package/dist/chunk-NKAGIDE2.js.map +0 -1
- package/dist/chunk-NPCTHQIO.js +0 -91
- package/dist/chunk-NPCTHQIO.js.map +0 -1
- package/dist/chunk-NYLOYM6N.js +0 -332
- package/dist/chunk-NYLOYM6N.js.map +0 -1
- package/dist/chunk-ONWEPEDO.js +0 -57
- package/dist/chunk-ONWEPEDO.js.map +0 -1
- package/dist/chunk-P5W7RQKK.js +0 -576
- package/dist/chunk-P5W7RQKK.js.map +0 -1
- package/dist/chunk-P6FYH6K4.js +0 -1161
- package/dist/chunk-P6FYH6K4.js.map +0 -1
- package/dist/chunk-PBE2LOSS.js +0 -669
- package/dist/chunk-PBE2LOSS.js.map +0 -1
- package/dist/chunk-PC4UYEBM.js +0 -166
- package/dist/chunk-PC4UYEBM.js.map +0 -1
- package/dist/chunk-PXE2VKMX.js +0 -140
- package/dist/chunk-PXE2VKMX.js.map +0 -1
- package/dist/chunk-PZ5AY32C.js +0 -10
- package/dist/chunk-PZ5AY32C.js.map +0 -1
- package/dist/chunk-RZTMDUO7.js +0 -49
- package/dist/chunk-RZTMDUO7.js.map +0 -1
- package/dist/chunk-S5YLIBFX.js +0 -136
- package/dist/chunk-S5YLIBFX.js.map +0 -1
- package/dist/chunk-SZLVEKMJ.js +0 -1446
- package/dist/chunk-SZLVEKMJ.js.map +0 -1
- package/dist/chunk-T4SQEITX.js +0 -95
- package/dist/chunk-T4SQEITX.js.map +0 -1
- package/dist/chunk-TBL77AUT.js +0 -355
- package/dist/chunk-TBL77AUT.js.map +0 -1
- package/dist/chunk-TSN7JT6D.js +0 -1646
- package/dist/chunk-TSN7JT6D.js.map +0 -1
- package/dist/chunk-TT4KNT67.js +0 -124
- package/dist/chunk-TT4KNT67.js.map +0 -1
- package/dist/chunk-UB2LOJ6Q.js +0 -4461
- package/dist/chunk-UB2LOJ6Q.js.map +0 -1
- package/dist/chunk-UWZZKKU7.js +0 -237
- package/dist/chunk-UWZZKKU7.js.map +0 -1
- package/dist/chunk-VBQ3CRKH.js +0 -291
- package/dist/chunk-VBQ3CRKH.js.map +0 -1
- package/dist/chunk-VGRCHJON.js +0 -163
- package/dist/chunk-VGRCHJON.js.map +0 -1
- package/dist/chunk-VI2UW6B6.js +0 -162
- package/dist/chunk-VI2UW6B6.js.map +0 -1
- package/dist/chunk-VLOATJQ2.js +0 -908
- package/dist/chunk-VLOATJQ2.js.map +0 -1
- package/dist/chunk-VQMK5FMP.js +0 -247
- package/dist/chunk-VQMK5FMP.js.map +0 -1
- package/dist/chunk-VZSRQ272.js +0 -149
- package/dist/chunk-VZSRQ272.js.map +0 -1
- package/dist/chunk-WGXIEX7P.js +0 -116
- package/dist/chunk-WGXIEX7P.js.map +0 -1
- package/dist/chunk-WS3NZZQQ.js +0 -929
- package/dist/chunk-WS3NZZQQ.js.map +0 -1
- package/dist/chunk-XDWDC2MP.js +0 -695
- package/dist/chunk-XDWDC2MP.js.map +0 -1
- package/dist/chunk-XPRT64IE.js +0 -766
- package/dist/chunk-XPRT64IE.js.map +0 -1
- package/dist/chunk-YJBNWCAA.js +0 -1056
- package/dist/chunk-YJBNWCAA.js.map +0 -1
- package/dist/chunk-ZET2UAYW.js +0 -89
- package/dist/chunk-ZET2UAYW.js.map +0 -1
- package/dist/chunk-ZUUWPZCV.js +0 -752
- package/dist/chunk-ZUUWPZCV.js.map +0 -1
- package/dist/control.js.map +0 -1
- package/dist/hosted/index.js.map +0 -1
- package/dist/matrix/index.js.map +0 -1
- package/dist/reporting.js.map +0 -1
- package/dist/rollout/index.js.map +0 -1
- package/dist/run-campaign-ISHFZ7FJ.js +0 -17
- package/dist/run-campaign-ISHFZ7FJ.js.map +0 -1
- package/dist/supervisor-run/index.js.map +0 -1
- package/dist/traces.js.map +0 -1
- package/dist/wire/index.js.map +0 -1
|
@@ -0,0 +1,514 @@
|
|
|
1
|
+
import { g as JudgeScore } from "./types-DGsxbAEd.js";
|
|
2
|
+
import { i as ContinuousAgreementOptions, r as ContinuousAgreement } from "./judge-calibration-DFtEMlde.js";
|
|
3
|
+
//#region src/statistics.d.ts
|
|
4
|
+
/** Identity: dimensions already follow "higher = better" by prompt convention
|
|
5
|
+
* (inverted dims like hallucination are scored 10 = best at the source). */
|
|
6
|
+
declare const normalizeScores: (scores: JudgeScore[]) => JudgeScore[];
|
|
7
|
+
/** Weighted mean — falls back to uniform weights when omitted */
|
|
8
|
+
declare function weightedMean(scores: {
|
|
9
|
+
score: number;
|
|
10
|
+
weight?: number;
|
|
11
|
+
}[]): number;
|
|
12
|
+
/** Bootstrap confidence interval */
|
|
13
|
+
declare function confidenceInterval(scores: number[], confidence?: number, opts?: {
|
|
14
|
+
seed?: number;
|
|
15
|
+
resamples?: number;
|
|
16
|
+
}): {
|
|
17
|
+
mean: number;
|
|
18
|
+
lower: number;
|
|
19
|
+
upper: number;
|
|
20
|
+
};
|
|
21
|
+
/**
|
|
22
|
+
* Inter-rater reliability — simplified Krippendorff's alpha.
|
|
23
|
+
*
|
|
24
|
+
* Each inner array is one judge's scores for all items.
|
|
25
|
+
* All arrays must have the same length (same items scored).
|
|
26
|
+
*/
|
|
27
|
+
declare function interRaterReliability(judgeScores: JudgeScore[][]): number;
|
|
28
|
+
/**
|
|
29
|
+
* Mann-Whitney U test for comparing two independent groups.
|
|
30
|
+
* Returns U statistic and approximate p-value (normal approximation).
|
|
31
|
+
*/
|
|
32
|
+
declare function mannWhitneyU(a: number[], b: number[]): {
|
|
33
|
+
u: number;
|
|
34
|
+
p: number;
|
|
35
|
+
};
|
|
36
|
+
/** Partial credit: returns 0-1 ratio of current toward target */
|
|
37
|
+
declare function partialCredit(current: number, target: number): number;
|
|
38
|
+
/**
|
|
39
|
+
* Paired t-test — before/after measurements on the SAME items.
|
|
40
|
+
* Pairing removes inter-item variance, giving tighter significance than
|
|
41
|
+
* an unpaired test when comparing prompt v1 vs prompt v2 on identical
|
|
42
|
+
* scenarios.
|
|
43
|
+
*/
|
|
44
|
+
declare function pairedTTest(before: number[], after: number[]): {
|
|
45
|
+
t: number;
|
|
46
|
+
df: number;
|
|
47
|
+
p: number;
|
|
48
|
+
};
|
|
49
|
+
/**
|
|
50
|
+
* Wilcoxon signed-rank test — paired non-parametric alternative.
|
|
51
|
+
* Use when the differences aren't normally distributed.
|
|
52
|
+
*/
|
|
53
|
+
declare function wilcoxonSignedRank(before: number[], after: number[]): {
|
|
54
|
+
w: number;
|
|
55
|
+
p: number;
|
|
56
|
+
};
|
|
57
|
+
/**
|
|
58
|
+
* Cohen's d — standardized effect size for two independent groups.
|
|
59
|
+
* Positive d means group b has higher mean than group a.
|
|
60
|
+
* Rule of thumb: |d| < 0.2 negligible, 0.2–0.5 small, 0.5–0.8 medium, > 0.8 large.
|
|
61
|
+
*/
|
|
62
|
+
declare function cohensD(a: number[], b: number[]): number;
|
|
63
|
+
/**
|
|
64
|
+
* Cohen's dz for paired observations: mean(after - before) divided by the
|
|
65
|
+
* sample standard deviation of those within-pair deltas.
|
|
66
|
+
*
|
|
67
|
+
* Returns null when fewer than two pairs exist or a non-zero constant delta
|
|
68
|
+
* has zero observed variance. In that case the standardized effect is
|
|
69
|
+
* undefined, not an arbitrarily large finite number.
|
|
70
|
+
*/
|
|
71
|
+
declare function pairedCohensDz(before: number[], after: number[]): number | null;
|
|
72
|
+
type CliffsMagnitude = 'negligible' | 'small' | 'medium' | 'large';
|
|
73
|
+
/**
|
|
74
|
+
* Cliff's delta — a non-parametric effect size for two independent samples.
|
|
75
|
+
* `δ = (#(after > before) − #(after < before)) / (n_before · n_after)`,
|
|
76
|
+
* ranging [-1, 1]. Positive ⇒ `after` tends to exceed `before` (improvement).
|
|
77
|
+
*
|
|
78
|
+
* Distribution-free counterpart to Cohen's d: no normality assumption, robust
|
|
79
|
+
* to the bounded/skewed score distributions judges produce. Pairs with
|
|
80
|
+
* `pairedBootstrap` / `wilcoxonSignedRank` for the non-parametric reporting
|
|
81
|
+
* path. Returns 0 when either sample is empty.
|
|
82
|
+
*/
|
|
83
|
+
declare function cliffsDelta(before: number[], after: number[]): number;
|
|
84
|
+
/**
|
|
85
|
+
* Map a Cliff's delta to a qualitative magnitude using the standard
|
|
86
|
+
* Romano et al. thresholds (|δ|): <0.147 negligible, <0.33 small,
|
|
87
|
+
* <0.474 medium, else large.
|
|
88
|
+
*/
|
|
89
|
+
declare function interpretCliffs(delta: number): CliffsMagnitude;
|
|
90
|
+
/**
|
|
91
|
+
* Average-rank-with-ties transform (1-indexed). Tied values receive the mean
|
|
92
|
+
* of the ranks they span, the standard correction for Spearman's ρ.
|
|
93
|
+
*/
|
|
94
|
+
declare function ranks(xs: number[]): number[];
|
|
95
|
+
/**
|
|
96
|
+
* Pearson product-moment correlation coefficient r ∈ [-1, 1] between two
|
|
97
|
+
* equal-length series. See the edge-case contract above: NaN for n < 2 or
|
|
98
|
+
* unequal lengths, 1 when both series are constant, 0 when exactly one is.
|
|
99
|
+
*/
|
|
100
|
+
declare function pearsonR(a: number[], b: number[]): number;
|
|
101
|
+
/**
|
|
102
|
+
* Spearman's rank correlation ρ — Pearson over the average-rank-with-ties
|
|
103
|
+
* transform of each series. Same edge-case contract as {@link pearsonR}.
|
|
104
|
+
*/
|
|
105
|
+
declare function spearmanR(a: number[], b: number[]): number;
|
|
106
|
+
interface WeightedCompositeInput {
|
|
107
|
+
/** Per-dimension scores (typically 0..1). */
|
|
108
|
+
dims: Record<string, number>;
|
|
109
|
+
/** Weight per dimension. Every weighted dimension MUST be present in
|
|
110
|
+
* `dims` — a weight for an absent dimension is a config error and throws,
|
|
111
|
+
* because silently dropping it would renormalise the composite onto a
|
|
112
|
+
* different denominator than intended. */
|
|
113
|
+
weights: Record<string, number>;
|
|
114
|
+
/** Optional pass threshold; when set, the result reports `pass`. */
|
|
115
|
+
threshold?: number;
|
|
116
|
+
}
|
|
117
|
+
interface WeightedCompositeResult {
|
|
118
|
+
composite: number;
|
|
119
|
+
pass?: boolean;
|
|
120
|
+
}
|
|
121
|
+
/**
|
|
122
|
+
* Weighted composite over judge dimensions: `Σ(score_d · w_d) / Σ(w_d)` across
|
|
123
|
+
* the weighted dimensions. The canonical replacement for the per-consumer
|
|
124
|
+
* hand-rolled composite math (tax/legal/creative/gtm each ship a copy).
|
|
125
|
+
*
|
|
126
|
+
* Fail-loud: throws if a weighted dimension is missing from `dims`, if any
|
|
127
|
+
* weight is negative, or if the weights sum to 0 — none of which can produce
|
|
128
|
+
* a meaningful composite.
|
|
129
|
+
*/
|
|
130
|
+
declare function weightedComposite(input: WeightedCompositeInput): WeightedCompositeResult;
|
|
131
|
+
interface CorpusScoreRecord {
|
|
132
|
+
/** Stable identifier for the rated item (scenario, span, turn, …). */
|
|
133
|
+
itemId: string;
|
|
134
|
+
/** Identifier for the judge that produced this score. */
|
|
135
|
+
judgeName: string;
|
|
136
|
+
/** Dimension name (matches `JudgeScore.dimension`). */
|
|
137
|
+
dimension: string;
|
|
138
|
+
/** Numeric score; must be finite. */
|
|
139
|
+
score: number;
|
|
140
|
+
}
|
|
141
|
+
interface CorpusAgreementPerDimension extends ContinuousAgreement {
|
|
142
|
+
dimension: string;
|
|
143
|
+
/** Item IDs that contributed to this dimension's matrix (every judge scored them). */
|
|
144
|
+
itemIds: string[];
|
|
145
|
+
/** Judge IDs that contributed to this dimension's matrix. */
|
|
146
|
+
judgeIds: string[];
|
|
147
|
+
}
|
|
148
|
+
interface CorpusAgreementReport {
|
|
149
|
+
/** Per-dimension ICC(2,1) + κ_w + Pearson + Spearman + bootstrap CIs. */
|
|
150
|
+
perDimension: CorpusAgreementPerDimension[];
|
|
151
|
+
/** Mean ICC across dimensions (NaN if no dimension yielded a finite ICC). */
|
|
152
|
+
overallIcc: number;
|
|
153
|
+
/** Mean weighted κ across dimensions (NaN if none finite). */
|
|
154
|
+
overallWeightedKappa: number;
|
|
155
|
+
/** Dimensions evaluated (sorted). */
|
|
156
|
+
dimensions: string[];
|
|
157
|
+
/** Judges seen across the corpus (sorted). */
|
|
158
|
+
judgeIds: string[];
|
|
159
|
+
}
|
|
160
|
+
interface CorpusAgreementOptions extends ContinuousAgreementOptions {
|
|
161
|
+
/**
|
|
162
|
+
* Restrict the audit to these dimensions. Default = every dimension
|
|
163
|
+
* that appears in the input. A dimension named here but absent from
|
|
164
|
+
* the input throws — silent omission would corrupt the overall metric.
|
|
165
|
+
*/
|
|
166
|
+
dimensions?: string[];
|
|
167
|
+
/**
|
|
168
|
+
* Restrict the audit to these judges. Default = every judge that
|
|
169
|
+
* appears in the input. A judge named here but absent from a
|
|
170
|
+
* dimension throws (see "fail loud" below).
|
|
171
|
+
*/
|
|
172
|
+
judges?: string[];
|
|
173
|
+
}
|
|
174
|
+
/**
|
|
175
|
+
* Corpus-wide inter-rater agreement across N items × M judges × D dimensions.
|
|
176
|
+
*
|
|
177
|
+
* For each dimension, builds the [n_items][n_judges] matrix of scores
|
|
178
|
+
* (keeping only items every judge rated on that dimension), then runs
|
|
179
|
+
* `continuousAgreement` to get ICC(2,1), κ_w, Pearson, Spearman, and
|
|
180
|
+
* bootstrap CIs. Reports a pooled mean across dimensions as a single
|
|
181
|
+
* "is this judge panel reliable on this corpus?" number.
|
|
182
|
+
*
|
|
183
|
+
* Fail-loud contract:
|
|
184
|
+
* - Empty input throws.
|
|
185
|
+
* - Fewer than 2 judges or fewer than 2 items per dimension throws.
|
|
186
|
+
* - A judge present in some dimensions but with zero scored items on
|
|
187
|
+
* another dimension throws (would silently shrink the matrix).
|
|
188
|
+
* - Duplicate (itemId, judgeName, dimension) records throw.
|
|
189
|
+
*/
|
|
190
|
+
declare function corpusInterRaterAgreement(records: CorpusScoreRecord[], opts?: CorpusAgreementOptions): CorpusAgreementReport;
|
|
191
|
+
/**
|
|
192
|
+
* Convenience adapter for `JudgeScore[]` data keyed externally by item.
|
|
193
|
+
*
|
|
194
|
+
* Use when you have per-item arrays of `JudgeScore[]` (e.g. one
|
|
195
|
+
* `ScenarioResult.judgeScores` per scenario) and want corpus-wide
|
|
196
|
+
* agreement without manually flattening. `itemId` must be unique per
|
|
197
|
+
* row of `itemsScores`.
|
|
198
|
+
*/
|
|
199
|
+
declare function corpusInterRaterAgreementFromJudgeScores(itemsScores: Array<{
|
|
200
|
+
itemId: string;
|
|
201
|
+
scores: JudgeScore[];
|
|
202
|
+
}>, opts?: CorpusAgreementOptions): CorpusAgreementReport;
|
|
203
|
+
/**
|
|
204
|
+
* Required N per arm for a two-sample comparison at target effect size,
|
|
205
|
+
* alpha, and power. Normal-approximation formula:
|
|
206
|
+
* n = 2 * ( (z_{1-α/2} + z_{1-β}) / d )^2
|
|
207
|
+
* where d is Cohen's d. Returns Infinity for effect ≤ 0.
|
|
208
|
+
*/
|
|
209
|
+
declare function requiredSampleSize(opts: {
|
|
210
|
+
effect: number;
|
|
211
|
+
alpha?: number;
|
|
212
|
+
power?: number;
|
|
213
|
+
twoSided?: boolean;
|
|
214
|
+
}): number;
|
|
215
|
+
/**
|
|
216
|
+
* Required number of paired observations for a target Cohen's dz.
|
|
217
|
+
* Unlike the independent-groups formula, this has no two-arm factor of two.
|
|
218
|
+
*/
|
|
219
|
+
declare function requiredPairedSampleSize(opts: {
|
|
220
|
+
effect: number;
|
|
221
|
+
alpha?: number;
|
|
222
|
+
power?: number;
|
|
223
|
+
twoSided?: boolean;
|
|
224
|
+
}): number;
|
|
225
|
+
/**
|
|
226
|
+
* Minimum detectable paired effect (standardised units) for a target paired
|
|
227
|
+
* sample size: d_min = (z_{1-α/2} + z_β) / sqrt(n_paired). Multiply by
|
|
228
|
+
* sd(deltas) for score units; treat as a lower bound — Wilcoxon and bootstrap
|
|
229
|
+
* have asymptotic relative efficiency below 1 vs the t-test on heavy tails.
|
|
230
|
+
*/
|
|
231
|
+
declare function pairedMde(opts: {
|
|
232
|
+
nPaired: number;
|
|
233
|
+
alpha?: number;
|
|
234
|
+
power?: number;
|
|
235
|
+
twoSided?: boolean;
|
|
236
|
+
}): number;
|
|
237
|
+
/**
|
|
238
|
+
* Number of paired observations needed for a McNemar test to reach a target
|
|
239
|
+
* power — the pre-registration companion to {@link mcnemar}. Parametrised by the
|
|
240
|
+
* expected discordant-cell probabilities `p10` (P[treatment wins on a pair]) and
|
|
241
|
+
* `p01` (P[control wins]); concordant pairs carry no information, so the count
|
|
242
|
+
* is driven entirely by the discordant rate. Lachin's (1992) asymptotic normal
|
|
243
|
+
* approximation: with discordant rate `pDisc = p10 + p01` and marginal effect
|
|
244
|
+
* `δ = p10 − p01`,
|
|
245
|
+
* n = ( z_{1-α/2}·√pDisc + z_{1-β}·√(pDisc − δ²) )² / δ².
|
|
246
|
+
* Returns Infinity when there is no effect (p10 === p01). Asymptotic — at the
|
|
247
|
+
* tiny discordant counts where the exact {@link mcnemar} differs from the normal
|
|
248
|
+
* approximation, treat the result as a lower bound and prefer the discordant-pair
|
|
249
|
+
* floor.
|
|
250
|
+
*/
|
|
251
|
+
declare function mcnemarRequiredN(opts: {
|
|
252
|
+
p10: number;
|
|
253
|
+
p01: number;
|
|
254
|
+
alpha?: number;
|
|
255
|
+
power?: number;
|
|
256
|
+
twoSided?: boolean;
|
|
257
|
+
}): number;
|
|
258
|
+
/**
|
|
259
|
+
* Power of a McNemar test at a given number of paired observations, the inverse
|
|
260
|
+
* of {@link mcnemarRequiredN} (same Lachin asymptotic model, same parameters).
|
|
261
|
+
* Returns a value in [0, 1]; equals `alpha` when there is no effect.
|
|
262
|
+
*/
|
|
263
|
+
declare function mcnemarPower(opts: {
|
|
264
|
+
p10: number;
|
|
265
|
+
p01: number;
|
|
266
|
+
nPairs: number;
|
|
267
|
+
alpha?: number;
|
|
268
|
+
twoSided?: boolean;
|
|
269
|
+
}): number;
|
|
270
|
+
/** Bonferroni adjustment: multiply every p-value by the test count, clamp at 1. */
|
|
271
|
+
declare function bonferroni(pValues: number[], alpha?: number): {
|
|
272
|
+
adjusted: number[];
|
|
273
|
+
significant: boolean[];
|
|
274
|
+
};
|
|
275
|
+
/**
|
|
276
|
+
* Holm step-down family-wise error adjustment.
|
|
277
|
+
*
|
|
278
|
+
* P-values are sorted from smallest to largest, multiplied by their remaining
|
|
279
|
+
* hypothesis count, and made monotonically non-decreasing before being mapped
|
|
280
|
+
* back to input order. This uniformly dominates plain Bonferroni while keeping
|
|
281
|
+
* strong family-wise error control under arbitrary dependence.
|
|
282
|
+
*/
|
|
283
|
+
declare function holm(pValues: readonly number[], alpha?: number): {
|
|
284
|
+
adjusted: number[];
|
|
285
|
+
significant: boolean[];
|
|
286
|
+
};
|
|
287
|
+
/**
|
|
288
|
+
* Benjamini–Hochberg false discovery rate. Returns adjusted q-values and
|
|
289
|
+
* significance at the target FDR; handles ties and preserves q monotonicity.
|
|
290
|
+
*/
|
|
291
|
+
declare function benjaminiHochberg(pValues: number[], fdr?: number): {
|
|
292
|
+
qValues: number[];
|
|
293
|
+
significant: boolean[];
|
|
294
|
+
};
|
|
295
|
+
interface PairedBootstrapResult {
|
|
296
|
+
/** Number of paired observations. */
|
|
297
|
+
n: number;
|
|
298
|
+
/** Median of paired deltas (after − before). */
|
|
299
|
+
median: number;
|
|
300
|
+
/** Mean of paired deltas. */
|
|
301
|
+
mean: number;
|
|
302
|
+
/** Lower bound of the bootstrap CI on the chosen statistic. */
|
|
303
|
+
low: number;
|
|
304
|
+
/** Upper bound of the bootstrap CI on the chosen statistic. */
|
|
305
|
+
high: number;
|
|
306
|
+
/** Confidence level used (e.g. 0.95). */
|
|
307
|
+
confidence: number;
|
|
308
|
+
/** Number of bootstrap resamples used. */
|
|
309
|
+
resamples: number;
|
|
310
|
+
}
|
|
311
|
+
interface PairedBootstrapOptions {
|
|
312
|
+
/** Confidence level. Default 0.95. */
|
|
313
|
+
confidence?: number;
|
|
314
|
+
/** Bootstrap resample count. Default 2000. */
|
|
315
|
+
resamples?: number;
|
|
316
|
+
/** Statistic to bootstrap. Default 'median'. */
|
|
317
|
+
statistic?: 'median' | 'mean';
|
|
318
|
+
/** Deterministic seed. If omitted, uses Math.random(). */
|
|
319
|
+
seed?: number;
|
|
320
|
+
}
|
|
321
|
+
/**
|
|
322
|
+
* Paired bootstrap on (after − before) deltas. Returns a CI on the chosen
|
|
323
|
+
* statistic (median by default); pairs are resampled with replacement. The
|
|
324
|
+
* lower bound is what the promotion gate checks — `low > threshold` means the
|
|
325
|
+
* gain is real at the confidence level. Throws on unequal sample sizes.
|
|
326
|
+
*/
|
|
327
|
+
declare function pairedBootstrap(before: number[], after: number[], opts?: PairedBootstrapOptions): PairedBootstrapResult;
|
|
328
|
+
/** Pre-registered direction for a one-sided paired sign test. */
|
|
329
|
+
type SignTestAlternative = 'greater' | 'less';
|
|
330
|
+
/** Exact one-sided sign-test result for paired numeric differences. */
|
|
331
|
+
interface PairedSignTestResult {
|
|
332
|
+
/** Total supplied differences, including zero ties. */
|
|
333
|
+
n: number;
|
|
334
|
+
/** Strictly positive differences. */
|
|
335
|
+
positive: number;
|
|
336
|
+
/** Strictly negative differences. */
|
|
337
|
+
negative: number;
|
|
338
|
+
/** Zero differences excluded from the binomial test. */
|
|
339
|
+
ties: number;
|
|
340
|
+
/** Non-zero differences used by the binomial test. */
|
|
341
|
+
nNonTies: number;
|
|
342
|
+
/** Direction of the pre-registered alternative hypothesis. */
|
|
343
|
+
alternative: SignTestAlternative;
|
|
344
|
+
/** Exact one-sided p-value under P(positive) = P(negative) = 0.5. */
|
|
345
|
+
pValue: number;
|
|
346
|
+
}
|
|
347
|
+
/**
|
|
348
|
+
* Exact one-sided sign test over paired differences.
|
|
349
|
+
*
|
|
350
|
+
* Pass `after[i] - before[i]` for each matched item. `alternative = 'greater'`
|
|
351
|
+
* tests whether positive signs are more likely than negative signs and returns
|
|
352
|
+
* `P(Binomial(nNonTies, 0.5) >= positive)`. `alternative = 'less'` treats
|
|
353
|
+
* negative signs as successes instead. With a continuous difference
|
|
354
|
+
* distribution this is the usual directional median test. Exact zero
|
|
355
|
+
* differences are ties and do not enter the binomial denominator. All-tie and
|
|
356
|
+
* empty inputs return p = 1. Every input difference must be finite, and the
|
|
357
|
+
* direction must be chosen explicitly so a caller cannot select it after
|
|
358
|
+
* seeing the signs.
|
|
359
|
+
*/
|
|
360
|
+
declare function pairedSignTest(differences: readonly number[], alternative: SignTestAlternative): PairedSignTestResult;
|
|
361
|
+
/** A binomial proportion estimate with a confidence interval. */
|
|
362
|
+
interface ProportionInterval {
|
|
363
|
+
/** Point estimate successes / n (0 when n = 0). */
|
|
364
|
+
estimate: number;
|
|
365
|
+
/** Lower bound, clamped to [0, 1]. */
|
|
366
|
+
lower: number;
|
|
367
|
+
/** Upper bound, clamped to [0, 1]. */
|
|
368
|
+
upper: number;
|
|
369
|
+
}
|
|
370
|
+
/**
|
|
371
|
+
* Wilson score interval for a binomial proportion. Correct at small n and near
|
|
372
|
+
* 0/1, where the normal (Wald) approximation produces bounds outside [0, 1] and
|
|
373
|
+
* understates coverage. Use this for any pass-rate / hit-rate / realness-rate
|
|
374
|
+
* CI — the continuous `confidenceInterval` assumes the wrong distribution for a
|
|
375
|
+
* proportion. `n = 0 ⇒ {0, 0, 0}`.
|
|
376
|
+
*/
|
|
377
|
+
declare function wilson(successes: number, n: number, confidence?: number): ProportionInterval;
|
|
378
|
+
/** Result of a McNemar paired-binary significance test. */
|
|
379
|
+
interface McNemarResult {
|
|
380
|
+
/** Total paired observations. */
|
|
381
|
+
n: number;
|
|
382
|
+
/** Discordant pairs (b + c) — the only ones that carry signal. */
|
|
383
|
+
nDiscordant: number;
|
|
384
|
+
/** Pairs where treatment succeeded and control failed ("newly correct"). */
|
|
385
|
+
b: number;
|
|
386
|
+
/** Pairs where control succeeded and treatment failed ("newly wrong"). */
|
|
387
|
+
c: number;
|
|
388
|
+
/** Continuity-corrected chi-square statistic (reference; exact p drives the call). */
|
|
389
|
+
statistic: number;
|
|
390
|
+
/** Two-sided p-value. Exact (binomial sign test on discordant pairs). */
|
|
391
|
+
pValue: number;
|
|
392
|
+
}
|
|
393
|
+
/**
|
|
394
|
+
* McNemar's test for paired binary outcomes — the correct significance test for
|
|
395
|
+
* "does treatment change the success rate vs control on the SAME items". Only
|
|
396
|
+
* discordant pairs (one arm right, the other wrong) carry information; concordant
|
|
397
|
+
* pairs are uninformative, so a paired t-test / two-proportion z-test on the raw
|
|
398
|
+
* rates is wrong here. The p-value is exact: under H0 the b "treatment-wins" are
|
|
399
|
+
* Binomial(b + c, 0.5), so the two-sided p is the doubled binomial tail — correct
|
|
400
|
+
* at the small discordant counts typical of eval runs (no continuity-corrected
|
|
401
|
+
* chi-square approximation needed, though it is returned as `statistic` for
|
|
402
|
+
* reference). Inputs are paired 0/1 (or boolean) arrays, control first to match
|
|
403
|
+
* the module's (before, after) convention. Throws on unequal lengths.
|
|
404
|
+
*/
|
|
405
|
+
declare function mcnemar(control: ArrayLike<number | boolean>, treatment: ArrayLike<number | boolean>): McNemarResult;
|
|
406
|
+
/** A paired binary effect size (treatment rate − control rate) with a CI. */
|
|
407
|
+
interface RiskDifferenceResult {
|
|
408
|
+
/** Total paired observations. */
|
|
409
|
+
n: number;
|
|
410
|
+
/** Discordant pairs: treatment-win count. */
|
|
411
|
+
b: number;
|
|
412
|
+
/** Discordant pairs: control-win count. */
|
|
413
|
+
c: number;
|
|
414
|
+
/** Paired risk difference p(treatment) − p(control) = (b − c) / n. */
|
|
415
|
+
riskDifference: number;
|
|
416
|
+
/** Lower bound of the CI, clamped to [-1, 1]. */
|
|
417
|
+
lower: number;
|
|
418
|
+
/** Upper bound of the CI, clamped to [-1, 1]. */
|
|
419
|
+
upper: number;
|
|
420
|
+
/** Confidence level used. */
|
|
421
|
+
confidence: number;
|
|
422
|
+
}
|
|
423
|
+
/**
|
|
424
|
+
* Paired risk difference (the effect-size companion to {@link mcnemar}): the
|
|
425
|
+
* change in success rate p(treatment) − p(control) on matched items, which for
|
|
426
|
+
* paired binary data equals (b − c) / n. The CI uses the paired variance from
|
|
427
|
+
* the discordant counts, not the independent-samples formula (which overstates
|
|
428
|
+
* the interval by ignoring the pairing). Inputs are paired 0/1 (or boolean)
|
|
429
|
+
* arrays, control first. Throws on unequal lengths.
|
|
430
|
+
*/
|
|
431
|
+
declare function pairedRiskDifference(control: ArrayLike<number | boolean>, treatment: ArrayLike<number | boolean>, confidence?: number): RiskDifferenceResult;
|
|
432
|
+
/**
|
|
433
|
+
* Unbiased pass@k for code generation (Chen et al. 2021, "Evaluating Large
|
|
434
|
+
* Language Models Trained on Code"). Given `n` independent samples for one
|
|
435
|
+
* problem of which `c` pass, the probability that at least one of a random k of
|
|
436
|
+
* them passes is 1 − C(n−c, k) / C(n, k). Estimating pass@k as "did any of the
|
|
437
|
+
* first k pass" is biased high at small n; this is the variance-reduced estimator
|
|
438
|
+
* averaged implicitly over all k-subsets. Average the per-problem values across
|
|
439
|
+
* the suite for the corpus pass@k. Computed in the numerically stable product
|
|
440
|
+
* form. Requires 1 ≤ k ≤ n and 0 ≤ c ≤ n.
|
|
441
|
+
*/
|
|
442
|
+
declare function passAtK(n: number, c: number, k: number): number;
|
|
443
|
+
interface EProcessOptions {
|
|
444
|
+
/** Type-I error budget. The process decides when wealth ≥ 1/alpha
|
|
445
|
+
* (Ville's inequality). Default 0.05. */
|
|
446
|
+
alpha?: number;
|
|
447
|
+
/** Truncation bound on the predictable bet λ ∈ [0, maxBet]. Must satisfy
|
|
448
|
+
* maxBet < 1/nullMean so every wealth factor stays strictly positive.
|
|
449
|
+
* Default 0.5. */
|
|
450
|
+
maxBet?: number;
|
|
451
|
+
/** The null boundary m₀ for H0: E[x] ≤ m₀ on x ∈ [0,1]. Default 0.5
|
|
452
|
+
* (the paired-delta encoding x = (d+1)/2 maps "no effect" to 1/2).
|
|
453
|
+
* A pre-registered minEffect shifts this — see `sequentialPairedGate`. */
|
|
454
|
+
nullMean?: number;
|
|
455
|
+
}
|
|
456
|
+
interface EProcessStep {
|
|
457
|
+
/** Current wealth W_n — the e-value against H0 after n observations. */
|
|
458
|
+
wealth: number;
|
|
459
|
+
/** Observations consumed so far. */
|
|
460
|
+
n: number;
|
|
461
|
+
/** True from the first n where W_n ≥ 1/alpha onward (sticky). */
|
|
462
|
+
decided: boolean;
|
|
463
|
+
}
|
|
464
|
+
interface EProcessState extends EProcessStep {
|
|
465
|
+
alpha: number;
|
|
466
|
+
maxBet: number;
|
|
467
|
+
nullMean: number;
|
|
468
|
+
/** The decision boundary 1/alpha. */
|
|
469
|
+
threshold: number;
|
|
470
|
+
/** Observation count at the first threshold crossing; undefined until decided. */
|
|
471
|
+
decidedAtN?: number;
|
|
472
|
+
}
|
|
473
|
+
interface EProcess {
|
|
474
|
+
/** Consume one observation x ∈ [0,1]. Throws on non-finite / out-of-range
|
|
475
|
+
* input — a silent clamp would corrupt the type-I guarantee. */
|
|
476
|
+
update(x: number): EProcessStep;
|
|
477
|
+
state(): EProcessState;
|
|
478
|
+
}
|
|
479
|
+
/**
|
|
480
|
+
* Betting test-martingale for bounded observations — the e-process core of
|
|
481
|
+
* anytime-valid sequential testing (Waudby-Smith & Ramdas, "Estimating means
|
|
482
|
+
* of bounded random variables by betting", JRSS-B 2024).
|
|
483
|
+
*
|
|
484
|
+
* Observations x_i ∈ [0,1]; H0: E[x] ≤ m₀ (`nullMean`, default 1/2). Wealth
|
|
485
|
+
*
|
|
486
|
+
* W_t = Π_{i≤t} (1 + λ_i (x_i − m₀)), W_0 = 1
|
|
487
|
+
*
|
|
488
|
+
* with the truncated GROW-style plug-in bet computed from PRIOR observations:
|
|
489
|
+
*
|
|
490
|
+
* λ_i = clamp((μ̂_{i−1} − m₀) / (σ̂²_{i−1} + (μ̂_{i−1} − m₀)²), 0, maxBet)
|
|
491
|
+
*
|
|
492
|
+
* where μ̂/σ̂² are the shrunk running estimates μ̂_t = (1/2 + Σx_i)/(t+1),
|
|
493
|
+
* σ̂²_t = (1/4 + Σ(x_i − μ̂_i)²)/(t+1).
|
|
494
|
+
*
|
|
495
|
+
* PREDICTABILITY INVARIANT (load-bearing): λ_i is a function of x_1..x_{i−1}
|
|
496
|
+
* ONLY — it may never see x_i. With λ_i ≥ 0 predictable, each factor has
|
|
497
|
+
* E[1 + λ_i(x_i − m₀) | past] ≤ 1 under H0, so W is a nonnegative
|
|
498
|
+
* supermartingale and Ville's inequality gives P(∃t: W_t ≥ 1/α) ≤ α — the
|
|
499
|
+
* type-I guarantee holds at ANY data-dependent stopping time. λ_1 is always 0
|
|
500
|
+
* (no prior evidence), so the first observation never moves wealth.
|
|
501
|
+
*
|
|
502
|
+
* `decided` latches at the first crossing W_t ≥ 1/α and never un-latches;
|
|
503
|
+
* wealth keeps updating after the crossing (the e-process remains valid), but
|
|
504
|
+
* the decision time is the first crossing.
|
|
505
|
+
*/
|
|
506
|
+
declare function eProcess(opts?: EProcessOptions): EProcess;
|
|
507
|
+
/** Tiny seedable PRNG (mulberry32) — deterministic resampling/shuffling, not
|
|
508
|
+
* cryptographic. Exported so e-process shuffles and bootstrap resampling
|
|
509
|
+
* share ONE PRNG implementation; a seed is REQUIRED (unseeded randomness in
|
|
510
|
+
* gate verdicts is non-reproducible by construction). */
|
|
511
|
+
declare function mulberry32(seed: number): () => number;
|
|
512
|
+
//#endregion
|
|
513
|
+
export { mannWhitneyU as A, pairedSignTest as B, confidenceInterval as C, holm as D, eProcess as E, normalizeScores as F, ranks as G, partialCredit as H, pairedBootstrap as I, spearmanR as J, requiredPairedSampleSize as K, pairedCohensDz as L, mcnemarPower as M, mcnemarRequiredN as N, interRaterReliability as O, mulberry32 as P, wilson as Q, pairedMde as R, cohensD as S, corpusInterRaterAgreementFromJudgeScores as T, passAtK as U, pairedTTest as V, pearsonR as W, weightedMean as X, weightedComposite as Y, wilcoxonSignedRank as Z, WeightedCompositeInput as _, CorpusScoreRecord as a, bonferroni as b, EProcessState as c, PairedBootstrapOptions as d, PairedBootstrapResult as f, SignTestAlternative as g, RiskDifferenceResult as h, CorpusAgreementReport as i, mcnemar as j, interpretCliffs as k, EProcessStep as l, ProportionInterval as m, CorpusAgreementOptions as n, EProcess as o, PairedSignTestResult as p, requiredSampleSize as q, CorpusAgreementPerDimension as r, EProcessOptions as s, CliffsMagnitude as t, McNemarResult as u, WeightedCompositeResult as v, corpusInterRaterAgreement as w, cliffsDelta as x, benjaminiHochberg as y, pairedRiskDifference as z };
|
|
514
|
+
//# sourceMappingURL=statistics-Cmj6nynr.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"statistics-Cmj6nynr.d.ts","names":[],"sources":["../src/statistics.ts"],"mappings":";;;;;cAUa,kBAAe,QAAY,iBAAe;;iBAGvC,aAAa;EAAU;EAAe;;;iBAatC,mBACd,kBACA,qBACA;EAAQ;EAAe;;EACpB;EAAc;EAAe;;;;;;;;iBAsClB,sBAAsB,aAAa;;;;;iBAwDnC,aAAa,aAAa;EAAgB;EAAW;;;iBA+CrD,cAAc,iBAAiB;;;;;;;iBAW/B,YACd,kBACA;EACG;EAAW;EAAY;;;;;;iBAyBZ,mBAAmB,kBAAkB;EAAoB;EAAW;;;;;;;iBAqCpE,QAAQ,aAAa;;;;;;;;;iBAqBrB,eAAe,kBAAkB;KAoBrC;;;;;;;;;;;iBAYI,YAAY,kBAAkB;;;;;;iBAkB9B,gBAAgB,gBAAgB;;;;;iBAuBhC,MAAM;;;;;;iBAmBN,SAAS,aAAa;;;;;iBAuBtB,UAAU,aAAa;UAKtB;;EAEf,MAAM;;;;;EAKN,SAAS;;EAET;;UAGe;EACf;EACA;;;;;;;;;;;iBAYc,kBAAkB,OAAO,yBAAyB;UA4CjD;;EAEf;;EAEA;;EAEA;;EAEA;;UAGe,oCAAoC;EACnD;;EAEA;;EAEA;;UAGe;;EAEf,cAAc;;EAEd;;EAEA;;EAEA;;EAEA;;UAGe,+BAA+B;;;;;;EAM9C;;;;;;EAMA;;;;;;;;;;;;;;;;;;iBAmBc,0BACd,SAAS,qBACT,OAAM,yBACL;;;;;;;;;iBA0Ha,yCACd,aAAa;EAAQ;EAAgB,QAAQ;IAC7C,OAAM,yBACL;;;;;;;iBA+Ga,mBAAmB;EACjC;EACA;EACA;EACA;;;;;;iBAiBc,yBAAyB;EACvC;EACA;EACA;EACA;;;;;;;;iBAkBc,UAAU;EACxB;EACA;EACA;EACA;;;;;;;;;;;;;;;;iBAyBc,iBAAiB;EAC/B;EACA;EACA;EACA;EACA;;;;;;;iBAyBc,aAAa;EAC3B;EACA;EACA;EACA;EACA;;;iBAkBc,WACd,mBACA;EACG;EAAoB;;;;;;;;;;iBAeT,KACd,4BACA;EACG;EAAoB;;;;;;iBA+BT,kBACd,mBACA;EACG;EAAmB;;UAoBP;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;UAGe;;EAEf;;EAEA;;EAEA;;EAEA;;;;;;;;iBASc,gBACd,kBACA,iBACA,OAAM,yBACL;;KAwDS;;UAGK;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA,aAAa;;EAEb;;;;;;;;;;;;;;;iBAgBc,eACd,gCACA,aAAa,sBACZ;;UAgDc;;EAEf;;EAEA;;EAEA;;;;;;;;;iBAUc,OAAO,mBAAmB,WAAW,sBAAoB;;UAmBxD;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;;;;;;;iBAec,QACd,SAAS,6BACT,WAAW,8BACV;;UAmBc;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;;;iBAWc,qBACd,SAAS,6BACT,WAAW,6BACX,sBACC;;;;;;;;;;;iBAyCa,QAAQ,WAAW,WAAW;UAgE7B;;;EAGf;;;;EAIA;;;;EAIA;;UAGe;;EAEf;;EAEA;;EAEA;;UAGe,sBAAsB;EACrC;EACA;EACA;;EAEA;;EAEA;;UAGe;;;EAGf,OAAO,YAAY;EACnB,SAAS;;;;;;;;;;;;;;;;;;;;;;;;;;;;;iBA8BK,SAAS,OAAM,kBAAuB;;;;;iBAsHtC,WAAW"}
|