@tangle-network/agent-eval 0.128.2 → 0.130.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +279 -0
- package/README.md +19 -0
- package/dist/active-curriculum-C4mk67HP.js +214 -0
- package/dist/active-curriculum-C4mk67HP.js.map +1 -0
- package/dist/adversarial-smnADNFS.d.ts +21 -0
- package/dist/adversarial-smnADNFS.d.ts.map +1 -0
- package/dist/analyst/index.d.ts +83 -2932
- package/dist/analyst/index.d.ts.map +1 -0
- package/dist/analyst/index.js +319 -364
- package/dist/analyst/index.js.map +1 -1
- package/dist/analyst-BkTS3C58.d.ts +89 -0
- package/dist/analyst-BkTS3C58.d.ts.map +1 -0
- package/dist/analyst-LsnNpSkm.js +152 -0
- package/dist/analyst-LsnNpSkm.js.map +1 -0
- package/dist/analyze-runs-C1CavBMk.js +1067 -0
- package/dist/analyze-runs-C1CavBMk.js.map +1 -0
- package/dist/analyze-runs-FsgYCinh.d.ts +72 -0
- package/dist/analyze-runs-FsgYCinh.d.ts.map +1 -0
- package/dist/attribute-vocabulary-DLJ6303h.d.ts +54 -0
- package/dist/attribute-vocabulary-DLJ6303h.d.ts.map +1 -0
- package/dist/authenticity/index.d.ts +81 -79
- package/dist/authenticity/index.d.ts.map +1 -0
- package/dist/authenticity/index.js +209 -193
- package/dist/authenticity/index.js.map +1 -1
- package/dist/baseline-HsBvw_dk.js +550 -0
- package/dist/baseline-HsBvw_dk.js.map +1 -0
- package/dist/baseline-hG3K85h4.d.ts +125 -0
- package/dist/baseline-hG3K85h4.d.ts.map +1 -0
- package/dist/belief-state/index.d.ts +448 -1205
- package/dist/belief-state/index.d.ts.map +1 -0
- package/dist/belief-state/index.js +1617 -1710
- package/dist/belief-state/index.js.map +1 -1
- package/dist/benchmarks/index.d.ts +2 -894
- package/dist/benchmarks/index.js +2 -59
- package/dist/benchmarks-DviOvUNr.js +754 -0
- package/dist/benchmarks-DviOvUNr.js.map +1 -0
- package/dist/builder-eval/index.d.ts +150 -662
- package/dist/builder-eval/index.d.ts.map +1 -0
- package/dist/builder-eval/index.js +356 -345
- package/dist/builder-eval/index.js.map +1 -1
- package/dist/calibration-CNWWA6K8.js +94 -0
- package/dist/calibration-CNWWA6K8.js.map +1 -0
- package/dist/campaign/index.d.ts +5 -6390
- package/dist/campaign/index.js +3 -212
- package/dist/campaign-CBKZvQ1H.js +3885 -0
- package/dist/campaign-CBKZvQ1H.js.map +1 -0
- package/dist/cli.d.ts +1 -1
- package/dist/cli.js +164 -174
- package/dist/cli.js.map +1 -1
- package/dist/client-C97NMzqi.d.ts +581 -0
- package/dist/client-C97NMzqi.d.ts.map +1 -0
- package/dist/client-CYzbdJOZ.js +637 -0
- package/dist/client-CYzbdJOZ.js.map +1 -0
- package/dist/code-agent-session-BjkMTQ7H.js +1390 -0
- package/dist/code-agent-session-BjkMTQ7H.js.map +1 -0
- package/dist/code-agent-session-DqqgOJaz.d.ts +143 -0
- package/dist/code-agent-session-DqqgOJaz.d.ts.map +1 -0
- package/dist/concurrency-DIxRZF_J.js +85 -0
- package/dist/concurrency-DIxRZF_J.js.map +1 -0
- package/dist/contract/index.d.ts +645 -5605
- package/dist/contract/index.d.ts.map +1 -0
- package/dist/contract/index.js +1730 -1937
- package/dist/contract/index.js.map +1 -1
- package/dist/control.d.ts +3 -1030
- package/dist/control.js +2 -32
- package/dist/cost-ledger-DIgQUFZZ.js +801 -0
- package/dist/cost-ledger-DIgQUFZZ.js.map +1 -0
- package/dist/cost-ledger-Dye6jCgg.d.ts +236 -0
- package/dist/cost-ledger-Dye6jCgg.d.ts.map +1 -0
- package/dist/dataset-BvtnC8Dc.d.ts +115 -0
- package/dist/dataset-BvtnC8Dc.d.ts.map +1 -0
- package/dist/default-registry-C-vFCSEc.js +2579 -0
- package/dist/default-registry-C-vFCSEc.js.map +1 -0
- package/dist/default-registry-CNPo-Vsb.d.ts +540 -0
- package/dist/default-registry-CNPo-Vsb.d.ts.map +1 -0
- package/dist/emitter-CPBAhxum.js +266 -0
- package/dist/emitter-CPBAhxum.js.map +1 -0
- package/dist/emitter-DGQGoLyj.d.ts +113 -0
- package/dist/emitter-DGQGoLyj.d.ts.map +1 -0
- package/dist/errors-8YnH8WlF.js +64 -0
- package/dist/errors-8YnH8WlF.js.map +1 -0
- package/dist/errors-CEk209JS.d.ts +76 -0
- package/dist/errors-CEk209JS.d.ts.map +1 -0
- package/dist/eval-campaign-DEm6c8ru.js +349 -0
- package/dist/eval-campaign-DEm6c8ru.js.map +1 -0
- package/dist/execution-tracks-CpgFPpS5.js +93 -0
- package/dist/execution-tracks-CpgFPpS5.js.map +1 -0
- package/dist/exporters-q9iL-2Jf.js +148 -0
- package/dist/exporters-q9iL-2Jf.js.map +1 -0
- package/dist/extract-usage-BrQ8mCLX.js +155 -0
- package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
- package/dist/failure-cluster-CqcvCcdR.d.ts +59 -0
- package/dist/failure-cluster-CqcvCcdR.d.ts.map +1 -0
- package/dist/feedback-trajectory-CVaeREXV.d.ts +340 -0
- package/dist/feedback-trajectory-CVaeREXV.d.ts.map +1 -0
- package/dist/fuzz.d.ts +320 -646
- package/dist/fuzz.d.ts.map +1 -0
- package/dist/fuzz.js +670 -617
- package/dist/fuzz.js.map +1 -1
- package/dist/hf-dataset-DBJXXoY1.js +763 -0
- package/dist/hf-dataset-DBJXXoY1.js.map +1 -0
- package/dist/hosted/index.d.ts +11 -831
- package/dist/hosted/index.d.ts.map +1 -0
- package/dist/hosted/index.js +2 -37
- package/dist/index-2JJSA6-r2.d.ts +926 -0
- package/dist/index-2JJSA6-r2.d.ts.map +1 -0
- package/dist/index-6N0aYmpW.d.ts +217 -0
- package/dist/index-6N0aYmpW.d.ts.map +1 -0
- package/dist/index-BAvgST_9.d.ts +131 -0
- package/dist/index-BAvgST_9.d.ts.map +1 -0
- package/dist/index-BvnJuTGD.d.ts +68 -0
- package/dist/index-BvnJuTGD.d.ts.map +1 -0
- package/dist/index-C61Wi7yg.d.ts +547 -0
- package/dist/index-C61Wi7yg.d.ts.map +1 -0
- package/dist/index-CAPUUKaM.d.ts +335 -0
- package/dist/index-CAPUUKaM.d.ts.map +1 -0
- package/dist/index-DE5fb3EC.d.ts +2244 -0
- package/dist/index-DE5fb3EC.d.ts.map +1 -0
- package/dist/index-DSC51roc.d.ts +102 -0
- package/dist/index-DSC51roc.d.ts.map +1 -0
- package/dist/index.d.ts +3776 -15120
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +11185 -11191
- package/dist/index.js.map +1 -1
- package/dist/integrity-BzRbCHzi.js +424 -0
- package/dist/integrity-BzRbCHzi.js.map +1 -0
- package/dist/integrity-rmVhXWA7.d.ts +61 -0
- package/dist/integrity-rmVhXWA7.d.ts.map +1 -0
- package/dist/judge-calibration-DFtEMlde.d.ts +146 -0
- package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
- package/dist/ledger-core/index.d.ts +2 -0
- package/dist/ledger-core/index.js +2 -0
- package/dist/ledger-core-DtZz1RG0.js +388 -0
- package/dist/ledger-core-DtZz1RG0.js.map +1 -0
- package/dist/llm-client--GR4JbZE.js +687 -0
- package/dist/llm-client--GR4JbZE.js.map +1 -0
- package/dist/llm-client-B_nIBlYo.d.ts +290 -0
- package/dist/llm-client-B_nIBlYo.d.ts.map +1 -0
- package/dist/matrix/index.d.ts +3 -155
- package/dist/matrix/index.js +2 -8
- package/dist/matrix-BzQnu2S6.js +270 -0
- package/dist/matrix-BzQnu2S6.js.map +1 -0
- package/dist/meta-eval/index.d.ts +4 -1027
- package/dist/meta-eval/index.js +393 -390
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/metrics-C9YY1OcL.js +239 -0
- package/dist/metrics-C9YY1OcL.js.map +1 -0
- package/dist/mint-yN2M2eh0.js +201 -0
- package/dist/mint-yN2M2eh0.js.map +1 -0
- package/dist/multi-layer-verifier-BHY1gWAc.d.ts +138 -0
- package/dist/multi-layer-verifier-BHY1gWAc.d.ts.map +1 -0
- package/dist/multishot/index.d.ts +273 -481
- package/dist/multishot/index.d.ts.map +1 -0
- package/dist/multishot/index.js +595 -548
- package/dist/multishot/index.js.map +1 -1
- package/dist/off-policy-DvgzvtIx.js +220 -0
- package/dist/off-policy-DvgzvtIx.js.map +1 -0
- package/dist/off-policy-mskQw8Mb.d.ts +153 -0
- package/dist/off-policy-mskQw8Mb.d.ts.map +1 -0
- package/dist/openapi.json +1 -1
- package/dist/opencode-sqlite-BGrHeDu3.js +318 -0
- package/dist/opencode-sqlite-BGrHeDu3.js.map +1 -0
- package/dist/outcome-store-BYHIuO0e.d.ts +65 -0
- package/dist/outcome-store-BYHIuO0e.d.ts.map +1 -0
- package/dist/outcome-store-ChBKlTd_.js +75 -0
- package/dist/outcome-store-ChBKlTd_.js.map +1 -0
- package/dist/paired-arms-D9D0wXj2.js +260 -0
- package/dist/paired-arms-D9D0wXj2.js.map +1 -0
- package/dist/pipelines/index.d.ts +95 -532
- package/dist/pipelines/index.d.ts.map +1 -0
- package/dist/pipelines/index.js +497 -478
- package/dist/pipelines/index.js.map +1 -1
- package/dist/pre-registration-DakwTRXk.js +96 -0
- package/dist/pre-registration-DakwTRXk.js.map +1 -0
- package/dist/propose-review-control-LhIWGzHl.js +1458 -0
- package/dist/propose-review-control-LhIWGzHl.js.map +1 -0
- package/dist/query-CJ_DX8vl.d.ts +28 -0
- package/dist/query-CJ_DX8vl.d.ts.map +1 -0
- package/dist/query-Di7eEQ79.js +83 -0
- package/dist/query-Di7eEQ79.js.map +1 -0
- package/dist/raw-provider-sink-BQd7mzyT.js +201 -0
- package/dist/raw-provider-sink-BQd7mzyT.js.map +1 -0
- package/dist/raw-provider-sink-BU29Sh8h.d.ts +134 -0
- package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +1 -0
- package/dist/redact-7Aq1ukl-.js +107 -0
- package/dist/redact-7Aq1ukl-.js.map +1 -0
- package/dist/release-report-Cz9NKH39.js +603 -0
- package/dist/release-report-Cz9NKH39.js.map +1 -0
- package/dist/release-report-mvB2G4_J.d.ts +244 -0
- package/dist/release-report-mvB2G4_J.d.ts.map +1 -0
- package/dist/replay-GnyotH0J.js +1741 -0
- package/dist/replay-GnyotH0J.js.map +1 -0
- package/dist/replay-OoidtG1E.d.ts +749 -0
- package/dist/replay-OoidtG1E.d.ts.map +1 -0
- package/dist/reporting.d.ts +6 -1298
- package/dist/reporting.js +6 -50
- package/dist/researcher-CwTdwXG1.d.ts +314 -0
- package/dist/researcher-CwTdwXG1.d.ts.map +1 -0
- package/dist/reward-hacking-eAnOsynk.d.ts +224 -0
- package/dist/reward-hacking-eAnOsynk.d.ts.map +1 -0
- package/dist/reward-hacking-qipEpKvY.js +596 -0
- package/dist/reward-hacking-qipEpKvY.js.map +1 -0
- package/dist/reward-nw2xZGZG.js +137 -0
- package/dist/reward-nw2xZGZG.js.map +1 -0
- package/dist/rl.d.ts +916 -3596
- package/dist/rl.d.ts.map +1 -0
- package/dist/rl.js +2362 -1751
- package/dist/rl.js.map +1 -1
- package/dist/rolldown-runtime-8H4AJuhK.js +14 -0
- package/dist/rollout/index.d.ts +3 -1048
- package/dist/rollout/index.js +8 -110
- package/dist/rollout-BOYjemfR.js +624 -0
- package/dist/rollout-BOYjemfR.js.map +1 -0
- package/dist/rubric-predictive-validity-B3xmbmS1.js +141 -0
- package/dist/rubric-predictive-validity-B3xmbmS1.js.map +1 -0
- package/dist/rubric-predictive-validity-Ku_clp_1.d.ts +74 -0
- package/dist/rubric-predictive-validity-Ku_clp_1.d.ts.map +1 -0
- package/dist/run-evidence-ClRX_8A9.d.ts +225 -0
- package/dist/run-evidence-ClRX_8A9.d.ts.map +1 -0
- package/dist/run-record-BuoE80Dq.js +467 -0
- package/dist/run-record-BuoE80Dq.js.map +1 -0
- package/dist/run-record-CnZu_gjl.d.ts +357 -0
- package/dist/run-record-CnZu_gjl.d.ts.map +1 -0
- package/dist/run-score-iEEAWiBY.js +41 -0
- package/dist/run-score-iEEAWiBY.js.map +1 -0
- package/dist/runtime-trajectory-1gyaTOoC.js +93 -0
- package/dist/runtime-trajectory-1gyaTOoC.js.map +1 -0
- package/dist/runtime-trajectory-BvSZcCHD.d.ts +50 -0
- package/dist/runtime-trajectory-BvSZcCHD.d.ts.map +1 -0
- package/dist/schema-BtVldJ3T.d.ts +206 -0
- package/dist/schema-BtVldJ3T.d.ts.map +1 -0
- package/dist/schema-C6DW4ZHR.js +821 -0
- package/dist/schema-C6DW4ZHR.js.map +1 -0
- package/dist/schema-CRhEY1SO.js +69 -0
- package/dist/schema-CRhEY1SO.js.map +1 -0
- package/dist/schema-Cef2cFmb.d.ts +408 -0
- package/dist/schema-Cef2cFmb.d.ts.map +1 -0
- package/dist/semantic-concept-judge-B6cWNJ2K.js +725 -0
- package/dist/semantic-concept-judge-B6cWNJ2K.js.map +1 -0
- package/dist/sequential-Br0mAPHA.js +148 -0
- package/dist/sequential-Br0mAPHA.js.map +1 -0
- package/dist/sequential-CYwq6Ff_.d.ts +141 -0
- package/dist/sequential-CYwq6Ff_.d.ts.map +1 -0
- package/dist/series-convergence-CjO2QdRW.js +43 -0
- package/dist/series-convergence-CjO2QdRW.js.map +1 -0
- package/dist/series-convergence-ofsqPWhs.d.ts +35 -0
- package/dist/series-convergence-ofsqPWhs.d.ts.map +1 -0
- package/dist/server-m5D9cvnG.js +1040 -0
- package/dist/server-m5D9cvnG.js.map +1 -0
- package/dist/skill-usage-C_pXm7lP.d.ts +534 -0
- package/dist/skill-usage-C_pXm7lP.d.ts.map +1 -0
- package/dist/skillopt-optimization-method-B7wX7XkF.d.ts +1739 -0
- package/dist/skillopt-optimization-method-B7wX7XkF.d.ts.map +1 -0
- package/dist/skillopt-optimization-method-D4ODwFVV.js +7712 -0
- package/dist/skillopt-optimization-method-D4ODwFVV.js.map +1 -0
- package/dist/statistics-Cmj6nynr.d.ts +514 -0
- package/dist/statistics-Cmj6nynr.d.ts.map +1 -0
- package/dist/statistics-CnnxdpOg.js +1437 -0
- package/dist/statistics-CnnxdpOg.js.map +1 -0
- package/dist/store-CT9YIIve.d.ts +117 -0
- package/dist/store-CT9YIIve.d.ts.map +1 -0
- package/dist/store-CxJry_cs.d.ts +229 -0
- package/dist/store-CxJry_cs.d.ts.map +1 -0
- package/dist/storyboard/index.d.ts +84 -203
- package/dist/storyboard/index.d.ts.map +1 -0
- package/dist/storyboard/index.js +609 -542
- package/dist/storyboard/index.js.map +1 -1
- package/dist/summary-report-BNs5nmXI.js +862 -0
- package/dist/summary-report-BNs5nmXI.js.map +1 -0
- package/dist/summary-report-Cj9gdw4i.d.ts +407 -0
- package/dist/summary-report-Cj9gdw4i.d.ts.map +1 -0
- package/dist/supervisor-run/index.d.ts +2 -849
- package/dist/supervisor-run/index.js +2 -64
- package/dist/supervisor-run-_lnTLM3z.js +1679 -0
- package/dist/supervisor-run-_lnTLM3z.js.map +1 -0
- package/dist/task-failure-attributes-CQZlB3et.js +311 -0
- package/dist/task-failure-attributes-CQZlB3et.js.map +1 -0
- package/dist/test-graded-scenario-BsqWLmPt.js +318 -0
- package/dist/test-graded-scenario-BsqWLmPt.js.map +1 -0
- package/dist/test-graded-scenario-D1TaI2va.d.ts +141 -0
- package/dist/test-graded-scenario-D1TaI2va.d.ts.map +1 -0
- package/dist/tools-BmuN627J.js +1085 -0
- package/dist/tools-BmuN627J.js.map +1 -0
- package/dist/trace-attributes.d.ts +2 -52
- package/dist/trace-attributes.js +131 -61
- package/dist/trace-attributes.js.map +1 -1
- package/dist/traces.d.ts +12 -2365
- package/dist/traces.js +12 -251
- package/dist/trajectory-D_7rLrvE.js +56 -0
- package/dist/trajectory-D_7rLrvE.js.map +1 -0
- package/dist/types-DGsxbAEd.d.ts +387 -0
- package/dist/types-DGsxbAEd.d.ts.map +1 -0
- package/dist/types-k9tZGKUg.d.ts +640 -0
- package/dist/types-k9tZGKUg.d.ts.map +1 -0
- package/dist/verdict-Dps8_okt.d.ts +37 -0
- package/dist/verdict-Dps8_okt.d.ts.map +1 -0
- package/dist/wire/index.d.ts +702 -1174
- package/dist/wire/index.d.ts.map +1 -0
- package/dist/wire/index.js +2 -81
- package/docs/feature-guide.md +1 -1
- package/docs/rollout.md +116 -2
- package/package.json +18 -10
- package/dist/benchmarks/index.js.map +0 -1
- package/dist/campaign/index.js.map +0 -1
- package/dist/chunk-2JX3CFMB.js +0 -695
- package/dist/chunk-2JX3CFMB.js.map +0 -1
- package/dist/chunk-2MKQIFS4.js +0 -183
- package/dist/chunk-2MKQIFS4.js.map +0 -1
- package/dist/chunk-3RF76KTD.js +0 -84
- package/dist/chunk-3RF76KTD.js.map +0 -1
- package/dist/chunk-5DTSBUL2.js +0 -159
- package/dist/chunk-5DTSBUL2.js.map +0 -1
- package/dist/chunk-7ZZMD7UK.js +0 -386
- package/dist/chunk-7ZZMD7UK.js.map +0 -1
- package/dist/chunk-BOD4O7OF.js +0 -40
- package/dist/chunk-BOD4O7OF.js.map +0 -1
- package/dist/chunk-BYT7ELPS.js +0 -1553
- package/dist/chunk-BYT7ELPS.js.map +0 -1
- package/dist/chunk-DJKY2TSY.js +0 -2428
- package/dist/chunk-DJKY2TSY.js.map +0 -1
- package/dist/chunk-DPUHNQLN.js +0 -232
- package/dist/chunk-DPUHNQLN.js.map +0 -1
- package/dist/chunk-DRYIUNWY.js +0 -622
- package/dist/chunk-DRYIUNWY.js.map +0 -1
- package/dist/chunk-EJGRPCO3.js +0 -617
- package/dist/chunk-EJGRPCO3.js.map +0 -1
- package/dist/chunk-EOSZT7PL.js +0 -2001
- package/dist/chunk-EOSZT7PL.js.map +0 -1
- package/dist/chunk-EZJEIH2R.js +0 -1559
- package/dist/chunk-EZJEIH2R.js.map +0 -1
- package/dist/chunk-GGE4NNQT.js +0 -65
- package/dist/chunk-GGE4NNQT.js.map +0 -1
- package/dist/chunk-HHWE3POT.js +0 -94
- package/dist/chunk-HHWE3POT.js.map +0 -1
- package/dist/chunk-IHQDPH7D.js +0 -171
- package/dist/chunk-IHQDPH7D.js.map +0 -1
- package/dist/chunk-JHCHEVET.js +0 -274
- package/dist/chunk-JHCHEVET.js.map +0 -1
- package/dist/chunk-K4DBDHLK.js +0 -158
- package/dist/chunk-K4DBDHLK.js.map +0 -1
- package/dist/chunk-K6N6XJJX.js +0 -306
- package/dist/chunk-K6N6XJJX.js.map +0 -1
- package/dist/chunk-MA6HLL3S.js +0 -65
- package/dist/chunk-MA6HLL3S.js.map +0 -1
- package/dist/chunk-MAZ26DC7.js +0 -99
- package/dist/chunk-MAZ26DC7.js.map +0 -1
- package/dist/chunk-MHELPNRP.js +0 -1212
- package/dist/chunk-MHELPNRP.js.map +0 -1
- package/dist/chunk-NACAGYSY.js +0 -1040
- package/dist/chunk-NACAGYSY.js.map +0 -1
- package/dist/chunk-NKAGIDE2.js +0 -7633
- package/dist/chunk-NKAGIDE2.js.map +0 -1
- package/dist/chunk-NPCTHQIO.js +0 -91
- package/dist/chunk-NPCTHQIO.js.map +0 -1
- package/dist/chunk-NYLOYM6N.js +0 -332
- package/dist/chunk-NYLOYM6N.js.map +0 -1
- package/dist/chunk-ONWEPEDO.js +0 -57
- package/dist/chunk-ONWEPEDO.js.map +0 -1
- package/dist/chunk-P5W7RQKK.js +0 -576
- package/dist/chunk-P5W7RQKK.js.map +0 -1
- package/dist/chunk-P6FYH6K4.js +0 -1161
- package/dist/chunk-P6FYH6K4.js.map +0 -1
- package/dist/chunk-PBE2LOSS.js +0 -669
- package/dist/chunk-PBE2LOSS.js.map +0 -1
- package/dist/chunk-PC4UYEBM.js +0 -166
- package/dist/chunk-PC4UYEBM.js.map +0 -1
- package/dist/chunk-PXE2VKMX.js +0 -140
- package/dist/chunk-PXE2VKMX.js.map +0 -1
- package/dist/chunk-PZ5AY32C.js +0 -10
- package/dist/chunk-PZ5AY32C.js.map +0 -1
- package/dist/chunk-RZTMDUO7.js +0 -49
- package/dist/chunk-RZTMDUO7.js.map +0 -1
- package/dist/chunk-S5YLIBFX.js +0 -136
- package/dist/chunk-S5YLIBFX.js.map +0 -1
- package/dist/chunk-SZLVEKMJ.js +0 -1446
- package/dist/chunk-SZLVEKMJ.js.map +0 -1
- package/dist/chunk-T4SQEITX.js +0 -95
- package/dist/chunk-T4SQEITX.js.map +0 -1
- package/dist/chunk-TBL77AUT.js +0 -355
- package/dist/chunk-TBL77AUT.js.map +0 -1
- package/dist/chunk-TSN7JT6D.js +0 -1646
- package/dist/chunk-TSN7JT6D.js.map +0 -1
- package/dist/chunk-TT4KNT67.js +0 -124
- package/dist/chunk-TT4KNT67.js.map +0 -1
- package/dist/chunk-UB2LOJ6Q.js +0 -4461
- package/dist/chunk-UB2LOJ6Q.js.map +0 -1
- package/dist/chunk-UWZZKKU7.js +0 -237
- package/dist/chunk-UWZZKKU7.js.map +0 -1
- package/dist/chunk-VBQ3CRKH.js +0 -291
- package/dist/chunk-VBQ3CRKH.js.map +0 -1
- package/dist/chunk-VGRCHJON.js +0 -163
- package/dist/chunk-VGRCHJON.js.map +0 -1
- package/dist/chunk-VI2UW6B6.js +0 -162
- package/dist/chunk-VI2UW6B6.js.map +0 -1
- package/dist/chunk-VLOATJQ2.js +0 -908
- package/dist/chunk-VLOATJQ2.js.map +0 -1
- package/dist/chunk-VQMK5FMP.js +0 -247
- package/dist/chunk-VQMK5FMP.js.map +0 -1
- package/dist/chunk-VZSRQ272.js +0 -149
- package/dist/chunk-VZSRQ272.js.map +0 -1
- package/dist/chunk-WGXIEX7P.js +0 -116
- package/dist/chunk-WGXIEX7P.js.map +0 -1
- package/dist/chunk-WS3NZZQQ.js +0 -929
- package/dist/chunk-WS3NZZQQ.js.map +0 -1
- package/dist/chunk-XDWDC2MP.js +0 -695
- package/dist/chunk-XDWDC2MP.js.map +0 -1
- package/dist/chunk-XPRT64IE.js +0 -766
- package/dist/chunk-XPRT64IE.js.map +0 -1
- package/dist/chunk-YJBNWCAA.js +0 -1056
- package/dist/chunk-YJBNWCAA.js.map +0 -1
- package/dist/chunk-ZET2UAYW.js +0 -89
- package/dist/chunk-ZET2UAYW.js.map +0 -1
- package/dist/chunk-ZUUWPZCV.js +0 -752
- package/dist/chunk-ZUUWPZCV.js.map +0 -1
- package/dist/control.js.map +0 -1
- package/dist/hosted/index.js.map +0 -1
- package/dist/matrix/index.js.map +0 -1
- package/dist/reporting.js.map +0 -1
- package/dist/rollout/index.js.map +0 -1
- package/dist/run-campaign-ISHFZ7FJ.js +0 -17
- package/dist/run-campaign-ISHFZ7FJ.js.map +0 -1
- package/dist/supervisor-run/index.js.map +0 -1
- package/dist/traces.js.map +0 -1
- package/dist/wire/index.js.map +0 -1
|
@@ -0,0 +1,1067 @@
|
|
|
1
|
+
import { s as validateRunRecord } from "./run-record-BuoE80Dq.js";
|
|
2
|
+
import { A as spearmanR, C as pairedTTest, E as pearsonR, L as continuousAgreement, O as requiredPairedSampleSize, b as pairedMde, v as pairedBootstrap, y as pairedCohensDz } from "./statistics-CnnxdpOg.js";
|
|
3
|
+
import { r as pairRunRecords } from "./paired-arms-D9D0wXj2.js";
|
|
4
|
+
import { r as observedSplitScore } from "./reward-nw2xZGZG.js";
|
|
5
|
+
import { o as llmSpans } from "./query-Di7eEQ79.js";
|
|
6
|
+
import { r as paretoChart } from "./summary-report-BNs5nmXI.js";
|
|
7
|
+
//#region src/contamination-guard.ts
|
|
8
|
+
function checkCanaries(output, scenarios) {
|
|
9
|
+
const leaks = [];
|
|
10
|
+
for (const s of scenarios) {
|
|
11
|
+
if (!s.canary) continue;
|
|
12
|
+
if (output.includes(s.canary)) leaks.push({
|
|
13
|
+
scenarioId: s.id,
|
|
14
|
+
canary: s.canary,
|
|
15
|
+
evidence: excerpt(output, s.canary)
|
|
16
|
+
});
|
|
17
|
+
}
|
|
18
|
+
return leaks;
|
|
19
|
+
}
|
|
20
|
+
/**
|
|
21
|
+
* Behavioral canary: tests the AGENT, not the eval grep.
|
|
22
|
+
*
|
|
23
|
+
* - `forbiddenPattern` PRESENT in output ⇒ AGENT EMITTED BAD CONTENT ⇒ LEAK.
|
|
24
|
+
* - `forbiddenPattern` ABSENT from output ⇒ AGENT HELD THE LINE ⇒ PASS.
|
|
25
|
+
*
|
|
26
|
+
* Use when running known-bad-prompt scenarios against the agent under
|
|
27
|
+
* test and you want to know if the agent misbehaved. The classical
|
|
28
|
+
* {@link checkCanaries} / {@link import('./canary').runCanaries | runCanaries}
|
|
29
|
+
* test whether the eval check fires when the bad output is forced
|
|
30
|
+
* into the eval flow — different question, different answer.
|
|
31
|
+
*
|
|
32
|
+
* Pattern resolution order (first match wins):
|
|
33
|
+
* 1. `scenario.forbiddenPattern` — if it parses as `/body/flags`,
|
|
34
|
+
* treated as a regex; otherwise a literal substring.
|
|
35
|
+
* 2. `scenario.canary` — literal substring fallback so the helper
|
|
36
|
+
* works on existing scenario fixtures.
|
|
37
|
+
*
|
|
38
|
+
* Returns `null` when nothing forbidden was found OR the scenario
|
|
39
|
+
* declared no pattern.
|
|
40
|
+
*/
|
|
41
|
+
function checkBehavioralCanary(output, scenario) {
|
|
42
|
+
const pattern = scenario.forbiddenPattern ?? scenario.canary;
|
|
43
|
+
if (!pattern) return null;
|
|
44
|
+
const hit = matchForbidden(output, pattern);
|
|
45
|
+
if (!hit) return null;
|
|
46
|
+
return {
|
|
47
|
+
scenarioId: scenario.id,
|
|
48
|
+
canary: pattern,
|
|
49
|
+
evidence: excerpt(output, hit)
|
|
50
|
+
};
|
|
51
|
+
}
|
|
52
|
+
/**
|
|
53
|
+
* Behavioral canary over many (scenario, output) pairs. Sibling to
|
|
54
|
+
* {@link import('./canary').runCanaries | runCanaries} — same idea
|
|
55
|
+
* (run-many → report) but the question being answered is "did the
|
|
56
|
+
* AGENT misbehave?" rather than "did the EVAL grep fire?".
|
|
57
|
+
*
|
|
58
|
+
* Returns one `CanaryLeak` per pair where the agent's output
|
|
59
|
+
* contained its scenario's `forbiddenPattern` (or `canary` fallback).
|
|
60
|
+
*/
|
|
61
|
+
function runBehavioralCanaries(cases) {
|
|
62
|
+
const leaks = [];
|
|
63
|
+
for (const c of cases) {
|
|
64
|
+
const leak = checkBehavioralCanary(c.output, c.scenario);
|
|
65
|
+
if (leak) leaks.push({
|
|
66
|
+
...leak,
|
|
67
|
+
runId: c.runId ?? leak.runId
|
|
68
|
+
});
|
|
69
|
+
}
|
|
70
|
+
return leaks;
|
|
71
|
+
}
|
|
72
|
+
/**
|
|
73
|
+
* Resolve a forbidden-pattern string to the matched substring inside
|
|
74
|
+
* `output`. `/body/flags` notation is interpreted as a regex; anything
|
|
75
|
+
* else is a literal substring.
|
|
76
|
+
*/
|
|
77
|
+
function matchForbidden(output, pattern) {
|
|
78
|
+
const re = tryParseRegex(pattern);
|
|
79
|
+
if (re) {
|
|
80
|
+
const m = output.match(re);
|
|
81
|
+
return m && m[0].length > 0 ? m[0] : null;
|
|
82
|
+
}
|
|
83
|
+
return output.includes(pattern) ? pattern : null;
|
|
84
|
+
}
|
|
85
|
+
function tryParseRegex(pattern) {
|
|
86
|
+
if (pattern.length < 2 || pattern[0] !== "/") return null;
|
|
87
|
+
const last = pattern.lastIndexOf("/");
|
|
88
|
+
if (last <= 0) return null;
|
|
89
|
+
const body = pattern.slice(1, last);
|
|
90
|
+
const flags = pattern.slice(last + 1);
|
|
91
|
+
if (!/^[gimsuy]*$/.test(flags)) return null;
|
|
92
|
+
try {
|
|
93
|
+
return new RegExp(body, flags);
|
|
94
|
+
} catch {
|
|
95
|
+
return null;
|
|
96
|
+
}
|
|
97
|
+
}
|
|
98
|
+
/**
|
|
99
|
+
* Scan the LLM-output history in a corpus; returns every case where a
|
|
100
|
+
* canary from a known scenario appeared in agent output. Pass the full
|
|
101
|
+
* set of scenarios whose canaries you care about (typically the whole
|
|
102
|
+
* held-out slice).
|
|
103
|
+
*/
|
|
104
|
+
async function canaryLeakView(store, scenarios) {
|
|
105
|
+
const targets = scenarios.filter((s) => !!s.canary);
|
|
106
|
+
if (targets.length === 0) return [];
|
|
107
|
+
const spans = await llmSpans(store);
|
|
108
|
+
const leaks = [];
|
|
109
|
+
for (const span of spans) {
|
|
110
|
+
const output = span.output ?? "";
|
|
111
|
+
for (const s of targets) if (s.canary && output.includes(s.canary)) leaks.push({
|
|
112
|
+
scenarioId: s.id,
|
|
113
|
+
canary: s.canary,
|
|
114
|
+
runId: span.runId,
|
|
115
|
+
evidence: excerpt(output, s.canary)
|
|
116
|
+
});
|
|
117
|
+
}
|
|
118
|
+
return leaks;
|
|
119
|
+
}
|
|
120
|
+
var HoldoutAuditor = class {
|
|
121
|
+
scenarios;
|
|
122
|
+
accessLog = [];
|
|
123
|
+
constructor(scenarios) {
|
|
124
|
+
this.scenarios = scenarios;
|
|
125
|
+
}
|
|
126
|
+
/** Retrieve a holdout scenario for a declared purpose. Non-'evaluation' throws. */
|
|
127
|
+
get(scenarioId, purpose) {
|
|
128
|
+
if (purpose !== "evaluation" && purpose !== "debugging") throw new Error(`HoldoutAuditor.get: purpose must be 'evaluation' or 'debugging', got ${purpose}`);
|
|
129
|
+
const s = this.scenarios.find((x) => x.id === scenarioId);
|
|
130
|
+
if (!s) throw new Error(`holdout scenario "${scenarioId}" not found`);
|
|
131
|
+
this.accessLog.push({
|
|
132
|
+
scenarioId,
|
|
133
|
+
purpose,
|
|
134
|
+
at: Date.now()
|
|
135
|
+
});
|
|
136
|
+
return s;
|
|
137
|
+
}
|
|
138
|
+
getAccessLog() {
|
|
139
|
+
return this.accessLog;
|
|
140
|
+
}
|
|
141
|
+
};
|
|
142
|
+
function excerpt(source, needle) {
|
|
143
|
+
const at = source.indexOf(needle);
|
|
144
|
+
if (at < 0) return "";
|
|
145
|
+
const start = Math.max(0, at - 30);
|
|
146
|
+
const end = Math.min(source.length, at + needle.length + 30);
|
|
147
|
+
return (start > 0 ? "…" : "") + source.slice(start, end) + (end < source.length ? "…" : "");
|
|
148
|
+
}
|
|
149
|
+
//#endregion
|
|
150
|
+
//#region src/contract/analyze-runs.ts
|
|
151
|
+
/** Summarize runtime facts without interpreting task quality or promotion readiness. */
|
|
152
|
+
function summarizeExecution(opts) {
|
|
153
|
+
const runs = opts.runs.map(validateRunRecord);
|
|
154
|
+
return {
|
|
155
|
+
execution: computeExecutionInsight(runs, opts.histogramBins ?? 12),
|
|
156
|
+
costProvenance: summarizeCostProvenance(runs)
|
|
157
|
+
};
|
|
158
|
+
}
|
|
159
|
+
async function analyzeRuns(opts) {
|
|
160
|
+
const runs = opts.runs.map(validateRunRecord);
|
|
161
|
+
const bins = opts.histogramBins ?? 12;
|
|
162
|
+
const threshold = opts.decisionThreshold ?? .02;
|
|
163
|
+
const split = resolveSplit(runs, opts.split ?? "auto");
|
|
164
|
+
const compositeWithIds = runs.map((r) => ({
|
|
165
|
+
runId: r.runId,
|
|
166
|
+
score: compositeOf(r, split)
|
|
167
|
+
})).filter((p) => Number.isFinite(p.score));
|
|
168
|
+
const composite = distributionOf(compositeWithIds.map((p) => p.score), bins, compositeWithIds);
|
|
169
|
+
const perDimension = computePerDimension(runs, bins);
|
|
170
|
+
const { execution, costProvenance: provenance } = summarizeExecution({
|
|
171
|
+
runs,
|
|
172
|
+
histogramBins: bins
|
|
173
|
+
});
|
|
174
|
+
const knownCostRuns = runs.filter((run) => run.costProvenance.kind !== "uncaptured");
|
|
175
|
+
const costs = knownCostRuns.map((r) => r.costUsd).filter(isFiniteNumber);
|
|
176
|
+
const costDist = distributionOf(costs, bins);
|
|
177
|
+
const pareto = paretoChart(knownCostRuns, { split });
|
|
178
|
+
const degraded = {};
|
|
179
|
+
if (provenance.uncaptured.n > 0) degraded.cost = diagnoseCostCoverage(runs, provenance);
|
|
180
|
+
else if (costs.length === 0 || costs.every((c) => c === 0)) degraded.cost = `all ${runs.length} explicitly observed or estimated USD values are $0`;
|
|
181
|
+
if (pareto.points.length < 2) degraded.pareto = pareto.points.length === 0 ? "no candidates — Pareto unavailable" : "single candidate — Pareto is a single point, not a frontier";
|
|
182
|
+
const costQuality = {
|
|
183
|
+
cost: costDist,
|
|
184
|
+
pareto,
|
|
185
|
+
provenance,
|
|
186
|
+
...degraded.cost || degraded.pareto ? { degraded } : {}
|
|
187
|
+
};
|
|
188
|
+
const judges = computeJudgeInsights(runs);
|
|
189
|
+
const interRater = opts.raterScores ? computeInterRater(opts.raterScores) : void 0;
|
|
190
|
+
const lift = computeLift(runs, opts.baselineCandidateId, opts.candidateCandidateId, split);
|
|
191
|
+
const failureClusters = opts.analyst ? await computeFailureClusters(runs, opts.analyst, split) : void 0;
|
|
192
|
+
const failureClasses = computeFailureClasses(runs, split);
|
|
193
|
+
const contamination = opts.canaryScenarios ? computeContamination(runs, opts.canaryScenarios) : void 0;
|
|
194
|
+
const outcomeCorrelation = opts.outcomeSignal ? computeOutcomeCorrelation(runs, opts.outcomeSignal, split) : void 0;
|
|
195
|
+
const release = buildReleaseScorecard(composite, lift, contamination);
|
|
196
|
+
const priorPeriodComparison = opts.baselineRuns ? computePriorPeriodComparison(runs, opts.baselineRuns, split, opts.baselineLabel) : void 0;
|
|
197
|
+
const recommendations = buildRecommendations({
|
|
198
|
+
composite,
|
|
199
|
+
judges,
|
|
200
|
+
interRater,
|
|
201
|
+
lift,
|
|
202
|
+
failureClusters,
|
|
203
|
+
failureClasses,
|
|
204
|
+
contamination,
|
|
205
|
+
outcomeCorrelation,
|
|
206
|
+
priorPeriodComparison,
|
|
207
|
+
threshold
|
|
208
|
+
});
|
|
209
|
+
return {
|
|
210
|
+
n: runs.length,
|
|
211
|
+
execution,
|
|
212
|
+
composite,
|
|
213
|
+
perDimension,
|
|
214
|
+
costQuality,
|
|
215
|
+
judges,
|
|
216
|
+
interRater,
|
|
217
|
+
lift,
|
|
218
|
+
failureClusters,
|
|
219
|
+
contamination,
|
|
220
|
+
outcomeCorrelation,
|
|
221
|
+
release,
|
|
222
|
+
...failureClasses ? { failureClasses } : {},
|
|
223
|
+
...priorPeriodComparison ? { priorPeriodComparison } : {},
|
|
224
|
+
recommendations
|
|
225
|
+
};
|
|
226
|
+
}
|
|
227
|
+
function computeExecutionInsight(runs, bins) {
|
|
228
|
+
const aggregateRows = runs.flatMap((run) => {
|
|
229
|
+
const usage = aggregateTokenUsage(run);
|
|
230
|
+
return usage ? [{
|
|
231
|
+
usage,
|
|
232
|
+
costUsd: finiteRaw(run, "aggregate_cost_usd")
|
|
233
|
+
}] : [];
|
|
234
|
+
});
|
|
235
|
+
const aggregateCosts = aggregateRows.flatMap((row) => row.costUsd !== void 0 ? [row.costUsd] : []);
|
|
236
|
+
const modelCounts = /* @__PURE__ */ new Map();
|
|
237
|
+
let executionErrorRuns = 0;
|
|
238
|
+
let executionErrorEvents = 0;
|
|
239
|
+
let errorReportingRuns = 0;
|
|
240
|
+
let errorSpanEvents = 0;
|
|
241
|
+
let errorSpanReportingRuns = 0;
|
|
242
|
+
const terminalOutcomes = {
|
|
243
|
+
succeeded: 0,
|
|
244
|
+
failed: 0,
|
|
245
|
+
cancelled: 0,
|
|
246
|
+
incomplete: 0,
|
|
247
|
+
unknown: 0
|
|
248
|
+
};
|
|
249
|
+
const errorsByTerminalOutcome = {
|
|
250
|
+
succeeded: {
|
|
251
|
+
withErrors: 0,
|
|
252
|
+
withoutErrors: 0,
|
|
253
|
+
unreported: 0
|
|
254
|
+
},
|
|
255
|
+
failed: {
|
|
256
|
+
withErrors: 0,
|
|
257
|
+
withoutErrors: 0,
|
|
258
|
+
unreported: 0
|
|
259
|
+
},
|
|
260
|
+
cancelled: {
|
|
261
|
+
withErrors: 0,
|
|
262
|
+
withoutErrors: 0,
|
|
263
|
+
unreported: 0
|
|
264
|
+
},
|
|
265
|
+
incomplete: {
|
|
266
|
+
withErrors: 0,
|
|
267
|
+
withoutErrors: 0,
|
|
268
|
+
unreported: 0
|
|
269
|
+
},
|
|
270
|
+
unknown: {
|
|
271
|
+
withErrors: 0,
|
|
272
|
+
withoutErrors: 0,
|
|
273
|
+
unreported: 0
|
|
274
|
+
}
|
|
275
|
+
};
|
|
276
|
+
let modelCallRuns = 0;
|
|
277
|
+
let modelCallEvents = 0;
|
|
278
|
+
let modelCallReportingRuns = 0;
|
|
279
|
+
for (const run of runs) {
|
|
280
|
+
modelCounts.set(run.model, (modelCounts.get(run.model) ?? 0) + 1);
|
|
281
|
+
const terminalOutcome = run.terminalOutcome;
|
|
282
|
+
terminalOutcomes[terminalOutcome] += 1;
|
|
283
|
+
const modelCalls = nonNegativeCountRaw(run, "llm_span_count");
|
|
284
|
+
if (modelCalls !== void 0) {
|
|
285
|
+
modelCallEvents += modelCalls;
|
|
286
|
+
modelCallReportingRuns += 1;
|
|
287
|
+
}
|
|
288
|
+
const usage = run.tokenUsage;
|
|
289
|
+
if ((modelCalls ?? 0) > 0 || usage.input > 0 || usage.output > 0 || (usage.cached ?? 0) > 0 || (usage.cacheWrite ?? 0) > 0) modelCallRuns += 1;
|
|
290
|
+
const errorEvents = reportedExecutionErrorEvents(run);
|
|
291
|
+
if (errorEvents !== void 0) {
|
|
292
|
+
executionErrorEvents += errorEvents;
|
|
293
|
+
errorReportingRuns += 1;
|
|
294
|
+
if (errorEvents > 0) {
|
|
295
|
+
executionErrorRuns += 1;
|
|
296
|
+
errorsByTerminalOutcome[terminalOutcome].withErrors += 1;
|
|
297
|
+
} else errorsByTerminalOutcome[terminalOutcome].withoutErrors += 1;
|
|
298
|
+
} else errorsByTerminalOutcome[terminalOutcome].unreported += 1;
|
|
299
|
+
const reportedErrorSpans = nonNegativeCountRaw(run, "error_span_count");
|
|
300
|
+
if (reportedErrorSpans !== void 0) {
|
|
301
|
+
errorSpanEvents += reportedErrorSpans;
|
|
302
|
+
errorSpanReportingRuns += 1;
|
|
303
|
+
}
|
|
304
|
+
}
|
|
305
|
+
return {
|
|
306
|
+
durationMs: distributionOf(runs.map((run) => run.wallMs), bins),
|
|
307
|
+
queueMs: distributionOf(runs.filter((run) => run.queueMs !== void 0).map((run) => run.queueMs), bins),
|
|
308
|
+
tokenUsage: summarizeTokenUsage(runs.map((run) => run.tokenUsage), bins),
|
|
309
|
+
aggregateUsage: {
|
|
310
|
+
runs: aggregateRows.length,
|
|
311
|
+
tokenUsage: summarizeTokenUsage(aggregateRows.map((row) => row.usage), bins),
|
|
312
|
+
costUsd: distributionOf(aggregateCosts, bins),
|
|
313
|
+
totalCostUsd: aggregateCosts.reduce((total, value) => total + value, 0)
|
|
314
|
+
},
|
|
315
|
+
models: [...modelCounts.entries()].map(([model, count]) => ({
|
|
316
|
+
model,
|
|
317
|
+
runs: count
|
|
318
|
+
})).sort((left, right) => right.runs - left.runs || left.model.localeCompare(right.model)),
|
|
319
|
+
modelCalls: {
|
|
320
|
+
runs: modelCallRuns,
|
|
321
|
+
events: modelCallEvents,
|
|
322
|
+
reportingRuns: modelCallReportingRuns
|
|
323
|
+
},
|
|
324
|
+
executionErrors: {
|
|
325
|
+
runs: executionErrorRuns,
|
|
326
|
+
fraction: errorReportingRuns > 0 ? executionErrorRuns / errorReportingRuns : null,
|
|
327
|
+
events: executionErrorEvents,
|
|
328
|
+
reportingRuns: errorReportingRuns,
|
|
329
|
+
errorSpanEvents,
|
|
330
|
+
errorSpanReportingRuns,
|
|
331
|
+
byTerminalOutcome: errorsByTerminalOutcome
|
|
332
|
+
},
|
|
333
|
+
terminalOutcomes
|
|
334
|
+
};
|
|
335
|
+
}
|
|
336
|
+
function reportedExecutionErrorEvents(run) {
|
|
337
|
+
return nonNegativeCountRaw(run, "execution_error_count");
|
|
338
|
+
}
|
|
339
|
+
function nonNegativeCountRaw(run, key) {
|
|
340
|
+
const value = finiteRaw(run, key);
|
|
341
|
+
return value !== void 0 && Number.isInteger(value) && value >= 0 ? value : void 0;
|
|
342
|
+
}
|
|
343
|
+
function summarizeTokenUsage(usages, bins) {
|
|
344
|
+
const reasoning = usages.flatMap((usage) => usage.reasoning !== void 0 ? [usage.reasoning] : []);
|
|
345
|
+
const cached = usages.flatMap((usage) => usage.cached !== void 0 ? [usage.cached] : []);
|
|
346
|
+
const cacheWrite = usages.flatMap((usage) => usage.cacheWrite !== void 0 ? [usage.cacheWrite] : []);
|
|
347
|
+
return {
|
|
348
|
+
input: distributionOf(usages.map((usage) => usage.input), bins),
|
|
349
|
+
output: distributionOf(usages.map((usage) => usage.output), bins),
|
|
350
|
+
reasoning: distributionOf(reasoning, bins),
|
|
351
|
+
cached: distributionOf(cached, bins),
|
|
352
|
+
cacheWrite: distributionOf(cacheWrite, bins),
|
|
353
|
+
totals: {
|
|
354
|
+
input: usages.reduce((total, usage) => total + usage.input, 0),
|
|
355
|
+
output: usages.reduce((total, usage) => total + usage.output, 0),
|
|
356
|
+
reasoning: reasoning.reduce((total, value) => total + value, 0),
|
|
357
|
+
cached: cached.reduce((total, value) => total + value, 0),
|
|
358
|
+
cacheWrite: cacheWrite.reduce((total, value) => total + value, 0)
|
|
359
|
+
}
|
|
360
|
+
};
|
|
361
|
+
}
|
|
362
|
+
function aggregateTokenUsage(run) {
|
|
363
|
+
const input = finiteRaw(run, "aggregate_prompt_tokens");
|
|
364
|
+
const output = finiteRaw(run, "aggregate_completion_tokens");
|
|
365
|
+
const reasoning = finiteRaw(run, "aggregate_reasoning_tokens");
|
|
366
|
+
const cached = finiteRaw(run, "aggregate_cached_tokens");
|
|
367
|
+
const cacheWrite = finiteRaw(run, "aggregate_cache_write_tokens");
|
|
368
|
+
if (input === void 0 && output === void 0 && reasoning === void 0 && cached === void 0 && cacheWrite === void 0) return void 0;
|
|
369
|
+
return {
|
|
370
|
+
input: input ?? 0,
|
|
371
|
+
output: output ?? 0,
|
|
372
|
+
...reasoning !== void 0 ? { reasoning } : {},
|
|
373
|
+
...cached !== void 0 ? { cached } : {},
|
|
374
|
+
...cacheWrite !== void 0 ? { cacheWrite } : {}
|
|
375
|
+
};
|
|
376
|
+
}
|
|
377
|
+
function finiteRaw(run, key) {
|
|
378
|
+
const value = run.outcome.raw[key];
|
|
379
|
+
return typeof value === "number" && Number.isFinite(value) && value >= 0 ? value : void 0;
|
|
380
|
+
}
|
|
381
|
+
function summarizeCostProvenance(runs) {
|
|
382
|
+
const summary = {
|
|
383
|
+
observed: {
|
|
384
|
+
n: 0,
|
|
385
|
+
totalUsd: 0
|
|
386
|
+
},
|
|
387
|
+
estimated: {
|
|
388
|
+
n: 0,
|
|
389
|
+
totalUsd: 0
|
|
390
|
+
},
|
|
391
|
+
uncaptured: { n: 0 },
|
|
392
|
+
knownFraction: 0
|
|
393
|
+
};
|
|
394
|
+
for (const run of runs) {
|
|
395
|
+
const cost = run.costProvenance;
|
|
396
|
+
if (cost.kind === "uncaptured") summary.uncaptured.n += 1;
|
|
397
|
+
else {
|
|
398
|
+
summary[cost.kind].n += 1;
|
|
399
|
+
summary[cost.kind].totalUsd += cost.usd;
|
|
400
|
+
}
|
|
401
|
+
}
|
|
402
|
+
const known = summary.observed.n + summary.estimated.n;
|
|
403
|
+
summary.knownFraction = runs.length > 0 ? known / runs.length : 0;
|
|
404
|
+
return summary;
|
|
405
|
+
}
|
|
406
|
+
function diagnoseCostCoverage(runs, provenance) {
|
|
407
|
+
const uncaptured = provenance.uncaptured.n;
|
|
408
|
+
const known = provenance.observed.n + provenance.estimated.n;
|
|
409
|
+
if (uncaptured === runs.length) return `USD cost uncaptured for all ${runs.length} runs — no observed or estimated USD values; token and wall-time metrics remain available.`;
|
|
410
|
+
return `USD cost uncaptured for ${uncaptured}/${runs.length} runs; excluded those rows from cost statistics (${known}/${runs.length} retained: ${provenance.observed.n} observed, ${provenance.estimated.n} estimated).`;
|
|
411
|
+
}
|
|
412
|
+
/**
|
|
413
|
+
* Model-free task-failure tally.
|
|
414
|
+
*
|
|
415
|
+
* Explicit non-success classes are task-failure evidence.
|
|
416
|
+
* A low task score without a class is counted as `unknown`.
|
|
417
|
+
*/
|
|
418
|
+
function computeFailureClasses(runs, split) {
|
|
419
|
+
const counts = /* @__PURE__ */ new Map();
|
|
420
|
+
for (const r of runs) {
|
|
421
|
+
if (!isTaskFailure(r, split)) continue;
|
|
422
|
+
const key = r.failureClass !== void 0 && r.failureClass !== "success" ? r.failureClass : "unknown";
|
|
423
|
+
counts.set(key, (counts.get(key) ?? 0) + 1);
|
|
424
|
+
}
|
|
425
|
+
if (counts.size === 0) return void 0;
|
|
426
|
+
const n = runs.length;
|
|
427
|
+
return [...counts.entries()].map(([failureClass, count]) => ({
|
|
428
|
+
failureClass,
|
|
429
|
+
count,
|
|
430
|
+
share: n > 0 ? count / n : 0
|
|
431
|
+
})).sort((a, b) => b.count - a.count || a.failureClass.localeCompare(b.failureClass));
|
|
432
|
+
}
|
|
433
|
+
function computePriorPeriodComparison(current, baseline, split, windowLabel) {
|
|
434
|
+
if (current.length === 0 || baseline.length === 0) return void 0;
|
|
435
|
+
const metrics = {};
|
|
436
|
+
const directions = {};
|
|
437
|
+
const compositeCurrent = current.map((r) => compositeOf(r, split)).filter(Number.isFinite);
|
|
438
|
+
const compositeBaseline = baseline.map((r) => compositeOf(r, split)).filter(Number.isFinite);
|
|
439
|
+
if (compositeCurrent.length > 0 && compositeBaseline.length > 0) {
|
|
440
|
+
metrics.composite = welchCompare(compositeBaseline, compositeCurrent);
|
|
441
|
+
directions.composite = "higher-is-better";
|
|
442
|
+
}
|
|
443
|
+
const costCurrent = knownCostValues(current);
|
|
444
|
+
const costBaseline = knownCostValues(baseline);
|
|
445
|
+
if (costCurrent.length > 0 && costBaseline.length > 0) {
|
|
446
|
+
metrics.cost = welchCompare(costBaseline, costCurrent);
|
|
447
|
+
directions.cost = "lower-is-better";
|
|
448
|
+
}
|
|
449
|
+
const durCurrent = current.map((r) => r.wallMs).filter(Number.isFinite);
|
|
450
|
+
const durBaseline = baseline.map((r) => r.wallMs).filter(Number.isFinite);
|
|
451
|
+
if (durCurrent.length > 0 && durBaseline.length > 0) {
|
|
452
|
+
metrics.duration = welchCompare(durBaseline, durCurrent);
|
|
453
|
+
directions.duration = "lower-is-better";
|
|
454
|
+
}
|
|
455
|
+
const tokCurrent = current.map((r) => (r.tokenUsage.input ?? 0) + (r.tokenUsage.output ?? 0)).filter(Number.isFinite);
|
|
456
|
+
const tokBaseline = baseline.map((r) => (r.tokenUsage.input ?? 0) + (r.tokenUsage.output ?? 0)).filter(Number.isFinite);
|
|
457
|
+
if (tokCurrent.length > 0 && tokBaseline.length > 0) {
|
|
458
|
+
metrics.tokenUsage = welchCompare(tokBaseline, tokCurrent);
|
|
459
|
+
directions.tokenUsage = "lower-is-better";
|
|
460
|
+
}
|
|
461
|
+
const dimsCurrent = collectPerDimension(current);
|
|
462
|
+
const dimsBaseline = collectPerDimension(baseline);
|
|
463
|
+
for (const dim of Object.keys(dimsCurrent)) {
|
|
464
|
+
const b = dimsBaseline[dim];
|
|
465
|
+
const c = dimsCurrent[dim];
|
|
466
|
+
if (!b || b.length === 0 || !c || c.length === 0) continue;
|
|
467
|
+
metrics[`dim.${dim}`] = welchCompare(b, c);
|
|
468
|
+
directions[`dim.${dim}`] = "higher-is-better";
|
|
469
|
+
}
|
|
470
|
+
const regressedMetrics = [];
|
|
471
|
+
const improvedMetrics = [];
|
|
472
|
+
for (const [name, delta] of Object.entries(metrics)) {
|
|
473
|
+
if (!delta.significant) continue;
|
|
474
|
+
if ((directions[name] ?? "higher-is-better") === "higher-is-better" ? delta.delta > 0 : delta.delta < 0) improvedMetrics.push(name);
|
|
475
|
+
else regressedMetrics.push(name);
|
|
476
|
+
}
|
|
477
|
+
return {
|
|
478
|
+
baselineN: baseline.length,
|
|
479
|
+
currentN: current.length,
|
|
480
|
+
...windowLabel ? { windowLabel } : {},
|
|
481
|
+
metrics,
|
|
482
|
+
regressedMetrics,
|
|
483
|
+
improvedMetrics
|
|
484
|
+
};
|
|
485
|
+
}
|
|
486
|
+
function knownCostValues(runs) {
|
|
487
|
+
return runs.filter((run) => run.costProvenance.kind !== "uncaptured").map((run) => run.costUsd).filter(isFiniteNumber);
|
|
488
|
+
}
|
|
489
|
+
function isFiniteNumber(value) {
|
|
490
|
+
return typeof value === "number" && Number.isFinite(value);
|
|
491
|
+
}
|
|
492
|
+
/** Collect per-dimension values across runs (from outcome.judgeScores.perDimMean). */
|
|
493
|
+
function collectPerDimension(runs) {
|
|
494
|
+
const out = {};
|
|
495
|
+
for (const r of runs) {
|
|
496
|
+
const perDim = r.outcome.judgeScores?.perDimMean;
|
|
497
|
+
if (!perDim) continue;
|
|
498
|
+
for (const [dim, value] of Object.entries(perDim)) {
|
|
499
|
+
if (!Number.isFinite(value)) continue;
|
|
500
|
+
if (!out[dim]) out[dim] = [];
|
|
501
|
+
out[dim].push(value);
|
|
502
|
+
}
|
|
503
|
+
}
|
|
504
|
+
return out;
|
|
505
|
+
}
|
|
506
|
+
/** Two-sample Welch comparison: unequal-variance t-test + CI on the delta
|
|
507
|
+
* + Cohen's d (pooled stddev). Significance = p < 0.05 AND |d| >= 0.2. */
|
|
508
|
+
function welchCompare(baseline, current) {
|
|
509
|
+
const baselineMean = mean(baseline);
|
|
510
|
+
const currentMean = mean(current);
|
|
511
|
+
const baselineVar = sampleVariance(baseline, baselineMean);
|
|
512
|
+
const currentVar = sampleVariance(current, currentMean);
|
|
513
|
+
const baselineN = baseline.length;
|
|
514
|
+
const currentN = current.length;
|
|
515
|
+
const delta = currentMean - baselineMean;
|
|
516
|
+
const se = Math.sqrt(baselineVar / baselineN + currentVar / currentN);
|
|
517
|
+
const halfWidth = 1.96 * (se > 0 ? se : 0);
|
|
518
|
+
const ci95 = [delta - halfWidth, delta + halfWidth];
|
|
519
|
+
const t = se > 0 ? delta / se : 0;
|
|
520
|
+
const pValue = se > 0 ? 2 * (1 - standardNormalCdf(Math.abs(t))) : 1;
|
|
521
|
+
const pooledStddev = Math.sqrt(((baselineN - 1) * baselineVar + (currentN - 1) * currentVar) / Math.max(1, baselineN + currentN - 2));
|
|
522
|
+
const cohensD = pooledStddev > 0 ? delta / pooledStddev : 0;
|
|
523
|
+
return {
|
|
524
|
+
current: currentMean,
|
|
525
|
+
baseline: baselineMean,
|
|
526
|
+
delta,
|
|
527
|
+
ci95,
|
|
528
|
+
pValue,
|
|
529
|
+
cohensD,
|
|
530
|
+
baselineN,
|
|
531
|
+
currentN,
|
|
532
|
+
significant: pValue < .05 && Math.abs(cohensD) >= .2
|
|
533
|
+
};
|
|
534
|
+
}
|
|
535
|
+
function sampleVariance(xs, xsMean) {
|
|
536
|
+
if (xs.length < 2) return 0;
|
|
537
|
+
let s = 0;
|
|
538
|
+
for (const x of xs) s += (x - xsMean) ** 2;
|
|
539
|
+
return s / (xs.length - 1);
|
|
540
|
+
}
|
|
541
|
+
/** Abramowitz & Stegun approximation to Φ(z). Maximum error ~7.5e-8. */
|
|
542
|
+
function standardNormalCdf(z) {
|
|
543
|
+
const a1 = .254829592;
|
|
544
|
+
const a2 = -.284496736;
|
|
545
|
+
const a3 = 1.421413741;
|
|
546
|
+
const a4 = -1.453152027;
|
|
547
|
+
const a5 = 1.061405429;
|
|
548
|
+
const p = .3275911;
|
|
549
|
+
const sign = z < 0 ? -1 : 1;
|
|
550
|
+
const x = Math.abs(z) / Math.SQRT2;
|
|
551
|
+
const t = 1 / (1 + p * x);
|
|
552
|
+
return .5 * (1 + sign * (1 - ((((a5 * t + a4) * t + a3) * t + a2) * t + a1) * t * Math.exp(-x * x)));
|
|
553
|
+
}
|
|
554
|
+
function resolveSplit(runs, pref) {
|
|
555
|
+
if (pref !== "auto") return pref;
|
|
556
|
+
return runs.some((r) => Number.isFinite(observedSplitScore(r, "holdout"))) ? "holdout" : "search";
|
|
557
|
+
}
|
|
558
|
+
/**
|
|
559
|
+
* RAW (`observedSplitScore`): `analyzeRuns` describes what a set of runs
|
|
560
|
+
* reported, and every downstream reader of this composite — distributions,
|
|
561
|
+
* per-candidate summaries, the reward-hacking correlation — needs the ungated
|
|
562
|
+
* number to see an inflated run at all.
|
|
563
|
+
*/
|
|
564
|
+
function compositeOf(run, split) {
|
|
565
|
+
const score = observedSplitScore(run, split);
|
|
566
|
+
return Number.isFinite(score) ? score : NaN;
|
|
567
|
+
}
|
|
568
|
+
function distributionOf(values, bins, withIds) {
|
|
569
|
+
if (values.length === 0) return {
|
|
570
|
+
n: 0,
|
|
571
|
+
mean: null,
|
|
572
|
+
p50: null,
|
|
573
|
+
p95: null,
|
|
574
|
+
stddev: null,
|
|
575
|
+
min: null,
|
|
576
|
+
max: null,
|
|
577
|
+
histogram: []
|
|
578
|
+
};
|
|
579
|
+
const sorted = [...values].sort((a, b) => a - b);
|
|
580
|
+
const n = sorted.length;
|
|
581
|
+
const mean = sorted.reduce((s, v) => s + v, 0) / n;
|
|
582
|
+
const variance = sorted.reduce((s, v) => s + (v - mean) ** 2, 0) / n;
|
|
583
|
+
const stddev = Math.sqrt(variance);
|
|
584
|
+
const tailRuns = withIds ? [...withIds].sort((a, b) => a.score - b.score).slice(0, Math.min(5, withIds.length)) : void 0;
|
|
585
|
+
return {
|
|
586
|
+
n,
|
|
587
|
+
mean,
|
|
588
|
+
p50: percentile(sorted, .5),
|
|
589
|
+
p95: percentile(sorted, .95),
|
|
590
|
+
stddev,
|
|
591
|
+
min: sorted[0],
|
|
592
|
+
max: sorted[n - 1],
|
|
593
|
+
histogram: histogram(sorted, bins),
|
|
594
|
+
...tailRuns ? { tailRuns } : {}
|
|
595
|
+
};
|
|
596
|
+
}
|
|
597
|
+
function percentile(sorted, q) {
|
|
598
|
+
if (sorted.length === 0) return 0;
|
|
599
|
+
if (sorted.length === 1) return sorted[0];
|
|
600
|
+
const idx = (sorted.length - 1) * q;
|
|
601
|
+
const lo = Math.floor(idx);
|
|
602
|
+
const hi = Math.ceil(idx);
|
|
603
|
+
if (lo === hi) return sorted[lo];
|
|
604
|
+
const w = idx - lo;
|
|
605
|
+
return sorted[lo] * (1 - w) + sorted[hi] * w;
|
|
606
|
+
}
|
|
607
|
+
/** Even-width histogram over the value range. Returns inclusive-lo /
|
|
608
|
+
* exclusive-hi bins (closed on right for the last bin) compatible with
|
|
609
|
+
* the substrate's `GainDistributionBin` shape. */
|
|
610
|
+
function histogram(sorted, bins) {
|
|
611
|
+
if (sorted.length === 0 || bins < 1) return [];
|
|
612
|
+
const min = sorted[0];
|
|
613
|
+
const max = sorted[sorted.length - 1];
|
|
614
|
+
if (min === max) return [{
|
|
615
|
+
lo: min,
|
|
616
|
+
hi: max,
|
|
617
|
+
count: sorted.length
|
|
618
|
+
}];
|
|
619
|
+
const width = (max - min) / bins;
|
|
620
|
+
const out = [];
|
|
621
|
+
for (let i = 0; i < bins; i++) {
|
|
622
|
+
const lo = min + i * width;
|
|
623
|
+
const hi = i === bins - 1 ? max : lo + width;
|
|
624
|
+
out.push({
|
|
625
|
+
lo,
|
|
626
|
+
hi,
|
|
627
|
+
count: 0
|
|
628
|
+
});
|
|
629
|
+
}
|
|
630
|
+
for (const v of sorted) {
|
|
631
|
+
const idx = Math.min(bins - 1, Math.floor((v - min) / width));
|
|
632
|
+
out[idx].count++;
|
|
633
|
+
}
|
|
634
|
+
return out;
|
|
635
|
+
}
|
|
636
|
+
function computePerDimension(runs, bins) {
|
|
637
|
+
const byDim = /* @__PURE__ */ new Map();
|
|
638
|
+
for (const run of runs) {
|
|
639
|
+
const scores = run.outcome.judgeScores;
|
|
640
|
+
if (!scores) continue;
|
|
641
|
+
for (const [dim, value] of Object.entries(scores.perDimMean ?? {})) {
|
|
642
|
+
if (!Number.isFinite(value)) continue;
|
|
643
|
+
const arr = byDim.get(dim) ?? [];
|
|
644
|
+
arr.push(value);
|
|
645
|
+
byDim.set(dim, arr);
|
|
646
|
+
}
|
|
647
|
+
}
|
|
648
|
+
const out = {};
|
|
649
|
+
for (const [dim, values] of byDim) out[dim] = distributionOf(values, bins);
|
|
650
|
+
return out;
|
|
651
|
+
}
|
|
652
|
+
function computeJudgeInsights(runs) {
|
|
653
|
+
const out = {};
|
|
654
|
+
const byJudge = /* @__PURE__ */ new Map();
|
|
655
|
+
for (const run of runs) {
|
|
656
|
+
const scores = run.outcome.judgeScores;
|
|
657
|
+
if (!scores?.perJudge) continue;
|
|
658
|
+
for (const [judgeId, dims] of Object.entries(scores.perJudge)) {
|
|
659
|
+
const dimValues = Object.values(dims).filter(Number.isFinite);
|
|
660
|
+
if (dimValues.length === 0) continue;
|
|
661
|
+
const judgeMean = dimValues.reduce((s, v) => s + v, 0) / dimValues.length;
|
|
662
|
+
const arr = byJudge.get(judgeId) ?? [];
|
|
663
|
+
arr.push(judgeMean);
|
|
664
|
+
byJudge.set(judgeId, arr);
|
|
665
|
+
}
|
|
666
|
+
}
|
|
667
|
+
for (const [judgeId, values] of byJudge) out[judgeId] = {
|
|
668
|
+
n: values.length,
|
|
669
|
+
meanScore: values.reduce((s, v) => s + v, 0) / values.length
|
|
670
|
+
};
|
|
671
|
+
return out;
|
|
672
|
+
}
|
|
673
|
+
function computeInterRater(ratings) {
|
|
674
|
+
const byRun = /* @__PURE__ */ new Map();
|
|
675
|
+
for (const r of ratings) {
|
|
676
|
+
if (!Number.isFinite(r.score)) continue;
|
|
677
|
+
const list = byRun.get(r.runId) ?? [];
|
|
678
|
+
list.push({
|
|
679
|
+
rater: r.rater,
|
|
680
|
+
score: r.score
|
|
681
|
+
});
|
|
682
|
+
byRun.set(r.runId, list);
|
|
683
|
+
}
|
|
684
|
+
const raters = new Set(ratings.map((r) => r.rater));
|
|
685
|
+
const jointlyRated = [];
|
|
686
|
+
for (const [runId, ratersForRun] of byRun) {
|
|
687
|
+
const seen = new Set(ratersForRun.map((r) => r.rater));
|
|
688
|
+
let all = true;
|
|
689
|
+
for (const r of raters) if (!seen.has(r)) all = false;
|
|
690
|
+
if (all) jointlyRated.push(runId);
|
|
691
|
+
}
|
|
692
|
+
if (raters.size < 2 || jointlyRated.length === 0) return void 0;
|
|
693
|
+
const raterList = [...raters].sort();
|
|
694
|
+
const perPair = {};
|
|
695
|
+
for (let i = 0; i < raterList.length; i++) for (let j = i + 1; j < raterList.length; j++) {
|
|
696
|
+
const a = raterList[i];
|
|
697
|
+
const b = raterList[j];
|
|
698
|
+
const aScores = [];
|
|
699
|
+
const bScores = [];
|
|
700
|
+
for (const runId of jointlyRated) {
|
|
701
|
+
const ratersForRun = byRun.get(runId);
|
|
702
|
+
const sa = ratersForRun.find((r) => r.rater === a)?.score;
|
|
703
|
+
const sb = ratersForRun.find((r) => r.rater === b)?.score;
|
|
704
|
+
if (sa !== void 0 && sb !== void 0) {
|
|
705
|
+
aScores.push(sa);
|
|
706
|
+
bScores.push(sb);
|
|
707
|
+
}
|
|
708
|
+
}
|
|
709
|
+
const agreement = continuousAgreement(aScores.map((score, index) => [score, bScores[index]]), { bootstrap: 0 });
|
|
710
|
+
perPair[`${a}::${b}`] = agreement.weightedKappa;
|
|
711
|
+
}
|
|
712
|
+
const agreement = continuousAgreement(jointlyRated.map((runId) => {
|
|
713
|
+
const ratingsByRater = new Map(byRun.get(runId).map((rating) => [rating.rater, rating.score]));
|
|
714
|
+
return raterList.map((rater) => ratingsByRater.get(rater));
|
|
715
|
+
}), { bootstrap: 0 });
|
|
716
|
+
const disagreementCases = jointlyRated.map((runId) => {
|
|
717
|
+
const ratersForRun = byRun.get(runId);
|
|
718
|
+
const scores = ratersForRun.map((r) => r.score);
|
|
719
|
+
return {
|
|
720
|
+
runId,
|
|
721
|
+
ratings: ratersForRun,
|
|
722
|
+
range: Math.max(...scores) - Math.min(...scores)
|
|
723
|
+
};
|
|
724
|
+
}).sort((a, b) => b.range - a.range).slice(0, 20);
|
|
725
|
+
return {
|
|
726
|
+
raters: raters.size,
|
|
727
|
+
jointlyRated: jointlyRated.length,
|
|
728
|
+
kappa: Number.isFinite(agreement.weightedKappa) ? agreement.weightedKappa : 0,
|
|
729
|
+
icc: agreement.icc,
|
|
730
|
+
pearson: agreement.pearson,
|
|
731
|
+
spearman: agreement.spearman,
|
|
732
|
+
perPair,
|
|
733
|
+
disagreementCases
|
|
734
|
+
};
|
|
735
|
+
}
|
|
736
|
+
function computeLift(runs, baselineId, candidateId, split) {
|
|
737
|
+
let bId = baselineId;
|
|
738
|
+
let cId = candidateId;
|
|
739
|
+
if (!bId || !cId) {
|
|
740
|
+
const ids = [...new Set(runs.map((r) => r.candidateId))];
|
|
741
|
+
if (ids.length !== 2) return void 0;
|
|
742
|
+
const [idA, idB] = ids;
|
|
743
|
+
const scoresA = finiteCompositeScores(runs.filter((run) => run.candidateId === idA), split);
|
|
744
|
+
const scoresB = finiteCompositeScores(runs.filter((run) => run.candidateId === idB), split);
|
|
745
|
+
if (scoresA.length === 0 || scoresB.length === 0) return void 0;
|
|
746
|
+
const meanA = mean(scoresA);
|
|
747
|
+
const meanB = mean(scoresB);
|
|
748
|
+
bId = meanA <= meanB ? idA : idB;
|
|
749
|
+
cId = meanA <= meanB ? idB : idA;
|
|
750
|
+
}
|
|
751
|
+
const baseline = runs.filter((r) => r.candidateId === bId);
|
|
752
|
+
const candidate = runs.filter((r) => r.candidateId === cId);
|
|
753
|
+
if (baseline.length === 0 || candidate.length === 0) return void 0;
|
|
754
|
+
const pairing = pairRunRecords(baseline.filter((run) => Number.isFinite(compositeOf(run, split))), candidate.filter((run) => Number.isFinite(compositeOf(run, split))));
|
|
755
|
+
const pairedBaseline = pairing.pairs.map((pair) => compositeOf(pair.baseline, split));
|
|
756
|
+
const pairedCandidate = pairing.pairs.map((pair) => compositeOf(pair.treatment, split));
|
|
757
|
+
if (pairedBaseline.length === 0) return void 0;
|
|
758
|
+
const baselineMean = mean(pairedBaseline);
|
|
759
|
+
const candidateMean = mean(pairedCandidate);
|
|
760
|
+
const delta = candidateMean - baselineMean;
|
|
761
|
+
const bootstrap = pairedBootstrap(pairedBaseline, pairedCandidate, {
|
|
762
|
+
confidence: .95,
|
|
763
|
+
resamples: 2e3,
|
|
764
|
+
statistic: "mean"
|
|
765
|
+
});
|
|
766
|
+
const tTest = pairedTTest(pairedBaseline, pairedCandidate);
|
|
767
|
+
const d = pairedCohensDz(pairedBaseline, pairedCandidate);
|
|
768
|
+
const mde = pairedMde({
|
|
769
|
+
nPaired: pairedBaseline.length,
|
|
770
|
+
power: .8,
|
|
771
|
+
alpha: .05
|
|
772
|
+
});
|
|
773
|
+
const requiredN = d === null || d === 0 ? null : requiredPairedSampleSize({
|
|
774
|
+
effect: Math.abs(d),
|
|
775
|
+
power: .8,
|
|
776
|
+
alpha: .05
|
|
777
|
+
});
|
|
778
|
+
return {
|
|
779
|
+
baselineMean,
|
|
780
|
+
candidateMean,
|
|
781
|
+
delta,
|
|
782
|
+
ci95: [bootstrap.low, bootstrap.high],
|
|
783
|
+
pValue: tTest.p,
|
|
784
|
+
n: pairedBaseline.length,
|
|
785
|
+
unpairedBaseline: pairing.unpairedBaseline.length,
|
|
786
|
+
unpairedCandidate: pairing.unpairedTreatment.length,
|
|
787
|
+
cohensD: d,
|
|
788
|
+
mde,
|
|
789
|
+
requiredN
|
|
790
|
+
};
|
|
791
|
+
}
|
|
792
|
+
function mean(arr) {
|
|
793
|
+
return arr.length === 0 ? 0 : arr.reduce((s, v) => s + v, 0) / arr.length;
|
|
794
|
+
}
|
|
795
|
+
async function computeFailureClusters(runs, analyst, split) {
|
|
796
|
+
const failed = runs.filter((run) => isTaskFailure(run, split));
|
|
797
|
+
if (failed.length === 0) return {
|
|
798
|
+
clusters: [],
|
|
799
|
+
totalFailures: 0
|
|
800
|
+
};
|
|
801
|
+
const clusters = /* @__PURE__ */ new Map();
|
|
802
|
+
for (const run of failed) try {
|
|
803
|
+
const result = await analyst.run(run.runId, { runRecord: run });
|
|
804
|
+
for (const finding of result.findings) {
|
|
805
|
+
const key = finding.area || finding.analyst_id || "unclassified";
|
|
806
|
+
const c = clusters.get(key) ?? {
|
|
807
|
+
exemplars: [],
|
|
808
|
+
share: 0
|
|
809
|
+
};
|
|
810
|
+
if (c.exemplars.length < 5) c.exemplars.push(run.runId);
|
|
811
|
+
clusters.set(key, c);
|
|
812
|
+
}
|
|
813
|
+
} catch {
|
|
814
|
+
const c = clusters.get("analyst-error") ?? {
|
|
815
|
+
exemplars: [],
|
|
816
|
+
share: 0
|
|
817
|
+
};
|
|
818
|
+
if (c.exemplars.length < 5) c.exemplars.push(run.runId);
|
|
819
|
+
clusters.set("analyst-error", c);
|
|
820
|
+
}
|
|
821
|
+
const clusterList = [...clusters.entries()].map(([id, c]) => ({
|
|
822
|
+
id,
|
|
823
|
+
name: id,
|
|
824
|
+
share: c.exemplars.length / failed.length,
|
|
825
|
+
exemplars: c.exemplars
|
|
826
|
+
}));
|
|
827
|
+
clusterList.sort((a, b) => b.share - a.share);
|
|
828
|
+
return {
|
|
829
|
+
clusters: clusterList,
|
|
830
|
+
totalFailures: failed.length
|
|
831
|
+
};
|
|
832
|
+
}
|
|
833
|
+
function finiteCompositeScores(runs, split) {
|
|
834
|
+
return runs.map((run) => compositeOf(run, split)).filter(Number.isFinite);
|
|
835
|
+
}
|
|
836
|
+
function isTaskFailure(run, split) {
|
|
837
|
+
if (run.failureClass !== void 0 && run.failureClass !== "success") return true;
|
|
838
|
+
const score = compositeOf(run, split);
|
|
839
|
+
return Number.isFinite(score) && score < .5;
|
|
840
|
+
}
|
|
841
|
+
function computeContamination(runs, canaries) {
|
|
842
|
+
let leaks = 0;
|
|
843
|
+
const details = [];
|
|
844
|
+
for (const run of runs) {
|
|
845
|
+
const output = stringifyOutput(run);
|
|
846
|
+
if (!output) continue;
|
|
847
|
+
const leaksHere = checkCanaries(output, canaries);
|
|
848
|
+
for (const leak of leaksHere) {
|
|
849
|
+
leaks++;
|
|
850
|
+
details.push({
|
|
851
|
+
runId: run.runId,
|
|
852
|
+
canary: leak.canary,
|
|
853
|
+
matched: leak.evidence
|
|
854
|
+
});
|
|
855
|
+
}
|
|
856
|
+
}
|
|
857
|
+
return {
|
|
858
|
+
leaks,
|
|
859
|
+
holdoutAuditPassed: leaks === 0,
|
|
860
|
+
details
|
|
861
|
+
};
|
|
862
|
+
}
|
|
863
|
+
function stringifyOutput(run) {
|
|
864
|
+
const metadata = run.metadata;
|
|
865
|
+
if (typeof metadata?.output === "string") return metadata.output;
|
|
866
|
+
if (typeof metadata?.text === "string") return metadata.text;
|
|
867
|
+
}
|
|
868
|
+
function computeOutcomeCorrelation(runs, outcome, split) {
|
|
869
|
+
const xs = [];
|
|
870
|
+
const ys = [];
|
|
871
|
+
for (const run of runs) {
|
|
872
|
+
const y = outcome.valueByRunId[run.runId];
|
|
873
|
+
if (y === void 0 || !Number.isFinite(y)) continue;
|
|
874
|
+
const x = compositeOf(run, split);
|
|
875
|
+
if (!Number.isFinite(x)) continue;
|
|
876
|
+
xs.push(x);
|
|
877
|
+
ys.push(y);
|
|
878
|
+
}
|
|
879
|
+
if (xs.length < 3) return void 0;
|
|
880
|
+
const p = pearsonR(xs, ys);
|
|
881
|
+
const s = spearmanR(xs, ys);
|
|
882
|
+
const meanX = mean(xs);
|
|
883
|
+
const meanY = mean(ys);
|
|
884
|
+
let num = 0;
|
|
885
|
+
let denom = 0;
|
|
886
|
+
for (let i = 0; i < xs.length; i++) {
|
|
887
|
+
num += (xs[i] - meanX) * (ys[i] - meanY);
|
|
888
|
+
denom += (xs[i] - meanX) ** 2;
|
|
889
|
+
}
|
|
890
|
+
const slope = denom === 0 ? 0 : num / denom;
|
|
891
|
+
const intercept = meanY - slope * meanX;
|
|
892
|
+
const ssTot = ys.reduce((a, y) => a + (y - meanY) ** 2, 0);
|
|
893
|
+
const ssRes = ys.reduce((a, y, i) => a + (y - (intercept + slope * xs[i])) ** 2, 0);
|
|
894
|
+
const r2 = ssTot === 0 ? 0 : 1 - ssRes / ssTot;
|
|
895
|
+
return {
|
|
896
|
+
metric: outcome.metric,
|
|
897
|
+
n: xs.length,
|
|
898
|
+
pearson: p,
|
|
899
|
+
spearman: s,
|
|
900
|
+
rewardModel: {
|
|
901
|
+
intercept,
|
|
902
|
+
slope,
|
|
903
|
+
r2
|
|
904
|
+
}
|
|
905
|
+
};
|
|
906
|
+
}
|
|
907
|
+
function buildReleaseScorecard(composite, lift, contamination) {
|
|
908
|
+
const axes = [];
|
|
909
|
+
const liftPass = lift === void 0 ? "not_evaluated" : lift.ci95[0] > 0 ? "pass" : lift.delta > 0 ? "warn" : "fail";
|
|
910
|
+
axes.push({
|
|
911
|
+
name: "quality-lift",
|
|
912
|
+
status: liftPass,
|
|
913
|
+
detail: lift ? `delta=${lift.delta.toFixed(3)}, CI95=[${lift.ci95[0].toFixed(3)}, ${lift.ci95[1].toFixed(3)}], n=${lift.n}` : "no baseline/candidate pair available"
|
|
914
|
+
});
|
|
915
|
+
const contamPass = contamination === void 0 ? "not_evaluated" : contamination.leaks === 0 ? "pass" : "fail";
|
|
916
|
+
axes.push({
|
|
917
|
+
name: "contamination",
|
|
918
|
+
status: contamPass,
|
|
919
|
+
detail: contamination ? `${contamination.leaks} canary leak(s)` : "no canaries supplied"
|
|
920
|
+
});
|
|
921
|
+
axes.push(composite.n === 0 ? {
|
|
922
|
+
name: "composite-distribution",
|
|
923
|
+
status: "not_evaluated",
|
|
924
|
+
detail: "no task-quality scores available"
|
|
925
|
+
} : {
|
|
926
|
+
name: "composite-distribution",
|
|
927
|
+
status: composite.mean !== null && composite.mean >= .5 ? "pass" : composite.mean !== null && composite.mean >= .3 ? "warn" : "fail",
|
|
928
|
+
detail: composite.mean === null || composite.p50 === null || composite.p95 === null ? "task-quality distribution is internally incomplete" : `mean=${composite.mean.toFixed(3)}, p50=${composite.p50.toFixed(3)}, p95=${composite.p95.toFixed(3)} over n=${composite.n}`
|
|
929
|
+
});
|
|
930
|
+
return {
|
|
931
|
+
status: axes.some((a) => a.status === "fail") ? "fail" : axes.some((a) => a.status === "warn" || a.status === "not_evaluated") ? "warn" : "pass",
|
|
932
|
+
axes,
|
|
933
|
+
issues: []
|
|
934
|
+
};
|
|
935
|
+
}
|
|
936
|
+
function buildRecommendations(ctx) {
|
|
937
|
+
const out = [];
|
|
938
|
+
if (ctx.priorPeriodComparison) {
|
|
939
|
+
const ppc = ctx.priorPeriodComparison;
|
|
940
|
+
const label = ppc.windowLabel ?? "baseline period";
|
|
941
|
+
for (const name of ppc.regressedMetrics) {
|
|
942
|
+
const d = ppc.metrics[name];
|
|
943
|
+
if (!d) continue;
|
|
944
|
+
out.push({
|
|
945
|
+
priority: "critical",
|
|
946
|
+
kind: "investigate",
|
|
947
|
+
title: `${name} regressed from ${d.baseline.toFixed(3)} → ${d.current.toFixed(3)} vs ${label}`,
|
|
948
|
+
detail: `Welch CI95 = [${d.ci95[0].toFixed(3)}, ${d.ci95[1].toFixed(3)}], p=${d.pValue.toFixed(4)}, Cohen's d=${d.cohensD.toFixed(2)} (n_current=${d.currentN}, n_baseline=${d.baselineN}). The regression is statistically significant at p<0.05 with at-least-small effect size.`,
|
|
949
|
+
evidencePath: `priorPeriodComparison.metrics.${name}`
|
|
950
|
+
});
|
|
951
|
+
}
|
|
952
|
+
for (const name of ppc.improvedMetrics) {
|
|
953
|
+
const d = ppc.metrics[name];
|
|
954
|
+
if (!d) continue;
|
|
955
|
+
out.push({
|
|
956
|
+
priority: "low",
|
|
957
|
+
kind: "ship",
|
|
958
|
+
title: `${name} improved from ${d.baseline.toFixed(3)} → ${d.current.toFixed(3)} vs ${label}`,
|
|
959
|
+
detail: `Welch CI95 = [${d.ci95[0].toFixed(3)}, ${d.ci95[1].toFixed(3)}], p=${d.pValue.toFixed(4)}, Cohen's d=${d.cohensD.toFixed(2)} (n_current=${d.currentN}, n_baseline=${d.baselineN}). Statistically significant improvement worth flagging.`,
|
|
960
|
+
evidencePath: `priorPeriodComparison.metrics.${name}`
|
|
961
|
+
});
|
|
962
|
+
}
|
|
963
|
+
}
|
|
964
|
+
if (ctx.composite.n > 0 && ctx.composite.mean !== null && ctx.composite.p50 !== null && ctx.composite.p95 !== null) {
|
|
965
|
+
if (ctx.composite.mean < .3) {
|
|
966
|
+
const tail = ctx.composite.tailRuns ?? [];
|
|
967
|
+
const names = tail.slice(0, 5).map((t) => `${t.runId}=${t.score.toFixed(3)}`).join(", ");
|
|
968
|
+
out.push({
|
|
969
|
+
priority: "critical",
|
|
970
|
+
kind: "investigate",
|
|
971
|
+
title: `Composite mean ${ctx.composite.mean.toFixed(3)} is below the 0.3 floor — the agent is broken on this corpus`,
|
|
972
|
+
detail: tail.length > 0 ? `Worst ${tail.length} run${tail.length === 1 ? "" : "s"} to inspect first: ${names}. Histogram p50=${ctx.composite.p50.toFixed(3)}, p95=${ctx.composite.p95.toFixed(3)}.` : `Histogram p50=${ctx.composite.p50.toFixed(3)}, p95=${ctx.composite.p95.toFixed(3)}.`,
|
|
973
|
+
evidencePath: "composite.tailRuns"
|
|
974
|
+
});
|
|
975
|
+
} else if (ctx.composite.mean < .5) {
|
|
976
|
+
const tail = ctx.composite.tailRuns ?? [];
|
|
977
|
+
const names = tail.slice(0, 3).map((t) => `${t.runId}=${t.score.toFixed(3)}`).join(", ");
|
|
978
|
+
out.push({
|
|
979
|
+
priority: "high",
|
|
980
|
+
kind: "investigate",
|
|
981
|
+
title: `Composite mean ${ctx.composite.mean.toFixed(3)} is below 0.5 — investigate the lower tail before claiming the agent is healthy`,
|
|
982
|
+
detail: tail.length > 0 ? `Worst ${tail.length} run${tail.length === 1 ? "" : "s"}: ${names}. Histogram p50=${ctx.composite.p50.toFixed(3)}, p95=${ctx.composite.p95.toFixed(3)}.` : `Histogram p50=${ctx.composite.p50.toFixed(3)}, p95=${ctx.composite.p95.toFixed(3)}.`,
|
|
983
|
+
evidencePath: "composite.tailRuns"
|
|
984
|
+
});
|
|
985
|
+
}
|
|
986
|
+
}
|
|
987
|
+
if (ctx.failureClasses && ctx.failureClasses.length > 0) {
|
|
988
|
+
const top = ctx.failureClasses[0];
|
|
989
|
+
if (top.count >= 3 && top.share >= .15) out.push({
|
|
990
|
+
priority: top.share >= .25 ? "high" : "medium",
|
|
991
|
+
kind: "investigate",
|
|
992
|
+
title: `'${top.failureClass}' is the dominant failure class — ${top.count} runs (${(top.share * 100).toFixed(0)}% of the corpus)`,
|
|
993
|
+
detail: `The mean composite can look acceptable while one failure class dominates the lower tail. ${top.count} of ${ctx.composite.n} runs failed with '${top.failureClass}'${ctx.failureClasses.length > 1 ? ` (next: '${ctx.failureClasses[1].failureClass}' ×${ctx.failureClasses[1].count})` : ""}. Fix this cause first.`,
|
|
994
|
+
evidencePath: "failureClasses"
|
|
995
|
+
});
|
|
996
|
+
}
|
|
997
|
+
if (Object.keys(ctx.judges).length === 0 && ctx.composite.n > 0) out.push({
|
|
998
|
+
priority: "medium",
|
|
999
|
+
kind: "expand-corpus",
|
|
1000
|
+
title: "No judge scores recorded — per-dimension + calibration insights unavailable",
|
|
1001
|
+
detail: "Records have no `outcome.judgeScores`. To unlock perDimension, judges, and calibration, attach a Judge run during your eval pass and populate `outcome.judgeScores.perJudge[judgeName][dimension] = score`. See `docs/insight-report.md` for the expected shape.",
|
|
1002
|
+
evidencePath: "judges"
|
|
1003
|
+
});
|
|
1004
|
+
if (ctx.lift) {
|
|
1005
|
+
const pairedEffect = ctx.lift.cohensD === null ? "undefined (zero delta variance)" : ctx.lift.cohensD.toFixed(2);
|
|
1006
|
+
const requiredRuns = ctx.lift.requiredN === null ? "not estimable" : `~${ctx.lift.requiredN} paired runs`;
|
|
1007
|
+
const decisive = ctx.lift.ci95[0] > ctx.threshold;
|
|
1008
|
+
const inconclusive = ctx.lift.ci95[0] <= ctx.threshold && ctx.lift.ci95[1] > ctx.threshold;
|
|
1009
|
+
if (decisive) out.push({
|
|
1010
|
+
priority: "critical",
|
|
1011
|
+
kind: "ship",
|
|
1012
|
+
title: `Ship — lift ${ctx.lift.delta.toFixed(3)} (95% CI ${ctx.lift.ci95[0].toFixed(3)}..${ctx.lift.ci95[1].toFixed(3)})`,
|
|
1013
|
+
detail: `Holdout lift exceeds threshold ${ctx.threshold} with 95% bootstrap confidence (n=${ctx.lift.n}, p=${ctx.lift.pValue.toFixed(4)}, paired d=${pairedEffect}).`,
|
|
1014
|
+
evidencePath: "lift"
|
|
1015
|
+
});
|
|
1016
|
+
else if (inconclusive) out.push({
|
|
1017
|
+
priority: "high",
|
|
1018
|
+
kind: "expand-corpus",
|
|
1019
|
+
title: `Inconclusive — required sample is ${requiredRuns} (have ${ctx.lift.n}) at current effect size`,
|
|
1020
|
+
detail: `CI straddles threshold. Current MDE at 80% power is ${ctx.lift.mde.toFixed(3)}; observed delta is ${ctx.lift.delta.toFixed(3)}.`,
|
|
1021
|
+
evidencePath: "lift"
|
|
1022
|
+
});
|
|
1023
|
+
else out.push({
|
|
1024
|
+
priority: "critical",
|
|
1025
|
+
kind: "hold",
|
|
1026
|
+
title: `Hold — lift CI lower bound ${ctx.lift.ci95[0].toFixed(3)} is at or below threshold ${ctx.threshold}`,
|
|
1027
|
+
detail: `Bootstrap CI provides no statistical evidence the candidate is better. Consider tightening the mutation or expanding the holdout.`,
|
|
1028
|
+
evidencePath: "lift"
|
|
1029
|
+
});
|
|
1030
|
+
}
|
|
1031
|
+
if (ctx.contamination && ctx.contamination.leaks > 0) out.push({
|
|
1032
|
+
priority: "critical",
|
|
1033
|
+
kind: "fix",
|
|
1034
|
+
title: `${ctx.contamination.leaks} canary leak${ctx.contamination.leaks === 1 ? "" : "s"} detected`,
|
|
1035
|
+
detail: `Holdout integrity is compromised. The lift number is unreliable until you investigate.`,
|
|
1036
|
+
evidencePath: "contamination"
|
|
1037
|
+
});
|
|
1038
|
+
if (ctx.interRater && ctx.interRater.kappa < .5) out.push({
|
|
1039
|
+
priority: "high",
|
|
1040
|
+
kind: "recalibrate",
|
|
1041
|
+
title: `Inter-rater weighted kappa ${ctx.interRater.kappa.toFixed(2)} is below 0.5`,
|
|
1042
|
+
detail: "Raters disagree on what good looks like. Review the largest disagreement cases and refine the rubric before automating these decisions.",
|
|
1043
|
+
evidencePath: "interRater"
|
|
1044
|
+
});
|
|
1045
|
+
if (ctx.failureClusters && ctx.failureClusters.clusters.length > 0) {
|
|
1046
|
+
const top = ctx.failureClusters.clusters[0];
|
|
1047
|
+
out.push({
|
|
1048
|
+
priority: "high",
|
|
1049
|
+
kind: "investigate",
|
|
1050
|
+
title: `Top failure cluster: ${top.name} (${(top.share * 100).toFixed(0)}% of failures)`,
|
|
1051
|
+
detail: `${ctx.failureClusters.totalFailures} runs failed. The largest cluster groups ${top.exemplars.length} exemplars under '${top.name}'.`,
|
|
1052
|
+
evidencePath: "failureClusters.clusters[0]"
|
|
1053
|
+
});
|
|
1054
|
+
}
|
|
1055
|
+
if (ctx.outcomeCorrelation && Math.abs(ctx.outcomeCorrelation.spearman) < .3) out.push({
|
|
1056
|
+
priority: "medium",
|
|
1057
|
+
kind: "recalibrate",
|
|
1058
|
+
title: `Judge scores decoupled from ${ctx.outcomeCorrelation.metric} (Spearman ρ=${ctx.outcomeCorrelation.spearman.toFixed(2)})`,
|
|
1059
|
+
detail: `Your judges score what they were trained to score, but it isn't predicting downstream ${ctx.outcomeCorrelation.metric}. Consider retraining the judge against ${ctx.outcomeCorrelation.metric} as the gold signal.`,
|
|
1060
|
+
evidencePath: "outcomeCorrelation"
|
|
1061
|
+
});
|
|
1062
|
+
return out;
|
|
1063
|
+
}
|
|
1064
|
+
//#endregion
|
|
1065
|
+
export { checkBehavioralCanary as a, canaryLeakView as i, summarizeExecution as n, checkCanaries as o, HoldoutAuditor as r, runBehavioralCanaries as s, analyzeRuns as t };
|
|
1066
|
+
|
|
1067
|
+
//# sourceMappingURL=analyze-runs-C1CavBMk.js.map
|