@tangle-network/agent-eval 0.128.2 → 0.130.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +279 -0
- package/README.md +19 -0
- package/dist/active-curriculum-C4mk67HP.js +214 -0
- package/dist/active-curriculum-C4mk67HP.js.map +1 -0
- package/dist/adversarial-smnADNFS.d.ts +21 -0
- package/dist/adversarial-smnADNFS.d.ts.map +1 -0
- package/dist/analyst/index.d.ts +83 -2932
- package/dist/analyst/index.d.ts.map +1 -0
- package/dist/analyst/index.js +319 -364
- package/dist/analyst/index.js.map +1 -1
- package/dist/analyst-BkTS3C58.d.ts +89 -0
- package/dist/analyst-BkTS3C58.d.ts.map +1 -0
- package/dist/analyst-LsnNpSkm.js +152 -0
- package/dist/analyst-LsnNpSkm.js.map +1 -0
- package/dist/analyze-runs-C1CavBMk.js +1067 -0
- package/dist/analyze-runs-C1CavBMk.js.map +1 -0
- package/dist/analyze-runs-FsgYCinh.d.ts +72 -0
- package/dist/analyze-runs-FsgYCinh.d.ts.map +1 -0
- package/dist/attribute-vocabulary-DLJ6303h.d.ts +54 -0
- package/dist/attribute-vocabulary-DLJ6303h.d.ts.map +1 -0
- package/dist/authenticity/index.d.ts +81 -79
- package/dist/authenticity/index.d.ts.map +1 -0
- package/dist/authenticity/index.js +209 -193
- package/dist/authenticity/index.js.map +1 -1
- package/dist/baseline-HsBvw_dk.js +550 -0
- package/dist/baseline-HsBvw_dk.js.map +1 -0
- package/dist/baseline-hG3K85h4.d.ts +125 -0
- package/dist/baseline-hG3K85h4.d.ts.map +1 -0
- package/dist/belief-state/index.d.ts +448 -1205
- package/dist/belief-state/index.d.ts.map +1 -0
- package/dist/belief-state/index.js +1617 -1710
- package/dist/belief-state/index.js.map +1 -1
- package/dist/benchmarks/index.d.ts +2 -894
- package/dist/benchmarks/index.js +2 -59
- package/dist/benchmarks-DviOvUNr.js +754 -0
- package/dist/benchmarks-DviOvUNr.js.map +1 -0
- package/dist/builder-eval/index.d.ts +150 -662
- package/dist/builder-eval/index.d.ts.map +1 -0
- package/dist/builder-eval/index.js +356 -345
- package/dist/builder-eval/index.js.map +1 -1
- package/dist/calibration-CNWWA6K8.js +94 -0
- package/dist/calibration-CNWWA6K8.js.map +1 -0
- package/dist/campaign/index.d.ts +5 -6390
- package/dist/campaign/index.js +3 -212
- package/dist/campaign-CBKZvQ1H.js +3885 -0
- package/dist/campaign-CBKZvQ1H.js.map +1 -0
- package/dist/cli.d.ts +1 -1
- package/dist/cli.js +164 -174
- package/dist/cli.js.map +1 -1
- package/dist/client-C97NMzqi.d.ts +581 -0
- package/dist/client-C97NMzqi.d.ts.map +1 -0
- package/dist/client-CYzbdJOZ.js +637 -0
- package/dist/client-CYzbdJOZ.js.map +1 -0
- package/dist/code-agent-session-BjkMTQ7H.js +1390 -0
- package/dist/code-agent-session-BjkMTQ7H.js.map +1 -0
- package/dist/code-agent-session-DqqgOJaz.d.ts +143 -0
- package/dist/code-agent-session-DqqgOJaz.d.ts.map +1 -0
- package/dist/concurrency-DIxRZF_J.js +85 -0
- package/dist/concurrency-DIxRZF_J.js.map +1 -0
- package/dist/contract/index.d.ts +645 -5605
- package/dist/contract/index.d.ts.map +1 -0
- package/dist/contract/index.js +1730 -1937
- package/dist/contract/index.js.map +1 -1
- package/dist/control.d.ts +3 -1030
- package/dist/control.js +2 -32
- package/dist/cost-ledger-DIgQUFZZ.js +801 -0
- package/dist/cost-ledger-DIgQUFZZ.js.map +1 -0
- package/dist/cost-ledger-Dye6jCgg.d.ts +236 -0
- package/dist/cost-ledger-Dye6jCgg.d.ts.map +1 -0
- package/dist/dataset-BvtnC8Dc.d.ts +115 -0
- package/dist/dataset-BvtnC8Dc.d.ts.map +1 -0
- package/dist/default-registry-C-vFCSEc.js +2579 -0
- package/dist/default-registry-C-vFCSEc.js.map +1 -0
- package/dist/default-registry-CNPo-Vsb.d.ts +540 -0
- package/dist/default-registry-CNPo-Vsb.d.ts.map +1 -0
- package/dist/emitter-CPBAhxum.js +266 -0
- package/dist/emitter-CPBAhxum.js.map +1 -0
- package/dist/emitter-DGQGoLyj.d.ts +113 -0
- package/dist/emitter-DGQGoLyj.d.ts.map +1 -0
- package/dist/errors-8YnH8WlF.js +64 -0
- package/dist/errors-8YnH8WlF.js.map +1 -0
- package/dist/errors-CEk209JS.d.ts +76 -0
- package/dist/errors-CEk209JS.d.ts.map +1 -0
- package/dist/eval-campaign-DEm6c8ru.js +349 -0
- package/dist/eval-campaign-DEm6c8ru.js.map +1 -0
- package/dist/execution-tracks-CpgFPpS5.js +93 -0
- package/dist/execution-tracks-CpgFPpS5.js.map +1 -0
- package/dist/exporters-q9iL-2Jf.js +148 -0
- package/dist/exporters-q9iL-2Jf.js.map +1 -0
- package/dist/extract-usage-BrQ8mCLX.js +155 -0
- package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
- package/dist/failure-cluster-CqcvCcdR.d.ts +59 -0
- package/dist/failure-cluster-CqcvCcdR.d.ts.map +1 -0
- package/dist/feedback-trajectory-CVaeREXV.d.ts +340 -0
- package/dist/feedback-trajectory-CVaeREXV.d.ts.map +1 -0
- package/dist/fuzz.d.ts +320 -646
- package/dist/fuzz.d.ts.map +1 -0
- package/dist/fuzz.js +670 -617
- package/dist/fuzz.js.map +1 -1
- package/dist/hf-dataset-DBJXXoY1.js +763 -0
- package/dist/hf-dataset-DBJXXoY1.js.map +1 -0
- package/dist/hosted/index.d.ts +11 -831
- package/dist/hosted/index.d.ts.map +1 -0
- package/dist/hosted/index.js +2 -37
- package/dist/index-2JJSA6-r2.d.ts +926 -0
- package/dist/index-2JJSA6-r2.d.ts.map +1 -0
- package/dist/index-6N0aYmpW.d.ts +217 -0
- package/dist/index-6N0aYmpW.d.ts.map +1 -0
- package/dist/index-BAvgST_9.d.ts +131 -0
- package/dist/index-BAvgST_9.d.ts.map +1 -0
- package/dist/index-BvnJuTGD.d.ts +68 -0
- package/dist/index-BvnJuTGD.d.ts.map +1 -0
- package/dist/index-C61Wi7yg.d.ts +547 -0
- package/dist/index-C61Wi7yg.d.ts.map +1 -0
- package/dist/index-CAPUUKaM.d.ts +335 -0
- package/dist/index-CAPUUKaM.d.ts.map +1 -0
- package/dist/index-DE5fb3EC.d.ts +2244 -0
- package/dist/index-DE5fb3EC.d.ts.map +1 -0
- package/dist/index-DSC51roc.d.ts +102 -0
- package/dist/index-DSC51roc.d.ts.map +1 -0
- package/dist/index.d.ts +3776 -15120
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +11185 -11191
- package/dist/index.js.map +1 -1
- package/dist/integrity-BzRbCHzi.js +424 -0
- package/dist/integrity-BzRbCHzi.js.map +1 -0
- package/dist/integrity-rmVhXWA7.d.ts +61 -0
- package/dist/integrity-rmVhXWA7.d.ts.map +1 -0
- package/dist/judge-calibration-DFtEMlde.d.ts +146 -0
- package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
- package/dist/ledger-core/index.d.ts +2 -0
- package/dist/ledger-core/index.js +2 -0
- package/dist/ledger-core-DtZz1RG0.js +388 -0
- package/dist/ledger-core-DtZz1RG0.js.map +1 -0
- package/dist/llm-client--GR4JbZE.js +687 -0
- package/dist/llm-client--GR4JbZE.js.map +1 -0
- package/dist/llm-client-B_nIBlYo.d.ts +290 -0
- package/dist/llm-client-B_nIBlYo.d.ts.map +1 -0
- package/dist/matrix/index.d.ts +3 -155
- package/dist/matrix/index.js +2 -8
- package/dist/matrix-BzQnu2S6.js +270 -0
- package/dist/matrix-BzQnu2S6.js.map +1 -0
- package/dist/meta-eval/index.d.ts +4 -1027
- package/dist/meta-eval/index.js +393 -390
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/metrics-C9YY1OcL.js +239 -0
- package/dist/metrics-C9YY1OcL.js.map +1 -0
- package/dist/mint-yN2M2eh0.js +201 -0
- package/dist/mint-yN2M2eh0.js.map +1 -0
- package/dist/multi-layer-verifier-BHY1gWAc.d.ts +138 -0
- package/dist/multi-layer-verifier-BHY1gWAc.d.ts.map +1 -0
- package/dist/multishot/index.d.ts +273 -481
- package/dist/multishot/index.d.ts.map +1 -0
- package/dist/multishot/index.js +595 -548
- package/dist/multishot/index.js.map +1 -1
- package/dist/off-policy-DvgzvtIx.js +220 -0
- package/dist/off-policy-DvgzvtIx.js.map +1 -0
- package/dist/off-policy-mskQw8Mb.d.ts +153 -0
- package/dist/off-policy-mskQw8Mb.d.ts.map +1 -0
- package/dist/openapi.json +1 -1
- package/dist/opencode-sqlite-BGrHeDu3.js +318 -0
- package/dist/opencode-sqlite-BGrHeDu3.js.map +1 -0
- package/dist/outcome-store-BYHIuO0e.d.ts +65 -0
- package/dist/outcome-store-BYHIuO0e.d.ts.map +1 -0
- package/dist/outcome-store-ChBKlTd_.js +75 -0
- package/dist/outcome-store-ChBKlTd_.js.map +1 -0
- package/dist/paired-arms-D9D0wXj2.js +260 -0
- package/dist/paired-arms-D9D0wXj2.js.map +1 -0
- package/dist/pipelines/index.d.ts +95 -532
- package/dist/pipelines/index.d.ts.map +1 -0
- package/dist/pipelines/index.js +497 -478
- package/dist/pipelines/index.js.map +1 -1
- package/dist/pre-registration-DakwTRXk.js +96 -0
- package/dist/pre-registration-DakwTRXk.js.map +1 -0
- package/dist/propose-review-control-LhIWGzHl.js +1458 -0
- package/dist/propose-review-control-LhIWGzHl.js.map +1 -0
- package/dist/query-CJ_DX8vl.d.ts +28 -0
- package/dist/query-CJ_DX8vl.d.ts.map +1 -0
- package/dist/query-Di7eEQ79.js +83 -0
- package/dist/query-Di7eEQ79.js.map +1 -0
- package/dist/raw-provider-sink-BQd7mzyT.js +201 -0
- package/dist/raw-provider-sink-BQd7mzyT.js.map +1 -0
- package/dist/raw-provider-sink-BU29Sh8h.d.ts +134 -0
- package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +1 -0
- package/dist/redact-7Aq1ukl-.js +107 -0
- package/dist/redact-7Aq1ukl-.js.map +1 -0
- package/dist/release-report-Cz9NKH39.js +603 -0
- package/dist/release-report-Cz9NKH39.js.map +1 -0
- package/dist/release-report-mvB2G4_J.d.ts +244 -0
- package/dist/release-report-mvB2G4_J.d.ts.map +1 -0
- package/dist/replay-GnyotH0J.js +1741 -0
- package/dist/replay-GnyotH0J.js.map +1 -0
- package/dist/replay-OoidtG1E.d.ts +749 -0
- package/dist/replay-OoidtG1E.d.ts.map +1 -0
- package/dist/reporting.d.ts +6 -1298
- package/dist/reporting.js +6 -50
- package/dist/researcher-CwTdwXG1.d.ts +314 -0
- package/dist/researcher-CwTdwXG1.d.ts.map +1 -0
- package/dist/reward-hacking-eAnOsynk.d.ts +224 -0
- package/dist/reward-hacking-eAnOsynk.d.ts.map +1 -0
- package/dist/reward-hacking-qipEpKvY.js +596 -0
- package/dist/reward-hacking-qipEpKvY.js.map +1 -0
- package/dist/reward-nw2xZGZG.js +137 -0
- package/dist/reward-nw2xZGZG.js.map +1 -0
- package/dist/rl.d.ts +916 -3596
- package/dist/rl.d.ts.map +1 -0
- package/dist/rl.js +2362 -1751
- package/dist/rl.js.map +1 -1
- package/dist/rolldown-runtime-8H4AJuhK.js +14 -0
- package/dist/rollout/index.d.ts +3 -1048
- package/dist/rollout/index.js +8 -110
- package/dist/rollout-BOYjemfR.js +624 -0
- package/dist/rollout-BOYjemfR.js.map +1 -0
- package/dist/rubric-predictive-validity-B3xmbmS1.js +141 -0
- package/dist/rubric-predictive-validity-B3xmbmS1.js.map +1 -0
- package/dist/rubric-predictive-validity-Ku_clp_1.d.ts +74 -0
- package/dist/rubric-predictive-validity-Ku_clp_1.d.ts.map +1 -0
- package/dist/run-evidence-ClRX_8A9.d.ts +225 -0
- package/dist/run-evidence-ClRX_8A9.d.ts.map +1 -0
- package/dist/run-record-BuoE80Dq.js +467 -0
- package/dist/run-record-BuoE80Dq.js.map +1 -0
- package/dist/run-record-CnZu_gjl.d.ts +357 -0
- package/dist/run-record-CnZu_gjl.d.ts.map +1 -0
- package/dist/run-score-iEEAWiBY.js +41 -0
- package/dist/run-score-iEEAWiBY.js.map +1 -0
- package/dist/runtime-trajectory-1gyaTOoC.js +93 -0
- package/dist/runtime-trajectory-1gyaTOoC.js.map +1 -0
- package/dist/runtime-trajectory-BvSZcCHD.d.ts +50 -0
- package/dist/runtime-trajectory-BvSZcCHD.d.ts.map +1 -0
- package/dist/schema-BtVldJ3T.d.ts +206 -0
- package/dist/schema-BtVldJ3T.d.ts.map +1 -0
- package/dist/schema-C6DW4ZHR.js +821 -0
- package/dist/schema-C6DW4ZHR.js.map +1 -0
- package/dist/schema-CRhEY1SO.js +69 -0
- package/dist/schema-CRhEY1SO.js.map +1 -0
- package/dist/schema-Cef2cFmb.d.ts +408 -0
- package/dist/schema-Cef2cFmb.d.ts.map +1 -0
- package/dist/semantic-concept-judge-B6cWNJ2K.js +725 -0
- package/dist/semantic-concept-judge-B6cWNJ2K.js.map +1 -0
- package/dist/sequential-Br0mAPHA.js +148 -0
- package/dist/sequential-Br0mAPHA.js.map +1 -0
- package/dist/sequential-CYwq6Ff_.d.ts +141 -0
- package/dist/sequential-CYwq6Ff_.d.ts.map +1 -0
- package/dist/series-convergence-CjO2QdRW.js +43 -0
- package/dist/series-convergence-CjO2QdRW.js.map +1 -0
- package/dist/series-convergence-ofsqPWhs.d.ts +35 -0
- package/dist/series-convergence-ofsqPWhs.d.ts.map +1 -0
- package/dist/server-m5D9cvnG.js +1040 -0
- package/dist/server-m5D9cvnG.js.map +1 -0
- package/dist/skill-usage-C_pXm7lP.d.ts +534 -0
- package/dist/skill-usage-C_pXm7lP.d.ts.map +1 -0
- package/dist/skillopt-optimization-method-B7wX7XkF.d.ts +1739 -0
- package/dist/skillopt-optimization-method-B7wX7XkF.d.ts.map +1 -0
- package/dist/skillopt-optimization-method-D4ODwFVV.js +7712 -0
- package/dist/skillopt-optimization-method-D4ODwFVV.js.map +1 -0
- package/dist/statistics-Cmj6nynr.d.ts +514 -0
- package/dist/statistics-Cmj6nynr.d.ts.map +1 -0
- package/dist/statistics-CnnxdpOg.js +1437 -0
- package/dist/statistics-CnnxdpOg.js.map +1 -0
- package/dist/store-CT9YIIve.d.ts +117 -0
- package/dist/store-CT9YIIve.d.ts.map +1 -0
- package/dist/store-CxJry_cs.d.ts +229 -0
- package/dist/store-CxJry_cs.d.ts.map +1 -0
- package/dist/storyboard/index.d.ts +84 -203
- package/dist/storyboard/index.d.ts.map +1 -0
- package/dist/storyboard/index.js +609 -542
- package/dist/storyboard/index.js.map +1 -1
- package/dist/summary-report-BNs5nmXI.js +862 -0
- package/dist/summary-report-BNs5nmXI.js.map +1 -0
- package/dist/summary-report-Cj9gdw4i.d.ts +407 -0
- package/dist/summary-report-Cj9gdw4i.d.ts.map +1 -0
- package/dist/supervisor-run/index.d.ts +2 -849
- package/dist/supervisor-run/index.js +2 -64
- package/dist/supervisor-run-_lnTLM3z.js +1679 -0
- package/dist/supervisor-run-_lnTLM3z.js.map +1 -0
- package/dist/task-failure-attributes-CQZlB3et.js +311 -0
- package/dist/task-failure-attributes-CQZlB3et.js.map +1 -0
- package/dist/test-graded-scenario-BsqWLmPt.js +318 -0
- package/dist/test-graded-scenario-BsqWLmPt.js.map +1 -0
- package/dist/test-graded-scenario-D1TaI2va.d.ts +141 -0
- package/dist/test-graded-scenario-D1TaI2va.d.ts.map +1 -0
- package/dist/tools-BmuN627J.js +1085 -0
- package/dist/tools-BmuN627J.js.map +1 -0
- package/dist/trace-attributes.d.ts +2 -52
- package/dist/trace-attributes.js +131 -61
- package/dist/trace-attributes.js.map +1 -1
- package/dist/traces.d.ts +12 -2365
- package/dist/traces.js +12 -251
- package/dist/trajectory-D_7rLrvE.js +56 -0
- package/dist/trajectory-D_7rLrvE.js.map +1 -0
- package/dist/types-DGsxbAEd.d.ts +387 -0
- package/dist/types-DGsxbAEd.d.ts.map +1 -0
- package/dist/types-k9tZGKUg.d.ts +640 -0
- package/dist/types-k9tZGKUg.d.ts.map +1 -0
- package/dist/verdict-Dps8_okt.d.ts +37 -0
- package/dist/verdict-Dps8_okt.d.ts.map +1 -0
- package/dist/wire/index.d.ts +702 -1174
- package/dist/wire/index.d.ts.map +1 -0
- package/dist/wire/index.js +2 -81
- package/docs/feature-guide.md +1 -1
- package/docs/rollout.md +116 -2
- package/package.json +18 -10
- package/dist/benchmarks/index.js.map +0 -1
- package/dist/campaign/index.js.map +0 -1
- package/dist/chunk-2JX3CFMB.js +0 -695
- package/dist/chunk-2JX3CFMB.js.map +0 -1
- package/dist/chunk-2MKQIFS4.js +0 -183
- package/dist/chunk-2MKQIFS4.js.map +0 -1
- package/dist/chunk-3RF76KTD.js +0 -84
- package/dist/chunk-3RF76KTD.js.map +0 -1
- package/dist/chunk-5DTSBUL2.js +0 -159
- package/dist/chunk-5DTSBUL2.js.map +0 -1
- package/dist/chunk-7ZZMD7UK.js +0 -386
- package/dist/chunk-7ZZMD7UK.js.map +0 -1
- package/dist/chunk-BOD4O7OF.js +0 -40
- package/dist/chunk-BOD4O7OF.js.map +0 -1
- package/dist/chunk-BYT7ELPS.js +0 -1553
- package/dist/chunk-BYT7ELPS.js.map +0 -1
- package/dist/chunk-DJKY2TSY.js +0 -2428
- package/dist/chunk-DJKY2TSY.js.map +0 -1
- package/dist/chunk-DPUHNQLN.js +0 -232
- package/dist/chunk-DPUHNQLN.js.map +0 -1
- package/dist/chunk-DRYIUNWY.js +0 -622
- package/dist/chunk-DRYIUNWY.js.map +0 -1
- package/dist/chunk-EJGRPCO3.js +0 -617
- package/dist/chunk-EJGRPCO3.js.map +0 -1
- package/dist/chunk-EOSZT7PL.js +0 -2001
- package/dist/chunk-EOSZT7PL.js.map +0 -1
- package/dist/chunk-EZJEIH2R.js +0 -1559
- package/dist/chunk-EZJEIH2R.js.map +0 -1
- package/dist/chunk-GGE4NNQT.js +0 -65
- package/dist/chunk-GGE4NNQT.js.map +0 -1
- package/dist/chunk-HHWE3POT.js +0 -94
- package/dist/chunk-HHWE3POT.js.map +0 -1
- package/dist/chunk-IHQDPH7D.js +0 -171
- package/dist/chunk-IHQDPH7D.js.map +0 -1
- package/dist/chunk-JHCHEVET.js +0 -274
- package/dist/chunk-JHCHEVET.js.map +0 -1
- package/dist/chunk-K4DBDHLK.js +0 -158
- package/dist/chunk-K4DBDHLK.js.map +0 -1
- package/dist/chunk-K6N6XJJX.js +0 -306
- package/dist/chunk-K6N6XJJX.js.map +0 -1
- package/dist/chunk-MA6HLL3S.js +0 -65
- package/dist/chunk-MA6HLL3S.js.map +0 -1
- package/dist/chunk-MAZ26DC7.js +0 -99
- package/dist/chunk-MAZ26DC7.js.map +0 -1
- package/dist/chunk-MHELPNRP.js +0 -1212
- package/dist/chunk-MHELPNRP.js.map +0 -1
- package/dist/chunk-NACAGYSY.js +0 -1040
- package/dist/chunk-NACAGYSY.js.map +0 -1
- package/dist/chunk-NKAGIDE2.js +0 -7633
- package/dist/chunk-NKAGIDE2.js.map +0 -1
- package/dist/chunk-NPCTHQIO.js +0 -91
- package/dist/chunk-NPCTHQIO.js.map +0 -1
- package/dist/chunk-NYLOYM6N.js +0 -332
- package/dist/chunk-NYLOYM6N.js.map +0 -1
- package/dist/chunk-ONWEPEDO.js +0 -57
- package/dist/chunk-ONWEPEDO.js.map +0 -1
- package/dist/chunk-P5W7RQKK.js +0 -576
- package/dist/chunk-P5W7RQKK.js.map +0 -1
- package/dist/chunk-P6FYH6K4.js +0 -1161
- package/dist/chunk-P6FYH6K4.js.map +0 -1
- package/dist/chunk-PBE2LOSS.js +0 -669
- package/dist/chunk-PBE2LOSS.js.map +0 -1
- package/dist/chunk-PC4UYEBM.js +0 -166
- package/dist/chunk-PC4UYEBM.js.map +0 -1
- package/dist/chunk-PXE2VKMX.js +0 -140
- package/dist/chunk-PXE2VKMX.js.map +0 -1
- package/dist/chunk-PZ5AY32C.js +0 -10
- package/dist/chunk-PZ5AY32C.js.map +0 -1
- package/dist/chunk-RZTMDUO7.js +0 -49
- package/dist/chunk-RZTMDUO7.js.map +0 -1
- package/dist/chunk-S5YLIBFX.js +0 -136
- package/dist/chunk-S5YLIBFX.js.map +0 -1
- package/dist/chunk-SZLVEKMJ.js +0 -1446
- package/dist/chunk-SZLVEKMJ.js.map +0 -1
- package/dist/chunk-T4SQEITX.js +0 -95
- package/dist/chunk-T4SQEITX.js.map +0 -1
- package/dist/chunk-TBL77AUT.js +0 -355
- package/dist/chunk-TBL77AUT.js.map +0 -1
- package/dist/chunk-TSN7JT6D.js +0 -1646
- package/dist/chunk-TSN7JT6D.js.map +0 -1
- package/dist/chunk-TT4KNT67.js +0 -124
- package/dist/chunk-TT4KNT67.js.map +0 -1
- package/dist/chunk-UB2LOJ6Q.js +0 -4461
- package/dist/chunk-UB2LOJ6Q.js.map +0 -1
- package/dist/chunk-UWZZKKU7.js +0 -237
- package/dist/chunk-UWZZKKU7.js.map +0 -1
- package/dist/chunk-VBQ3CRKH.js +0 -291
- package/dist/chunk-VBQ3CRKH.js.map +0 -1
- package/dist/chunk-VGRCHJON.js +0 -163
- package/dist/chunk-VGRCHJON.js.map +0 -1
- package/dist/chunk-VI2UW6B6.js +0 -162
- package/dist/chunk-VI2UW6B6.js.map +0 -1
- package/dist/chunk-VLOATJQ2.js +0 -908
- package/dist/chunk-VLOATJQ2.js.map +0 -1
- package/dist/chunk-VQMK5FMP.js +0 -247
- package/dist/chunk-VQMK5FMP.js.map +0 -1
- package/dist/chunk-VZSRQ272.js +0 -149
- package/dist/chunk-VZSRQ272.js.map +0 -1
- package/dist/chunk-WGXIEX7P.js +0 -116
- package/dist/chunk-WGXIEX7P.js.map +0 -1
- package/dist/chunk-WS3NZZQQ.js +0 -929
- package/dist/chunk-WS3NZZQQ.js.map +0 -1
- package/dist/chunk-XDWDC2MP.js +0 -695
- package/dist/chunk-XDWDC2MP.js.map +0 -1
- package/dist/chunk-XPRT64IE.js +0 -766
- package/dist/chunk-XPRT64IE.js.map +0 -1
- package/dist/chunk-YJBNWCAA.js +0 -1056
- package/dist/chunk-YJBNWCAA.js.map +0 -1
- package/dist/chunk-ZET2UAYW.js +0 -89
- package/dist/chunk-ZET2UAYW.js.map +0 -1
- package/dist/chunk-ZUUWPZCV.js +0 -752
- package/dist/chunk-ZUUWPZCV.js.map +0 -1
- package/dist/control.js.map +0 -1
- package/dist/hosted/index.js.map +0 -1
- package/dist/matrix/index.js.map +0 -1
- package/dist/reporting.js.map +0 -1
- package/dist/rollout/index.js.map +0 -1
- package/dist/run-campaign-ISHFZ7FJ.js +0 -17
- package/dist/run-campaign-ISHFZ7FJ.js.map +0 -1
- package/dist/supervisor-run/index.js.map +0 -1
- package/dist/traces.js.map +0 -1
- package/dist/wire/index.js.map +0 -1
package/dist/contract/index.js
CHANGED
|
@@ -1,2097 +1,1890 @@
|
|
|
1
|
-
import {
|
|
2
|
-
|
|
3
|
-
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
} from "../
|
|
11
|
-
import {
|
|
12
|
-
|
|
13
|
-
} from "../
|
|
14
|
-
import {
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
emitLoopProvenance,
|
|
28
|
-
externalTextOptimizationMethod,
|
|
29
|
-
gepaOptimizationMethod,
|
|
30
|
-
heldOutGate,
|
|
31
|
-
heldoutSignificance,
|
|
32
|
-
llmJudge,
|
|
33
|
-
loopProvenanceArgsFromResult,
|
|
34
|
-
paretoPolicy,
|
|
35
|
-
paretoSignificanceGate,
|
|
36
|
-
powerPreflight,
|
|
37
|
-
runEval,
|
|
38
|
-
runImprovementLoop,
|
|
39
|
-
runReferenceEquivalenceJudge,
|
|
40
|
-
skillOptOptimizationMethod,
|
|
41
|
-
surfaceContentHash,
|
|
42
|
-
surfaceHash
|
|
43
|
-
} from "../chunk-NKAGIDE2.js";
|
|
44
|
-
import {
|
|
45
|
-
campaignSplitDigest,
|
|
46
|
-
createRunCostLedger,
|
|
47
|
-
fsCampaignStorage,
|
|
48
|
-
inMemoryCampaignStorage,
|
|
49
|
-
resolveRunDir,
|
|
50
|
-
runCampaign
|
|
51
|
-
} from "../chunk-EZJEIH2R.js";
|
|
52
|
-
import {
|
|
53
|
-
buildDefaultAnalystRegistry,
|
|
54
|
-
createChatClient
|
|
55
|
-
} from "../chunk-DJKY2TSY.js";
|
|
56
|
-
import "../chunk-HHWE3POT.js";
|
|
57
|
-
import "../chunk-WGXIEX7P.js";
|
|
58
|
-
import {
|
|
59
|
-
FileSystemOutcomeStore,
|
|
60
|
-
InMemoryOutcomeStore
|
|
61
|
-
} from "../chunk-3RF76KTD.js";
|
|
62
|
-
import "../chunk-NYLOYM6N.js";
|
|
63
|
-
import {
|
|
64
|
-
campaignCellExecutionEvidence,
|
|
65
|
-
campaignCellJudgeDimensions,
|
|
66
|
-
campaignCellTaskScore,
|
|
67
|
-
campaignCellToRunRecord
|
|
68
|
-
} from "../chunk-2MKQIFS4.js";
|
|
69
|
-
import "../chunk-PBE2LOSS.js";
|
|
70
|
-
import "../chunk-VLOATJQ2.js";
|
|
71
|
-
import "../chunk-DPUHNQLN.js";
|
|
72
|
-
import {
|
|
73
|
-
pairedBootstrap
|
|
74
|
-
} from "../chunk-MHELPNRP.js";
|
|
75
|
-
import "../chunk-WS3NZZQQ.js";
|
|
76
|
-
import "../chunk-VI2UW6B6.js";
|
|
77
|
-
import {
|
|
78
|
-
readTaskFailureLabels,
|
|
79
|
-
recordAggregateMeasurements,
|
|
80
|
-
summarizeExecutionMeasurements,
|
|
81
|
-
summarizeTraceErrors
|
|
82
|
-
} from "../chunk-7ZZMD7UK.js";
|
|
83
|
-
import "../chunk-PXE2VKMX.js";
|
|
84
|
-
import "../chunk-ZET2UAYW.js";
|
|
85
|
-
import "../chunk-GGE4NNQT.js";
|
|
86
|
-
import {
|
|
87
|
-
classifyOtlpSpanRole,
|
|
88
|
-
isOtlpModelCall
|
|
89
|
-
} from "../chunk-P6FYH6K4.js";
|
|
90
|
-
import "../chunk-PC4UYEBM.js";
|
|
91
|
-
import {
|
|
92
|
-
modelHasSnapshot,
|
|
93
|
-
parseRunRecordSafe
|
|
94
|
-
} from "../chunk-2JX3CFMB.js";
|
|
95
|
-
import "../chunk-MA6HLL3S.js";
|
|
96
|
-
import {
|
|
97
|
-
ValidationError
|
|
98
|
-
} from "../chunk-ONWEPEDO.js";
|
|
99
|
-
import {
|
|
100
|
-
LLM_MODEL_ATTR_KEYS,
|
|
101
|
-
SPAN_KIND_ATTR_KEYS
|
|
102
|
-
} from "../chunk-K4DBDHLK.js";
|
|
103
|
-
import "../chunk-PZ5AY32C.js";
|
|
104
|
-
|
|
105
|
-
// src/contract/self-improve.ts
|
|
1
|
+
import { s as ValidationError } from "../errors-8YnH8WlF.js";
|
|
2
|
+
import { i as parseRunRecordSafe, r as modelHasSnapshot } from "../run-record-BuoE80Dq.js";
|
|
3
|
+
import { L as createChatClient, t as buildDefaultAnalystRegistry } from "../default-registry-C-vFCSEc.js";
|
|
4
|
+
import { LLM_MODEL_ATTR_KEYS, SPAN_KIND_ATTR_KEYS } from "../trace-attributes.js";
|
|
5
|
+
import { b as classifyOtlpSpanRole, x as isOtlpModelCall } from "../tools-BmuN627J.js";
|
|
6
|
+
import { B as surfaceContentHash, Ct as llmJudge, M as compareOptimizationMethods, O as composeGate, Q as inMemoryCampaignStorage, S as defaultProductionGate, T as heldoutSignificance, V as surfaceHash, X as createRunCostLedger, Y as runCampaign, Z as fsCampaignStorage, _ as buildEvidenceVector, a as emitLoopProvenance, b as powerPreflight, ct as campaignSplitDigest, d as runImprovementLoop, dt as REFERENCE_EQUIVALENCE_INPUT_LIMITS, ft as REFERENCE_EQUIVALENCE_JUDGE_VERSION, g as gepaOptimizationMethod, j as assertOptimizationResult, k as externalTextOptimizationMethod, mt as runReferenceEquivalenceJudge, o as loopProvenanceArgsFromResult, p as runEval, pt as createReferenceEquivalenceJudge, rt as resolveRunDir, t as skillOptOptimizationMethod, v as paretoPolicy, x as heldOutGate, y as paretoSignificanceGate } from "../skillopt-optimization-method-D4ODwFVV.js";
|
|
7
|
+
import { v as pairedBootstrap } from "../statistics-CnnxdpOg.js";
|
|
8
|
+
import { i as summarizeTraceErrors, n as recordAggregateMeasurements, r as summarizeExecutionMeasurements, t as readTaskFailureLabels } from "../task-failure-attributes-CQZlB3et.js";
|
|
9
|
+
import { n as summarizeExecution, t as analyzeRuns } from "../analyze-runs-C1CavBMk.js";
|
|
10
|
+
import { a as campaignCellExecutionEvidence, c as campaignCellToRunRecord, o as campaignCellJudgeDimensions, s as campaignCellTaskScore } from "../reward-hacking-qipEpKvY.js";
|
|
11
|
+
import { n as InMemoryOutcomeStore, t as FileSystemOutcomeStore } from "../outcome-store-ChBKlTd_.js";
|
|
12
|
+
import { a as fromPiSession, c as observeCodeAgentSession, i as fromOpenCodeSession, n as fromCodexSession, o as fromPigraphSession, r as fromKimiCodeSession, s as parseCodeAgentJsonl, t as fromClaudeCodeSession } from "../code-agent-session-BjkMTQ7H.js";
|
|
13
|
+
import { t as createHostedClient } from "../client-CYzbdJOZ.js";
|
|
14
|
+
import { dirname, join } from "node:path";
|
|
15
|
+
import { mkdir, readFile, readdir, stat, writeFile } from "node:fs/promises";
|
|
16
|
+
import { agentCandidateBenchmarkSuiteSchema, agentCandidateBenchmarkTaskSchema, agentCandidateBundleSchema, agentCandidateEvaluationPolicySchema, agentCandidateExperimentSchema, agentImprovementMeasuredComparisonSchema, candidateExecutionEvidenceSchema, canonicalCandidateDigest, omitTopLevelDigest } from "@tangle-network/agent-interface";
|
|
17
|
+
//#region src/contract/self-improve.ts
|
|
18
|
+
/**
|
|
19
|
+
* Run one complete improvement job.
|
|
20
|
+
*
|
|
21
|
+
* A caller-owned `proposer` can generate candidates across local generations.
|
|
22
|
+
* An external `method`, such as official GEPA or SkillOpt, owns its complete
|
|
23
|
+
* search and returns one candidate. Both paths remeasure the selected candidate
|
|
24
|
+
* against cases that candidate generation never receives.
|
|
25
|
+
*/
|
|
26
|
+
/** Failed self-improvement run with an immutable receipt snapshot. */
|
|
106
27
|
var SelfImproveRunError = class extends Error {
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
28
|
+
cost;
|
|
29
|
+
receipts;
|
|
30
|
+
constructor(cause, ledger) {
|
|
31
|
+
const original = cause instanceof Error ? cause : new Error(String(cause));
|
|
32
|
+
super(original.message, { cause: original });
|
|
33
|
+
this.name = "SelfImproveRunError";
|
|
34
|
+
this.cost = ledger.summary();
|
|
35
|
+
this.receipts = ledger.list();
|
|
36
|
+
}
|
|
116
37
|
};
|
|
117
38
|
function assertSelfImproveSearchMode(opts) {
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
throw new Error("selfImprove: method must have a trimmed name and optimize(input)");
|
|
129
|
-
}
|
|
130
|
-
const budget = opts.budget;
|
|
131
|
-
if (budget?.generations !== void 0 && budget.generations !== 1) {
|
|
132
|
-
throw new Error("selfImprove: method owns its rounds; budget.generations must be 1 when set");
|
|
133
|
-
}
|
|
134
|
-
if (budget?.populationSize !== void 0 && budget.populationSize !== 1) {
|
|
135
|
-
throw new Error(
|
|
136
|
-
"selfImprove: method owns its candidates; budget.populationSize must be 1 when set"
|
|
137
|
-
);
|
|
138
|
-
}
|
|
139
|
-
if (budget?.candidateConcurrency !== void 0 || budget?.maxImprovementShots !== void 0 || opts.analyzeGeneration !== void 0 || opts.findings !== void 0) {
|
|
140
|
-
throw new Error(
|
|
141
|
-
"selfImprove: candidateConcurrency, maxImprovementShots, analyzeGeneration, and findings apply only to proposer mode"
|
|
142
|
-
);
|
|
143
|
-
}
|
|
39
|
+
if (opts.method && opts.proposer) throw new Error("selfImprove: method and proposer are mutually exclusive");
|
|
40
|
+
if (!opts.method) {
|
|
41
|
+
if (opts.selectionScenarios !== void 0) throw new Error("selfImprove: selectionScenarios requires method");
|
|
42
|
+
return;
|
|
43
|
+
}
|
|
44
|
+
if (typeof opts.method.name !== "string" || !opts.method.name.trim() || opts.method.name.trim() !== opts.method.name || typeof opts.method.optimize !== "function") throw new Error("selfImprove: method must have a trimmed name and optimize(input)");
|
|
45
|
+
const budget = opts.budget;
|
|
46
|
+
if (budget?.generations !== void 0 && budget.generations !== 1) throw new Error("selfImprove: method owns its rounds; budget.generations must be 1 when set");
|
|
47
|
+
if (budget?.populationSize !== void 0 && budget.populationSize !== 1) throw new Error("selfImprove: method owns its candidates; budget.populationSize must be 1 when set");
|
|
48
|
+
if (budget?.candidateConcurrency !== void 0 || budget?.maxImprovementShots !== void 0 || opts.analyzeGeneration !== void 0 || opts.findings !== void 0) throw new Error("selfImprove: candidateConcurrency, maxImprovementShots, analyzeGeneration, and findings apply only to proposer mode");
|
|
144
49
|
}
|
|
145
50
|
function splitMethodPartitions(searchScenarios, explicitSelection, fraction) {
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
throw new Error("selfImprove: method train split is empty");
|
|
175
|
-
}
|
|
176
|
-
return {
|
|
177
|
-
train,
|
|
178
|
-
selection: explicitSelection.map((scenario) => byId.get(scenario.id))
|
|
179
|
-
};
|
|
180
|
-
}
|
|
181
|
-
if (searchScenarios.length < 2) {
|
|
182
|
-
throw new Error("selfImprove: method requires at least two non-final scenarios");
|
|
183
|
-
}
|
|
184
|
-
const sorted = [...searchScenarios].sort(
|
|
185
|
-
(a, b) => stableScenarioHash(a.id) - stableScenarioHash(b.id)
|
|
186
|
-
);
|
|
187
|
-
const count = Math.max(1, Math.min(sorted.length - 1, Math.round(sorted.length * fraction)));
|
|
188
|
-
return {
|
|
189
|
-
selection: sorted.slice(0, count),
|
|
190
|
-
train: sorted.slice(count)
|
|
191
|
-
};
|
|
51
|
+
if (!Number.isFinite(fraction) || fraction <= 0 || fraction >= 1) throw new Error("selfImprove: budget.selectionFraction must be in (0, 1)");
|
|
52
|
+
const byId = /* @__PURE__ */ new Map();
|
|
53
|
+
for (const scenario of searchScenarios) {
|
|
54
|
+
if (byId.has(scenario.id)) throw new Error(`selfImprove: duplicate scenario id '${scenario.id}'`);
|
|
55
|
+
byId.set(scenario.id, scenario);
|
|
56
|
+
}
|
|
57
|
+
if (explicitSelection) {
|
|
58
|
+
if (explicitSelection.length === 0) throw new Error("selfImprove: selectionScenarios must not be empty");
|
|
59
|
+
const selectionIds = /* @__PURE__ */ new Set();
|
|
60
|
+
for (const scenario of explicitSelection) {
|
|
61
|
+
if (!byId.has(scenario.id)) throw new Error(`selfImprove: selection scenario '${scenario.id}' is absent from the non-final cases`);
|
|
62
|
+
if (selectionIds.has(scenario.id)) throw new Error(`selfImprove: duplicate selection scenario id '${scenario.id}'`);
|
|
63
|
+
selectionIds.add(scenario.id);
|
|
64
|
+
}
|
|
65
|
+
const train = searchScenarios.filter((scenario) => !selectionIds.has(scenario.id));
|
|
66
|
+
if (train.length === 0) throw new Error("selfImprove: method train split is empty");
|
|
67
|
+
return {
|
|
68
|
+
train,
|
|
69
|
+
selection: explicitSelection.map((scenario) => byId.get(scenario.id))
|
|
70
|
+
};
|
|
71
|
+
}
|
|
72
|
+
if (searchScenarios.length < 2) throw new Error("selfImprove: method requires at least two non-final scenarios");
|
|
73
|
+
const sorted = [...searchScenarios].sort((a, b) => stableScenarioHash(a.id) - stableScenarioHash(b.id));
|
|
74
|
+
const count = Math.max(1, Math.min(sorted.length - 1, Math.round(sorted.length * fraction)));
|
|
75
|
+
return {
|
|
76
|
+
selection: sorted.slice(0, count),
|
|
77
|
+
train: sorted.slice(count)
|
|
78
|
+
};
|
|
192
79
|
}
|
|
193
80
|
function safeRunComponent(value) {
|
|
194
|
-
|
|
81
|
+
return value.replace(/[^a-zA-Z0-9._-]/g, "_");
|
|
195
82
|
}
|
|
196
83
|
function stableScenarioHash(value) {
|
|
197
|
-
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
|
|
202
|
-
|
|
203
|
-
}
|
|
84
|
+
let hash = 2166136261;
|
|
85
|
+
for (let index = 0; index < value.length; index++) {
|
|
86
|
+
hash ^= value.charCodeAt(index);
|
|
87
|
+
hash = Math.imul(hash, 16777619) >>> 0;
|
|
88
|
+
}
|
|
89
|
+
return hash;
|
|
90
|
+
}
|
|
91
|
+
/**
|
|
92
|
+
* Deterministic train/holdout split by a stable hash of `scenario.id`,
|
|
93
|
+
* so the same scenario set always splits the same way across runs.
|
|
94
|
+
*/
|
|
204
95
|
function splitTrainHoldout(scenarios, fraction) {
|
|
205
|
-
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
|
|
96
|
+
const sorted = [...scenarios].sort((a, b) => stableScenarioHash(a.id) - stableScenarioHash(b.id));
|
|
97
|
+
const nHoldout = Math.max(1, Math.min(sorted.length - 1, Math.round(sorted.length * fraction)));
|
|
98
|
+
return {
|
|
99
|
+
holdout: sorted.slice(0, nHoldout),
|
|
100
|
+
train: sorted.slice(nHoldout)
|
|
101
|
+
};
|
|
211
102
|
}
|
|
212
103
|
function meanComposite(byScenario) {
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
|
|
221
|
-
|
|
222
|
-
|
|
223
|
-
}
|
|
104
|
+
const perScenario = {};
|
|
105
|
+
const values = [];
|
|
106
|
+
for (const [id, agg] of Object.entries(byScenario)) {
|
|
107
|
+
perScenario[id] = agg.meanComposite;
|
|
108
|
+
values.push(agg.meanComposite);
|
|
109
|
+
}
|
|
110
|
+
return {
|
|
111
|
+
compositeMean: values.length === 0 ? 0 : values.reduce((s, v) => s + v, 0) / values.length,
|
|
112
|
+
perScenario
|
|
113
|
+
};
|
|
114
|
+
}
|
|
115
|
+
/**
|
|
116
|
+
* Latest search campaign measured for the winner surface; the baseline search
|
|
117
|
+
* campaign when the winner IS the baseline. Used by the deferred-holdout
|
|
118
|
+
* summary, where no holdout campaign exists to summarize.
|
|
119
|
+
*/
|
|
224
120
|
function winnerSearchCampaign(result) {
|
|
225
|
-
|
|
226
|
-
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
|
|
230
|
-
|
|
231
|
-
|
|
232
|
-
|
|
121
|
+
for (let i = result.generations.length - 1; i >= 0; i--) {
|
|
122
|
+
const measured = result.generations[i]?.surfaces.find((s) => s.surfaceHash === result.winnerSurfaceHash);
|
|
123
|
+
if (measured) return measured.campaign;
|
|
124
|
+
}
|
|
125
|
+
return result.baselineCampaign;
|
|
126
|
+
}
|
|
127
|
+
/**
|
|
128
|
+
* One-shot self-improvement loop. See module docstring for defaults +
|
|
129
|
+
* extension points.
|
|
130
|
+
*
|
|
131
|
+
* @example Minimum:
|
|
132
|
+
*
|
|
133
|
+
* const result = await selfImprove({
|
|
134
|
+
* agent: (surface, scenario, ctx) => myAgent(surface, scenario, ctx.signal),
|
|
135
|
+
* scenarios,
|
|
136
|
+
* judge,
|
|
137
|
+
* baselineSurface: DEFAULT_PROMPT,
|
|
138
|
+
* proposer,
|
|
139
|
+
* })
|
|
140
|
+
* console.log(`lift: ${result.lift.toFixed(3)} (${result.gateDecision})`)
|
|
141
|
+
*
|
|
142
|
+
* @example Distributed (workers in three regions):
|
|
143
|
+
*
|
|
144
|
+
* await selfImprove({
|
|
145
|
+
* agent: httpDispatch({ resolveUrl: ({ placement }) => REGION_URLS[placement!] }),
|
|
146
|
+
* scenarios,
|
|
147
|
+
* judge,
|
|
148
|
+
* baselineSurface: DEFAULT_PROMPT,
|
|
149
|
+
* cellPlacement: ({ scenario }) => scenario.region,
|
|
150
|
+
* budget: { maxConcurrency: 12 },
|
|
151
|
+
* })
|
|
152
|
+
*/
|
|
233
153
|
async function selfImprove(opts) {
|
|
234
|
-
|
|
235
|
-
|
|
236
|
-
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
|
|
242
|
-
|
|
243
|
-
|
|
244
|
-
|
|
245
|
-
|
|
246
|
-
|
|
247
|
-
}
|
|
154
|
+
const startedAt = Date.now();
|
|
155
|
+
const runDir = resolveRunDir(opts.runDir ?? (opts.method ? `.agent-eval/runs/self-improve-${startedAt}` : `mem://selfImprove-${startedAt}`));
|
|
156
|
+
const storage = opts.storage ?? (runDir.startsWith("mem://") ? inMemoryCampaignStorage() : fsCampaignStorage());
|
|
157
|
+
const costLedger = createRunCostLedger({
|
|
158
|
+
storage,
|
|
159
|
+
runDir,
|
|
160
|
+
costCeilingUsd: opts.budget?.dollars
|
|
161
|
+
});
|
|
162
|
+
try {
|
|
163
|
+
return await runSelfImprove(opts, costLedger, startedAt, runDir, storage);
|
|
164
|
+
} catch (error) {
|
|
165
|
+
throw new SelfImproveRunError(error, costLedger);
|
|
166
|
+
}
|
|
248
167
|
}
|
|
249
168
|
async function runSelfImprove(opts, costLedger, startedAt, runDir, storage) {
|
|
250
|
-
|
|
251
|
-
|
|
252
|
-
|
|
253
|
-
|
|
254
|
-
|
|
255
|
-
|
|
256
|
-
|
|
257
|
-
|
|
258
|
-
|
|
259
|
-
|
|
260
|
-
|
|
261
|
-
|
|
262
|
-
|
|
263
|
-
|
|
264
|
-
|
|
265
|
-
|
|
266
|
-
|
|
267
|
-
|
|
268
|
-
|
|
269
|
-
|
|
270
|
-
|
|
271
|
-
|
|
272
|
-
|
|
273
|
-
|
|
274
|
-
|
|
275
|
-
|
|
276
|
-
|
|
277
|
-
|
|
278
|
-
|
|
279
|
-
|
|
280
|
-
|
|
281
|
-
|
|
282
|
-
|
|
283
|
-
|
|
284
|
-
|
|
285
|
-
|
|
286
|
-
|
|
287
|
-
|
|
288
|
-
|
|
289
|
-
|
|
290
|
-
|
|
291
|
-
|
|
292
|
-
|
|
293
|
-
|
|
294
|
-
|
|
295
|
-
|
|
296
|
-
|
|
297
|
-
|
|
298
|
-
|
|
299
|
-
|
|
300
|
-
|
|
301
|
-
|
|
302
|
-
|
|
303
|
-
|
|
304
|
-
|
|
305
|
-
|
|
306
|
-
|
|
307
|
-
|
|
308
|
-
|
|
309
|
-
|
|
310
|
-
|
|
311
|
-
|
|
312
|
-
|
|
313
|
-
|
|
314
|
-
|
|
315
|
-
|
|
316
|
-
|
|
317
|
-
|
|
318
|
-
|
|
319
|
-
|
|
320
|
-
|
|
321
|
-
|
|
322
|
-
|
|
323
|
-
|
|
324
|
-
|
|
325
|
-
|
|
326
|
-
|
|
327
|
-
|
|
328
|
-
|
|
329
|
-
|
|
330
|
-
|
|
331
|
-
|
|
332
|
-
|
|
333
|
-
|
|
334
|
-
|
|
335
|
-
|
|
336
|
-
|
|
337
|
-
|
|
338
|
-
|
|
339
|
-
|
|
340
|
-
|
|
341
|
-
|
|
342
|
-
|
|
343
|
-
|
|
344
|
-
|
|
345
|
-
|
|
346
|
-
|
|
347
|
-
|
|
348
|
-
|
|
349
|
-
|
|
350
|
-
|
|
351
|
-
|
|
352
|
-
|
|
353
|
-
|
|
354
|
-
|
|
355
|
-
|
|
356
|
-
|
|
357
|
-
|
|
358
|
-
|
|
359
|
-
|
|
360
|
-
|
|
361
|
-
|
|
362
|
-
|
|
363
|
-
|
|
364
|
-
|
|
365
|
-
|
|
366
|
-
|
|
367
|
-
|
|
368
|
-
|
|
369
|
-
|
|
370
|
-
|
|
371
|
-
|
|
372
|
-
|
|
373
|
-
|
|
374
|
-
|
|
375
|
-
|
|
376
|
-
|
|
377
|
-
|
|
378
|
-
|
|
379
|
-
|
|
380
|
-
|
|
381
|
-
|
|
382
|
-
|
|
383
|
-
|
|
384
|
-
|
|
385
|
-
|
|
386
|
-
|
|
387
|
-
|
|
388
|
-
|
|
389
|
-
|
|
390
|
-
|
|
391
|
-
|
|
392
|
-
|
|
393
|
-
|
|
394
|
-
|
|
395
|
-
|
|
396
|
-
|
|
397
|
-
|
|
398
|
-
|
|
399
|
-
|
|
400
|
-
|
|
401
|
-
|
|
402
|
-
|
|
403
|
-
|
|
404
|
-
|
|
405
|
-
|
|
406
|
-
|
|
407
|
-
|
|
408
|
-
|
|
409
|
-
|
|
410
|
-
|
|
411
|
-
|
|
412
|
-
|
|
413
|
-
|
|
414
|
-
|
|
415
|
-
|
|
416
|
-
|
|
417
|
-
|
|
418
|
-
|
|
419
|
-
|
|
420
|
-
|
|
421
|
-
|
|
422
|
-
|
|
423
|
-
|
|
424
|
-
|
|
425
|
-
|
|
426
|
-
|
|
427
|
-
|
|
428
|
-
|
|
429
|
-
|
|
430
|
-
|
|
431
|
-
|
|
432
|
-
|
|
433
|
-
|
|
434
|
-
|
|
435
|
-
|
|
436
|
-
|
|
437
|
-
|
|
438
|
-
|
|
439
|
-
|
|
440
|
-
|
|
441
|
-
|
|
442
|
-
|
|
443
|
-
|
|
444
|
-
|
|
445
|
-
}
|
|
446
|
-
} : {},
|
|
447
|
-
storage,
|
|
448
|
-
hostedClient: opts.hostedTenant ? createHostedClient(opts.hostedTenant) : void 0
|
|
449
|
-
});
|
|
450
|
-
if (opts.onProvenance) opts.onProvenance(provenance);
|
|
451
|
-
const summary = {
|
|
452
|
-
baseline,
|
|
453
|
-
winner: {
|
|
454
|
-
...winnerStats,
|
|
455
|
-
surface: result.winnerSurface,
|
|
456
|
-
...result.winnerLabel ? { label: result.winnerLabel } : {},
|
|
457
|
-
...result.winnerRationale ? { rationale: result.winnerRationale } : {}
|
|
458
|
-
},
|
|
459
|
-
...holdoutDeferred ? {} : { lift: winnerStats.compositeMean - baseline.compositeMean },
|
|
460
|
-
diff: result.promotedDiff,
|
|
461
|
-
provenance,
|
|
462
|
-
gateDecision: result.gateResult.decision,
|
|
463
|
-
generationsExplored: result.generations.length,
|
|
464
|
-
durationMs,
|
|
465
|
-
totalCostUsd: totalCost,
|
|
466
|
-
cost,
|
|
467
|
-
receipts: costLedger.list(),
|
|
468
|
-
...optimizationResult ? {
|
|
469
|
-
optimization: {
|
|
470
|
-
name: opts.method.name,
|
|
471
|
-
cost: structuredClone(optimizationResult.cost),
|
|
472
|
-
...optimizationResult.durationMs === void 0 ? {} : { durationMs: optimizationResult.durationMs },
|
|
473
|
-
...optimizationResult.provenance === void 0 ? {} : { provenance: structuredClone(optimizationResult.provenance) }
|
|
474
|
-
}
|
|
475
|
-
} : {},
|
|
476
|
-
insight,
|
|
477
|
-
...power ? { power } : {},
|
|
478
|
-
raw: result
|
|
479
|
-
};
|
|
480
|
-
if (opts.hostedTenant) {
|
|
481
|
-
try {
|
|
482
|
-
await shipEvalRunToHosted(opts.hostedTenant, opts, summary, result, runDir);
|
|
483
|
-
} catch (err) {
|
|
484
|
-
const msg = err instanceof Error ? err.message : String(err);
|
|
485
|
-
console.warn(`[agent-eval] hosted ingest failed (continuing): ${msg}`);
|
|
486
|
-
}
|
|
487
|
-
}
|
|
488
|
-
return summary;
|
|
169
|
+
const budget = opts.budget ?? {};
|
|
170
|
+
assertSelfImproveSearchMode(opts);
|
|
171
|
+
const generations = opts.method ? 1 : budget.generations ?? 3;
|
|
172
|
+
const populationSize = opts.method ? 1 : budget.populationSize ?? 2;
|
|
173
|
+
const maxConcurrency = budget.maxConcurrency ?? 2;
|
|
174
|
+
const holdoutFraction = budget.holdoutFraction ?? .25;
|
|
175
|
+
const holdoutMode = budget.holdout ?? "measured";
|
|
176
|
+
const holdoutDeferred = holdoutMode === "deferred";
|
|
177
|
+
const expectUsage = opts.expectUsage ?? "assert";
|
|
178
|
+
const explicitHoldout = budget.holdoutScenarios;
|
|
179
|
+
const { train, holdout } = explicitHoldout ? {
|
|
180
|
+
train: opts.scenarios.filter((s) => !explicitHoldout.some((h) => h.id === s.id)),
|
|
181
|
+
holdout: explicitHoldout
|
|
182
|
+
} : holdoutDeferred ? {
|
|
183
|
+
train: opts.scenarios,
|
|
184
|
+
holdout: []
|
|
185
|
+
} : splitTrainHoldout(opts.scenarios, holdoutFraction);
|
|
186
|
+
if (train.length === 0) throw new Error("selfImprove: train split is empty. Reduce holdoutFraction or pass more scenarios.");
|
|
187
|
+
if (holdout.length === 0 && !holdoutDeferred) throw new Error("selfImprove: holdout split is empty. Pass more scenarios.");
|
|
188
|
+
if (generations > 0 && !opts.proposer && !opts.method) throw new Error("selfImprove: method or proposer is required when budget.generations is greater than zero");
|
|
189
|
+
let optimizationResult;
|
|
190
|
+
const methodPartitions = opts.method ? splitMethodPartitions(train, opts.selectionScenarios, budget.selectionFraction ?? .25) : void 0;
|
|
191
|
+
const proposer = opts.method ? {
|
|
192
|
+
kind: `method:${opts.method.name}`,
|
|
193
|
+
propose: async (context) => {
|
|
194
|
+
if (context.generation > 0) return [];
|
|
195
|
+
const result = await opts.method.optimize(Object.freeze({
|
|
196
|
+
baselineSurface: structuredClone(context.currentSurface),
|
|
197
|
+
trainScenarios: Object.freeze(methodPartitions.train.map((scenario) => structuredClone(scenario))),
|
|
198
|
+
selectionScenarios: Object.freeze(methodPartitions.selection.map((scenario) => structuredClone(scenario))),
|
|
199
|
+
dispatchWithSurface: opts.agent,
|
|
200
|
+
judges: Object.freeze([opts.judge]),
|
|
201
|
+
runDir: `${runDir}/optimization/${safeRunComponent(opts.method.name)}`,
|
|
202
|
+
seed: 42,
|
|
203
|
+
runOptions: Object.freeze({
|
|
204
|
+
storage,
|
|
205
|
+
maxConcurrency,
|
|
206
|
+
reps: budget.reps,
|
|
207
|
+
dispatchTimeoutMs: opts.dispatchTimeoutMs,
|
|
208
|
+
expectUsage,
|
|
209
|
+
costCeiling: budget.dollars
|
|
210
|
+
}),
|
|
211
|
+
costLedger
|
|
212
|
+
}));
|
|
213
|
+
assertOptimizationResult(opts.method.name, result);
|
|
214
|
+
optimizationResult = structuredClone(result);
|
|
215
|
+
return [{
|
|
216
|
+
surface: structuredClone(result.winnerSurface),
|
|
217
|
+
label: opts.method.name,
|
|
218
|
+
rationale: `${opts.method.name} selected this surface without final cases.`
|
|
219
|
+
}];
|
|
220
|
+
}
|
|
221
|
+
} : opts.proposer ?? {
|
|
222
|
+
kind: "baseline-only",
|
|
223
|
+
propose: async () => []
|
|
224
|
+
};
|
|
225
|
+
const gate = opts.gate ?? defaultProductionGate({
|
|
226
|
+
holdoutScenarios: holdout,
|
|
227
|
+
deltaThreshold: .05
|
|
228
|
+
});
|
|
229
|
+
if (opts.onProgress) opts.onProgress({
|
|
230
|
+
kind: "baseline.started",
|
|
231
|
+
scenarios: opts.scenarios.length
|
|
232
|
+
});
|
|
233
|
+
const result = await runImprovementLoop({
|
|
234
|
+
scenarios: train,
|
|
235
|
+
baselineSurface: opts.baselineSurface,
|
|
236
|
+
premeasuredBaseline: opts.premeasuredBaseline,
|
|
237
|
+
dispatchWithSurface: opts.agent,
|
|
238
|
+
proposer,
|
|
239
|
+
judges: [opts.judge],
|
|
240
|
+
populationSize,
|
|
241
|
+
maxGenerations: generations,
|
|
242
|
+
candidateConcurrency: budget.candidateConcurrency,
|
|
243
|
+
reps: budget.reps,
|
|
244
|
+
maxImprovementShots: budget.maxImprovementShots,
|
|
245
|
+
holdoutScenarios: holdout,
|
|
246
|
+
holdout: holdoutMode,
|
|
247
|
+
gate,
|
|
248
|
+
neutralize: opts.neutralize,
|
|
249
|
+
autoOnPromote: opts.autoOnPromote ?? "none",
|
|
250
|
+
ghOwner: opts.ghOwner,
|
|
251
|
+
ghRepo: opts.ghRepo,
|
|
252
|
+
storage,
|
|
253
|
+
runDir,
|
|
254
|
+
maxConcurrency,
|
|
255
|
+
cellPlacement: opts.cellPlacement,
|
|
256
|
+
dispatchTimeoutMs: opts.dispatchTimeoutMs,
|
|
257
|
+
costLedger,
|
|
258
|
+
expectUsage,
|
|
259
|
+
labeledStore: opts.labeledStore,
|
|
260
|
+
captureSource: opts.captureSource,
|
|
261
|
+
analyzeGeneration: opts.analyzeGeneration,
|
|
262
|
+
findings: opts.findings,
|
|
263
|
+
selectionRankKey: opts.selectionRankKey
|
|
264
|
+
});
|
|
265
|
+
const reportSplit = holdoutDeferred ? "search" : "holdout";
|
|
266
|
+
const reportBaselineCampaign = holdoutDeferred ? result.baselineCampaign : result.baselineOnHoldout;
|
|
267
|
+
const reportWinnerCampaign = holdoutDeferred ? winnerSearchCampaign(result) : result.winnerOnHoldout;
|
|
268
|
+
const baseline = meanComposite(reportBaselineCampaign.aggregates.byScenario);
|
|
269
|
+
const winnerStats = meanComposite(reportWinnerCampaign.aggregates.byScenario);
|
|
270
|
+
let power;
|
|
271
|
+
const baselineHoldoutComposites = result.baselineOnHoldout.cells.filter((cell) => !cell.error).map((cell) => {
|
|
272
|
+
const scores = Object.values(cell.judgeScores);
|
|
273
|
+
return scores.length === 0 ? NaN : scores.reduce((sum, s) => sum + s.composite, 0) / scores.length;
|
|
274
|
+
}).filter((v) => Number.isFinite(v));
|
|
275
|
+
if (baselineHoldoutComposites.length >= 3) {
|
|
276
|
+
power = powerPreflight({
|
|
277
|
+
baselineComposites: baselineHoldoutComposites,
|
|
278
|
+
sharedScorerChannel: true
|
|
279
|
+
});
|
|
280
|
+
if (opts.onProgress) opts.onProgress({
|
|
281
|
+
kind: "power.estimated",
|
|
282
|
+
n: power.n,
|
|
283
|
+
sd: power.sd,
|
|
284
|
+
mde: power.mde,
|
|
285
|
+
underpowered: power.underpowered
|
|
286
|
+
});
|
|
287
|
+
if (power.underpowered && generations > 0) console.warn(`[selfImprove] ${power.recommendation}`);
|
|
288
|
+
}
|
|
289
|
+
if (opts.onProgress) {
|
|
290
|
+
opts.onProgress({
|
|
291
|
+
kind: "baseline.completed",
|
|
292
|
+
compositeMean: baseline.compositeMean,
|
|
293
|
+
durationMs: Date.now() - startedAt
|
|
294
|
+
});
|
|
295
|
+
opts.onProgress({
|
|
296
|
+
kind: "gate.decided",
|
|
297
|
+
decision: result.gateResult.decision,
|
|
298
|
+
...holdoutDeferred ? {} : { lift: winnerStats.compositeMean - baseline.compositeMean }
|
|
299
|
+
});
|
|
300
|
+
}
|
|
301
|
+
const cost = result.cost;
|
|
302
|
+
const totalCost = cost.totalCostUsd;
|
|
303
|
+
const insight = await analyzeRuns({
|
|
304
|
+
runs: [...cellsToRunRecords(reportBaselineCampaign.cells, "baseline", runDir, opts.baselineSurface, reportSplit, opts.model), ...cellsToRunRecords(reportWinnerCampaign.cells, "winner", runDir, result.winnerSurface, reportSplit, opts.model)],
|
|
305
|
+
baselineCandidateId: "baseline",
|
|
306
|
+
candidateCandidateId: "winner"
|
|
307
|
+
});
|
|
308
|
+
const durationMs = Date.now() - startedAt;
|
|
309
|
+
const { record: provenance } = await emitLoopProvenance({
|
|
310
|
+
...loopProvenanceArgsFromResult({
|
|
311
|
+
runId: `${runDir}#${startedAt}`,
|
|
312
|
+
runDir,
|
|
313
|
+
timestamp: new Date(startedAt).toISOString(),
|
|
314
|
+
baselineSurface: opts.baselineSurface,
|
|
315
|
+
result,
|
|
316
|
+
costReceipts: costLedger.list(),
|
|
317
|
+
totalCostUsd: totalCost,
|
|
318
|
+
totalDurationMs: durationMs
|
|
319
|
+
}),
|
|
320
|
+
...optimizationResult ? { optimizationMethod: {
|
|
321
|
+
name: opts.method.name,
|
|
322
|
+
cost: structuredClone(optimizationResult.cost),
|
|
323
|
+
...optimizationResult.durationMs === void 0 ? {} : { durationMs: optimizationResult.durationMs },
|
|
324
|
+
...optimizationResult.provenance === void 0 ? {} : { provenance: structuredClone(optimizationResult.provenance) }
|
|
325
|
+
} } : {},
|
|
326
|
+
storage,
|
|
327
|
+
hostedClient: opts.hostedTenant ? createHostedClient(opts.hostedTenant) : void 0
|
|
328
|
+
});
|
|
329
|
+
if (opts.onProvenance) opts.onProvenance(provenance);
|
|
330
|
+
const summary = {
|
|
331
|
+
baseline,
|
|
332
|
+
winner: {
|
|
333
|
+
...winnerStats,
|
|
334
|
+
surface: result.winnerSurface,
|
|
335
|
+
...result.winnerLabel ? { label: result.winnerLabel } : {},
|
|
336
|
+
...result.winnerRationale ? { rationale: result.winnerRationale } : {}
|
|
337
|
+
},
|
|
338
|
+
...holdoutDeferred ? {} : { lift: winnerStats.compositeMean - baseline.compositeMean },
|
|
339
|
+
diff: result.promotedDiff,
|
|
340
|
+
provenance,
|
|
341
|
+
gateDecision: result.gateResult.decision,
|
|
342
|
+
generationsExplored: result.generations.length,
|
|
343
|
+
durationMs,
|
|
344
|
+
totalCostUsd: totalCost,
|
|
345
|
+
cost,
|
|
346
|
+
receipts: costLedger.list(),
|
|
347
|
+
...optimizationResult ? { optimization: {
|
|
348
|
+
name: opts.method.name,
|
|
349
|
+
cost: structuredClone(optimizationResult.cost),
|
|
350
|
+
...optimizationResult.durationMs === void 0 ? {} : { durationMs: optimizationResult.durationMs },
|
|
351
|
+
...optimizationResult.provenance === void 0 ? {} : { provenance: structuredClone(optimizationResult.provenance) }
|
|
352
|
+
} } : {},
|
|
353
|
+
insight,
|
|
354
|
+
...power ? { power } : {},
|
|
355
|
+
raw: result
|
|
356
|
+
};
|
|
357
|
+
if (opts.hostedTenant) try {
|
|
358
|
+
await shipEvalRunToHosted(opts.hostedTenant, opts, summary, result, runDir);
|
|
359
|
+
} catch (err) {
|
|
360
|
+
const msg = err instanceof Error ? err.message : String(err);
|
|
361
|
+
console.warn(`[agent-eval] hosted ingest failed (continuing): ${msg}`);
|
|
362
|
+
}
|
|
363
|
+
return summary;
|
|
489
364
|
}
|
|
490
365
|
async function shipEvalRunToHosted(tenant, opts, summary, raw, runDir) {
|
|
491
|
-
|
|
492
|
-
|
|
493
|
-
|
|
494
|
-
|
|
495
|
-
|
|
496
|
-
|
|
497
|
-
|
|
498
|
-
|
|
499
|
-
|
|
500
|
-
|
|
501
|
-
|
|
502
|
-
|
|
503
|
-
|
|
504
|
-
|
|
505
|
-
|
|
506
|
-
|
|
507
|
-
|
|
508
|
-
|
|
509
|
-
|
|
510
|
-
|
|
511
|
-
|
|
512
|
-
|
|
513
|
-
|
|
514
|
-
|
|
515
|
-
|
|
516
|
-
|
|
517
|
-
|
|
518
|
-
|
|
519
|
-
|
|
520
|
-
|
|
521
|
-
|
|
522
|
-
|
|
523
|
-
|
|
524
|
-
|
|
525
|
-
|
|
526
|
-
|
|
527
|
-
|
|
528
|
-
|
|
529
|
-
|
|
530
|
-
|
|
531
|
-
|
|
532
|
-
|
|
533
|
-
|
|
534
|
-
|
|
535
|
-
|
|
536
|
-
|
|
537
|
-
|
|
538
|
-
|
|
539
|
-
gateDecision: summary.gateDecision,
|
|
540
|
-
holdoutLift: summary.lift,
|
|
541
|
-
totalCostUsd: summary.totalCostUsd,
|
|
542
|
-
totalDurationMs: summary.durationMs,
|
|
543
|
-
insightReport: summary.insight
|
|
544
|
-
};
|
|
545
|
-
await client.ingestEvalRun(event);
|
|
366
|
+
const client = createHostedClient(tenant);
|
|
367
|
+
function snapshotFromCampaign(index, surface, campaign, durationMs) {
|
|
368
|
+
const cells = campaign.cells.map((cell) => {
|
|
369
|
+
const execution = campaignCellExecutionEvidence(cell);
|
|
370
|
+
return {
|
|
371
|
+
scenarioId: cell.scenarioId,
|
|
372
|
+
rep: cell.rep,
|
|
373
|
+
compositeMean: campaignCellTaskScore(cell) ?? null,
|
|
374
|
+
dimensions: campaignCellJudgeDimensions(cell),
|
|
375
|
+
terminalOutcome: execution.terminalOutcome,
|
|
376
|
+
executionErrorCount: execution.executionErrorCount ?? null,
|
|
377
|
+
errorMessage: cell.error ?? void 0
|
|
378
|
+
};
|
|
379
|
+
});
|
|
380
|
+
const scoredCells = cells.flatMap((cell) => cell.compositeMean === null ? [] : [cell.compositeMean]);
|
|
381
|
+
const compositeMean = scoredCells.length === 0 ? null : scoredCells.reduce((sum, score) => sum + score, 0) / scoredCells.length;
|
|
382
|
+
return {
|
|
383
|
+
index,
|
|
384
|
+
surfaceHash: surfaceHash(surface),
|
|
385
|
+
surface,
|
|
386
|
+
cells,
|
|
387
|
+
compositeMean,
|
|
388
|
+
costUsd: campaign.aggregates.cost.totalCostUsd,
|
|
389
|
+
durationMs
|
|
390
|
+
};
|
|
391
|
+
}
|
|
392
|
+
const generations = [];
|
|
393
|
+
generations.push(snapshotFromCampaign(0, opts.baselineSurface, raw.baselineCampaign, 0));
|
|
394
|
+
for (const gen of raw.generations) {
|
|
395
|
+
const winner = gen.surfaces.reduce((best, s) => s.campaign.aggregates.cellsExecuted > 0 && (best === void 0 || averageComposite(s.campaign) > averageComposite(best.campaign)) ? s : best, gen.surfaces[0]);
|
|
396
|
+
if (!winner) continue;
|
|
397
|
+
generations.push(snapshotFromCampaign(gen.record.generationIndex + 1, winner.surface, winner.campaign, 0));
|
|
398
|
+
}
|
|
399
|
+
const event = {
|
|
400
|
+
runId: `${runDir}#${Date.now()}`,
|
|
401
|
+
runDir,
|
|
402
|
+
timestamp: (/* @__PURE__ */ new Date()).toISOString(),
|
|
403
|
+
status: "finished",
|
|
404
|
+
labels: opts.hostedLabels ?? {},
|
|
405
|
+
baseline: generations[0],
|
|
406
|
+
generations,
|
|
407
|
+
gateDecision: summary.gateDecision,
|
|
408
|
+
holdoutLift: summary.lift,
|
|
409
|
+
totalCostUsd: summary.totalCostUsd,
|
|
410
|
+
totalDurationMs: summary.durationMs,
|
|
411
|
+
insightReport: summary.insight
|
|
412
|
+
};
|
|
413
|
+
await client.ingestEvalRun(event);
|
|
546
414
|
}
|
|
547
415
|
function averageComposite(campaign) {
|
|
548
|
-
|
|
549
|
-
|
|
416
|
+
const aggs = Object.values(campaign.aggregates.byScenario);
|
|
417
|
+
return aggs.length === 0 ? 0 : aggs.reduce((s, a) => s + a.meanComposite, 0) / aggs.length;
|
|
550
418
|
}
|
|
551
419
|
function hashString(s) {
|
|
552
|
-
|
|
553
|
-
|
|
554
|
-
|
|
555
|
-
|
|
556
|
-
|
|
557
|
-
|
|
558
|
-
}
|
|
420
|
+
let h = 2166136261;
|
|
421
|
+
for (let i = 0; i < s.length; i++) {
|
|
422
|
+
h ^= s.charCodeAt(i);
|
|
423
|
+
h = Math.imul(h, 16777619) >>> 0;
|
|
424
|
+
}
|
|
425
|
+
return h.toString(16).padStart(8, "0");
|
|
426
|
+
}
|
|
427
|
+
/**
|
|
428
|
+
* Adapt campaign cells into the `RunRecord` shape `analyzeRuns()` consumes.
|
|
429
|
+
* Each cell becomes one run; `candidateId` is the caller-supplied label so
|
|
430
|
+
* baseline + winner pair cleanly on `(experimentId, scenarioId, seed)`.
|
|
431
|
+
*
|
|
432
|
+
* `promptHash` is the REAL sha256 content hash of the surface this cell ran
|
|
433
|
+
* (baseline vs winner are byte-distinguishable + byte-identical-verifiable);
|
|
434
|
+
* `configHash` is the sha256 of the candidate label so the two candidates'
|
|
435
|
+
* config rows differ. Both were previously the literal `'sha256:cell'`, which
|
|
436
|
+
* made baseline and winner indistinguishable in every downstream record.
|
|
437
|
+
*/
|
|
559
438
|
function cellsToRunRecords(cells, candidateId, runId, surface, splitTag, fallbackModel) {
|
|
560
|
-
|
|
561
|
-
|
|
562
|
-
|
|
563
|
-
|
|
564
|
-
|
|
565
|
-
|
|
566
|
-
|
|
567
|
-
|
|
568
|
-
|
|
569
|
-
|
|
570
|
-
|
|
571
|
-
|
|
572
|
-
|
|
573
|
-
|
|
574
|
-
|
|
575
|
-
|
|
576
|
-
|
|
577
|
-
|
|
578
|
-
|
|
579
|
-
|
|
580
|
-
|
|
581
|
-
|
|
582
|
-
|
|
583
|
-
|
|
584
|
-
|
|
585
|
-
|
|
586
|
-
|
|
587
|
-
|
|
588
|
-
|
|
589
|
-
// src/contract/define-agent-eval.ts
|
|
439
|
+
const promptHash = surfaceContentHash(surface);
|
|
440
|
+
const configHash = surfaceContentHash(candidateId);
|
|
441
|
+
return cells.map((cell) => {
|
|
442
|
+
const model = cell.resolvedModel ?? fallbackModel;
|
|
443
|
+
if (!model) throw new ValidationError(`selfImprove.model is required when cell ${cell.cellId} has no paid-call model receipt`);
|
|
444
|
+
if (!modelHasSnapshot(model)) throw new ValidationError(`selfImprove model "${model}" lacks a snapshot version for cell ${cell.cellId}`);
|
|
445
|
+
return campaignCellToRunRecord(cell, {
|
|
446
|
+
runId: `${runId}::${candidateId}::${cell.cellId}`,
|
|
447
|
+
experimentId: runId,
|
|
448
|
+
candidateId,
|
|
449
|
+
seed: cell.rep * 1e6 + hashString(cell.scenarioId).slice(0, 6).split("").reduce((a, c) => a * 31 + c.charCodeAt(0) >>> 0, 0),
|
|
450
|
+
model,
|
|
451
|
+
promptHash,
|
|
452
|
+
configHash,
|
|
453
|
+
commitSha: "cell",
|
|
454
|
+
splitTag
|
|
455
|
+
});
|
|
456
|
+
});
|
|
457
|
+
}
|
|
458
|
+
//#endregion
|
|
459
|
+
//#region src/contract/define-agent-eval.ts
|
|
460
|
+
/**
|
|
461
|
+
* Define an agent eval once, then either score a surface with `evaluate()` or
|
|
462
|
+
* run the closed loop with `improve()`.
|
|
463
|
+
*
|
|
464
|
+
* This is a DX wrapper only: it delegates to `runEval()` and `selfImprove()` and
|
|
465
|
+
* returns their native result shapes.
|
|
466
|
+
*/
|
|
590
467
|
function defineAgentEval(defaults) {
|
|
591
|
-
|
|
592
|
-
|
|
593
|
-
|
|
594
|
-
|
|
595
|
-
|
|
596
|
-
|
|
597
|
-
|
|
598
|
-
|
|
599
|
-
|
|
600
|
-
|
|
601
|
-
|
|
602
|
-
|
|
603
|
-
|
|
604
|
-
|
|
605
|
-
|
|
606
|
-
|
|
607
|
-
|
|
608
|
-
|
|
609
|
-
|
|
610
|
-
|
|
611
|
-
|
|
612
|
-
|
|
613
|
-
|
|
614
|
-
|
|
615
|
-
|
|
616
|
-
|
|
617
|
-
|
|
618
|
-
|
|
619
|
-
|
|
620
|
-
|
|
621
|
-
|
|
622
|
-
|
|
623
|
-
|
|
624
|
-
|
|
625
|
-
...budget ? { budget } : {},
|
|
626
|
-
...hostedTenant ? { hostedTenant } : {}
|
|
627
|
-
});
|
|
628
|
-
}
|
|
629
|
-
};
|
|
468
|
+
const defaultEvaluateOptions = evaluateDefaults(defaults);
|
|
469
|
+
return {
|
|
470
|
+
scenarios: defaults.scenarios,
|
|
471
|
+
baselineSurface: defaults.baselineSurface,
|
|
472
|
+
async evaluate(opts = {}) {
|
|
473
|
+
const { agent, judge, judges, runDir, scenarios, surface, ...campaignOpts } = opts;
|
|
474
|
+
const selectedAgent = agent ?? defaults.agent;
|
|
475
|
+
const selectedSurface = surface ?? defaults.baselineSurface;
|
|
476
|
+
const selectedRunDir = runDir ?? defaults.runDir ?? `mem://defineAgentEval-${Date.now()}`;
|
|
477
|
+
const selectedStorage = campaignOpts.storage ?? defaultEvaluateOptions.storage ?? (selectedRunDir.startsWith("mem://") ? inMemoryCampaignStorage() : void 0);
|
|
478
|
+
const evalOptions = {
|
|
479
|
+
...defaultEvaluateOptions,
|
|
480
|
+
...campaignOpts,
|
|
481
|
+
...selectedStorage ? { storage: selectedStorage } : {},
|
|
482
|
+
runDir: selectedRunDir,
|
|
483
|
+
scenarios: scenarios ?? defaults.scenarios,
|
|
484
|
+
dispatch: (scenario, ctx) => selectedAgent(selectedSurface, scenario, ctx),
|
|
485
|
+
judges: evaluateJudges(judges, judge ?? defaults.judge)
|
|
486
|
+
};
|
|
487
|
+
if (evalOptions.reps !== void 0) evalOptions.reps = requirePositiveInteger(evalOptions.reps, "reps");
|
|
488
|
+
return runEval(evalOptions);
|
|
489
|
+
},
|
|
490
|
+
async improve(opts = {}) {
|
|
491
|
+
const { budget: budgetOverride, hostedTenant: hostedTenantOverride, ...topLevelOverrides } = opts;
|
|
492
|
+
const merged = mergeDefined(defaults, topLevelOverrides);
|
|
493
|
+
const budget = mergeBudget(defaults.budget, budgetOverride);
|
|
494
|
+
const hostedTenant = mergeHostedTenant(defaults.hostedTenant, hostedTenantOverride);
|
|
495
|
+
return selfImprove({
|
|
496
|
+
...merged,
|
|
497
|
+
...budget ? { budget } : {},
|
|
498
|
+
...hostedTenant ? { hostedTenant } : {}
|
|
499
|
+
});
|
|
500
|
+
}
|
|
501
|
+
};
|
|
630
502
|
}
|
|
631
503
|
function evaluateDefaults(defaults) {
|
|
632
|
-
|
|
633
|
-
|
|
634
|
-
|
|
635
|
-
|
|
636
|
-
|
|
637
|
-
|
|
638
|
-
|
|
639
|
-
|
|
640
|
-
|
|
641
|
-
|
|
642
|
-
out.reps = requirePositiveInteger(defaults.budget.reps, "budget.reps");
|
|
643
|
-
return out;
|
|
504
|
+
const out = {};
|
|
505
|
+
if (defaults.storage) out.storage = defaults.storage;
|
|
506
|
+
if (defaults.labeledStore) out.labeledStore = defaults.labeledStore;
|
|
507
|
+
if (defaults.captureSource) out.captureSource = defaults.captureSource;
|
|
508
|
+
if (defaults.cellPlacement) out.cellPlacement = defaults.cellPlacement;
|
|
509
|
+
if (defaults.expectUsage) out.expectUsage = defaults.expectUsage;
|
|
510
|
+
if (defaults.budget?.dollars !== void 0) out.costCeiling = defaults.budget.dollars;
|
|
511
|
+
if (defaults.budget?.maxConcurrency !== void 0) out.maxConcurrency = defaults.budget.maxConcurrency;
|
|
512
|
+
if (defaults.budget?.reps !== void 0) out.reps = requirePositiveInteger(defaults.budget.reps, "budget.reps");
|
|
513
|
+
return out;
|
|
644
514
|
}
|
|
645
515
|
function mergeBudget(defaults, overrides) {
|
|
646
|
-
|
|
647
|
-
|
|
648
|
-
|
|
516
|
+
const merged = mergeOptionalObject(defaults, overrides);
|
|
517
|
+
if (merged?.reps !== void 0) merged.reps = requirePositiveInteger(merged.reps, "budget.reps");
|
|
518
|
+
return merged;
|
|
649
519
|
}
|
|
650
520
|
function mergeHostedTenant(defaults, overrides) {
|
|
651
|
-
|
|
652
|
-
|
|
653
|
-
|
|
654
|
-
|
|
655
|
-
"defineAgentEval.improve: hostedTenant requires endpoint, apiKey, and tenantId after merging defaults and overrides"
|
|
656
|
-
);
|
|
657
|
-
}
|
|
658
|
-
return merged;
|
|
521
|
+
const merged = mergeOptionalObject(defaults, overrides);
|
|
522
|
+
if (!merged) return void 0;
|
|
523
|
+
if (!merged.endpoint?.trim() || !merged.apiKey?.trim() || !merged.tenantId?.trim()) throw new Error("defineAgentEval.improve: hostedTenant requires endpoint, apiKey, and tenantId after merging defaults and overrides");
|
|
524
|
+
return merged;
|
|
659
525
|
}
|
|
660
526
|
function mergeDefined(defaults, overrides) {
|
|
661
|
-
|
|
662
|
-
|
|
663
|
-
|
|
664
|
-
|
|
665
|
-
}
|
|
666
|
-
return merged;
|
|
527
|
+
if (!overrides) return defaults;
|
|
528
|
+
const merged = { ...defaults };
|
|
529
|
+
for (const [key, value] of Object.entries(overrides)) if (value !== void 0) merged[key] = value;
|
|
530
|
+
return merged;
|
|
667
531
|
}
|
|
668
532
|
function mergeOptionalObject(defaults, overrides) {
|
|
669
|
-
|
|
670
|
-
|
|
533
|
+
if (!defaults && !overrides) return void 0;
|
|
534
|
+
return mergeDefined(defaults ?? {}, overrides);
|
|
671
535
|
}
|
|
672
536
|
function evaluateJudges(judges, defaultJudge) {
|
|
673
|
-
|
|
674
|
-
|
|
675
|
-
|
|
676
|
-
|
|
677
|
-
|
|
678
|
-
}
|
|
679
|
-
return [defaultJudge];
|
|
537
|
+
if (judges !== void 0) {
|
|
538
|
+
if (judges.length === 0) throw new Error("defineAgentEval.evaluate: judges must not be empty");
|
|
539
|
+
return judges;
|
|
540
|
+
}
|
|
541
|
+
return [defaultJudge];
|
|
680
542
|
}
|
|
681
543
|
function requirePositiveInteger(value, field) {
|
|
682
|
-
|
|
683
|
-
|
|
684
|
-
}
|
|
685
|
-
return value;
|
|
544
|
+
if (!Number.isInteger(value) || value < 1) throw new Error(`defineAgentEval: ${field} must be a positive integer`);
|
|
545
|
+
return value;
|
|
686
546
|
}
|
|
687
|
-
|
|
688
|
-
|
|
689
|
-
|
|
690
|
-
agentCandidateBenchmarkSuiteSchema,
|
|
691
|
-
agentCandidateBenchmarkTaskSchema,
|
|
692
|
-
agentCandidateBundleSchema,
|
|
693
|
-
agentCandidateEvaluationPolicySchema,
|
|
694
|
-
agentCandidateExperimentSchema,
|
|
695
|
-
agentImprovementMeasuredComparisonSchema,
|
|
696
|
-
candidateExecutionEvidenceSchema,
|
|
697
|
-
canonicalCandidateDigest,
|
|
698
|
-
omitTopLevelDigest
|
|
699
|
-
} from "@tangle-network/agent-interface";
|
|
547
|
+
//#endregion
|
|
548
|
+
//#region src/contract/measured-comparison.ts
|
|
549
|
+
/** Content-address one task before any measured execution can see it. */
|
|
700
550
|
function sealCandidateBenchmarkTask(material) {
|
|
701
|
-
|
|
702
|
-
|
|
703
|
-
|
|
704
|
-
|
|
551
|
+
return agentCandidateBenchmarkTaskSchema.parse({
|
|
552
|
+
...material,
|
|
553
|
+
digest: canonicalCandidateDigest(material)
|
|
554
|
+
});
|
|
705
555
|
}
|
|
556
|
+
/** Freeze task order, repetitions, and every seed before either arm runs. */
|
|
706
557
|
function sealCandidateBenchmarkSuite(options) {
|
|
707
|
-
|
|
708
|
-
|
|
709
|
-
|
|
710
|
-
|
|
711
|
-
|
|
712
|
-
|
|
713
|
-
|
|
714
|
-
|
|
715
|
-
|
|
716
|
-
|
|
717
|
-
|
|
718
|
-
|
|
719
|
-
|
|
720
|
-
|
|
558
|
+
for (const task of options.tasks) verifyCandidateBenchmarkTask(task);
|
|
559
|
+
const material = {
|
|
560
|
+
kind: "agent-candidate-benchmark-suite",
|
|
561
|
+
digestAlgorithm: "rfc8785-sha256",
|
|
562
|
+
taskDigests: options.tasks.map((task) => task.digest),
|
|
563
|
+
reps: options.reps,
|
|
564
|
+
seeds: options.seeds
|
|
565
|
+
};
|
|
566
|
+
return {
|
|
567
|
+
suite: agentCandidateBenchmarkSuiteSchema.parse({
|
|
568
|
+
...material,
|
|
569
|
+
digest: canonicalCandidateDigest(material)
|
|
570
|
+
}),
|
|
571
|
+
tasks: options.tasks
|
|
572
|
+
};
|
|
573
|
+
}
|
|
574
|
+
/** Freeze both complete agent states and their exact held-out work. */
|
|
721
575
|
function sealCandidateExperiment(material) {
|
|
722
|
-
|
|
723
|
-
|
|
724
|
-
|
|
725
|
-
|
|
726
|
-
return verifyCandidateExperiment(parsed);
|
|
576
|
+
return verifyCandidateExperiment(agentCandidateExperimentSchema.parse({
|
|
577
|
+
...material,
|
|
578
|
+
digest: canonicalCandidateDigest(material)
|
|
579
|
+
}));
|
|
727
580
|
}
|
|
728
581
|
function verifyCandidateExperiment(input) {
|
|
729
|
-
|
|
730
|
-
|
|
731
|
-
|
|
732
|
-
|
|
733
|
-
|
|
734
|
-
|
|
735
|
-
|
|
736
|
-
|
|
737
|
-
|
|
738
|
-
}
|
|
582
|
+
const experiment = agentCandidateExperimentSchema.parse(input);
|
|
583
|
+
verifySelfAddressed(experiment, "candidate experiment");
|
|
584
|
+
verifyBundle(experiment.baseline, "baseline bundle");
|
|
585
|
+
verifyBundle(experiment.candidate, "candidate bundle");
|
|
586
|
+
if (experiment.baseline.digest === experiment.candidate.digest) throw new Error("candidate experiment baseline and candidate bundles are identical");
|
|
587
|
+
verifyCandidateBenchmarkSuiteInputs(experiment.benchmark);
|
|
588
|
+
return experiment;
|
|
589
|
+
}
|
|
590
|
+
/** Execute each signed cell for both arms. The callback is Runtime's one executor. */
|
|
739
591
|
async function runCandidateExperiment(options) {
|
|
740
|
-
|
|
741
|
-
|
|
742
|
-
|
|
743
|
-
|
|
744
|
-
|
|
745
|
-
|
|
746
|
-
|
|
747
|
-
|
|
748
|
-
|
|
749
|
-
|
|
750
|
-
|
|
751
|
-
|
|
752
|
-
|
|
753
|
-
|
|
754
|
-
|
|
755
|
-
|
|
756
|
-
|
|
757
|
-
|
|
758
|
-
|
|
759
|
-
|
|
760
|
-
|
|
761
|
-
|
|
762
|
-
|
|
763
|
-
|
|
764
|
-
|
|
765
|
-
|
|
766
|
-
|
|
767
|
-
|
|
768
|
-
|
|
769
|
-
|
|
770
|
-
|
|
771
|
-
|
|
772
|
-
|
|
773
|
-
|
|
774
|
-
|
|
775
|
-
|
|
776
|
-
|
|
777
|
-
|
|
778
|
-
|
|
779
|
-
|
|
780
|
-
|
|
781
|
-
|
|
782
|
-
|
|
783
|
-
|
|
784
|
-
|
|
785
|
-
|
|
786
|
-
|
|
787
|
-
|
|
788
|
-
|
|
789
|
-
|
|
790
|
-
|
|
791
|
-
|
|
792
|
-
|
|
793
|
-
|
|
794
|
-
|
|
795
|
-
|
|
592
|
+
const experiment = verifyCandidateExperiment(options.experiment);
|
|
593
|
+
const { suite, tasks } = experiment.benchmark;
|
|
594
|
+
const maxConcurrency = options.maxConcurrency ?? 2;
|
|
595
|
+
if (!Number.isSafeInteger(maxConcurrency) || maxConcurrency < 1) throw new Error("candidate experiment maxConcurrency must be a positive integer");
|
|
596
|
+
const measurements = new Array(suite.taskDigests.length * suite.reps);
|
|
597
|
+
let nextIndex = 0;
|
|
598
|
+
const lanes = Array.from({ length: Math.min(maxConcurrency, measurements.length) }, async () => {
|
|
599
|
+
while (true) {
|
|
600
|
+
if (options.signal?.aborted) throw abortError(options.signal);
|
|
601
|
+
const index = nextIndex;
|
|
602
|
+
nextIndex += 1;
|
|
603
|
+
if (index >= measurements.length) return;
|
|
604
|
+
const taskIndex = Math.floor(index / suite.reps);
|
|
605
|
+
const repetition = index % suite.reps;
|
|
606
|
+
const task = tasks[taskIndex];
|
|
607
|
+
const seed = suite.seeds[index];
|
|
608
|
+
if (!task || seed === void 0) throw new Error(`candidate experiment cell ${index} has no signed task or seed`);
|
|
609
|
+
const benchmarkCell = {
|
|
610
|
+
suiteDigest: suite.digest,
|
|
611
|
+
taskIndex,
|
|
612
|
+
repetition
|
|
613
|
+
};
|
|
614
|
+
const [baseline, candidate] = await Promise.all([options.execute({
|
|
615
|
+
experiment,
|
|
616
|
+
arm: "baseline",
|
|
617
|
+
bundle: experiment.baseline,
|
|
618
|
+
task,
|
|
619
|
+
benchmarkCell,
|
|
620
|
+
seed,
|
|
621
|
+
...options.signal ? { signal: options.signal } : {}
|
|
622
|
+
}), options.execute({
|
|
623
|
+
experiment,
|
|
624
|
+
arm: "candidate",
|
|
625
|
+
bundle: experiment.candidate,
|
|
626
|
+
task,
|
|
627
|
+
benchmarkCell,
|
|
628
|
+
seed,
|
|
629
|
+
...options.signal ? { signal: options.signal } : {}
|
|
630
|
+
})]);
|
|
631
|
+
const measurement = {
|
|
632
|
+
baseline,
|
|
633
|
+
candidate
|
|
634
|
+
};
|
|
635
|
+
verifyMeasurement(experiment, measurement, index);
|
|
636
|
+
measurements[index] = measurement;
|
|
637
|
+
}
|
|
638
|
+
});
|
|
639
|
+
await Promise.all(lanes);
|
|
640
|
+
return measurements;
|
|
641
|
+
}
|
|
642
|
+
/**
|
|
643
|
+
* Calculate the shared paired decision from any complete receipt shape.
|
|
644
|
+
*
|
|
645
|
+
* Callers still own sealing their tasks, verifying each receipt against its
|
|
646
|
+
* expected arm and state, and proving every expected cell exists. This function
|
|
647
|
+
* only validates the projected measurements and derives their shared decision.
|
|
648
|
+
*/
|
|
796
649
|
function evaluatePairedMeasurements(options) {
|
|
797
|
-
|
|
798
|
-
|
|
799
|
-
|
|
800
|
-
|
|
801
|
-
|
|
802
|
-
|
|
803
|
-
|
|
804
|
-
|
|
805
|
-
|
|
806
|
-
|
|
807
|
-
|
|
808
|
-
|
|
809
|
-
|
|
810
|
-
|
|
811
|
-
|
|
812
|
-
|
|
813
|
-
|
|
814
|
-
|
|
815
|
-
|
|
816
|
-
|
|
817
|
-
|
|
818
|
-
|
|
819
|
-
|
|
820
|
-
|
|
821
|
-
|
|
822
|
-
|
|
823
|
-
|
|
824
|
-
|
|
825
|
-
|
|
826
|
-
|
|
827
|
-
|
|
828
|
-
|
|
829
|
-
|
|
830
|
-
|
|
831
|
-
|
|
832
|
-
|
|
833
|
-
|
|
834
|
-
|
|
835
|
-
|
|
836
|
-
|
|
837
|
-
|
|
838
|
-
|
|
839
|
-
|
|
840
|
-
|
|
841
|
-
|
|
842
|
-
|
|
843
|
-
|
|
844
|
-
|
|
845
|
-
|
|
846
|
-
|
|
847
|
-
|
|
848
|
-
|
|
849
|
-
|
|
850
|
-
|
|
851
|
-
|
|
852
|
-
|
|
853
|
-
|
|
854
|
-
|
|
855
|
-
|
|
856
|
-
|
|
857
|
-
|
|
858
|
-
|
|
859
|
-
|
|
860
|
-
|
|
861
|
-
|
|
862
|
-
|
|
863
|
-
|
|
864
|
-
|
|
865
|
-
|
|
866
|
-
|
|
867
|
-
|
|
868
|
-
|
|
869
|
-
|
|
870
|
-
|
|
871
|
-
|
|
872
|
-
|
|
873
|
-
|
|
874
|
-
|
|
875
|
-
|
|
876
|
-
|
|
877
|
-
|
|
878
|
-
|
|
879
|
-
|
|
880
|
-
|
|
881
|
-
|
|
882
|
-
|
|
883
|
-
|
|
884
|
-
|
|
885
|
-
|
|
886
|
-
|
|
887
|
-
|
|
888
|
-
|
|
889
|
-
|
|
890
|
-
|
|
891
|
-
|
|
892
|
-
|
|
893
|
-
|
|
894
|
-
|
|
895
|
-
|
|
896
|
-
|
|
897
|
-
|
|
898
|
-
|
|
899
|
-
|
|
900
|
-
|
|
901
|
-
|
|
902
|
-
|
|
903
|
-
|
|
904
|
-
|
|
905
|
-
|
|
906
|
-
|
|
907
|
-
|
|
908
|
-
|
|
909
|
-
|
|
910
|
-
|
|
911
|
-
|
|
912
|
-
|
|
913
|
-
|
|
914
|
-
|
|
915
|
-
|
|
916
|
-
|
|
917
|
-
|
|
918
|
-
|
|
919
|
-
|
|
920
|
-
|
|
921
|
-
|
|
922
|
-
|
|
923
|
-
|
|
924
|
-
|
|
925
|
-
|
|
926
|
-
|
|
927
|
-
|
|
928
|
-
|
|
929
|
-
|
|
930
|
-
|
|
931
|
-
|
|
932
|
-
|
|
933
|
-
|
|
934
|
-
|
|
935
|
-
|
|
936
|
-
|
|
937
|
-
|
|
938
|
-
|
|
939
|
-
|
|
940
|
-
|
|
941
|
-
|
|
942
|
-
|
|
943
|
-
|
|
944
|
-
|
|
945
|
-
|
|
946
|
-
|
|
947
|
-
|
|
948
|
-
|
|
949
|
-
|
|
950
|
-
|
|
951
|
-
|
|
952
|
-
significance.fewRuns ? `only ${significance.n} paired runs; ${minProductiveRuns} required` : `paired interval lower bound ${significance.bootstrap.low} did not clear ${deltaThreshold}`
|
|
953
|
-
],
|
|
954
|
-
...powerSufficient ? [] : [power?.recommendation ?? `need at least ${Math.max(3, minProductiveRuns)} paired runs`],
|
|
955
|
-
...regressions.length === 0 ? [] : [`critical dimensions regressed: ${regressions.map((entry) => entry.name).join(", ")}`],
|
|
956
|
-
...missingCriticalDimensions.length === 0 ? [] : [`critical dimensions missing: ${missingCriticalDimensions.join(", ")}`],
|
|
957
|
-
...incompleteRuns.length === 0 ? [] : [`${incompleteRuns.length} benchmark executions did not exit successfully`],
|
|
958
|
-
...failedCandidateResults.length === 0 ? [] : [`candidate failed ${failedCandidateResults.length} benchmark tasks`],
|
|
959
|
-
...budgetPassed ? [] : [`total cost ${totalCostUsd} exceeded budget ${budgetUsd}`]
|
|
960
|
-
];
|
|
961
|
-
return {
|
|
962
|
-
overall: {
|
|
963
|
-
name: "composite",
|
|
964
|
-
direction: "higher-is-better",
|
|
965
|
-
unit: "score",
|
|
966
|
-
...overall
|
|
967
|
-
},
|
|
968
|
-
objectives,
|
|
969
|
-
decision: {
|
|
970
|
-
outcome: shipped ? "ship" : significance.fewRuns || !powerSufficient ? "need_more_work" : "hold",
|
|
971
|
-
reasons: reasons.length > 0 ? reasons : ["all measured checks passed"],
|
|
972
|
-
contributingChecks: checks
|
|
973
|
-
},
|
|
974
|
-
power: {
|
|
975
|
-
sufficient: powerSufficient,
|
|
976
|
-
n: baselineScores.length,
|
|
977
|
-
minimumDetectableDelta: power?.mde ?? 1,
|
|
978
|
-
confidenceLevel: confidence,
|
|
979
|
-
scaleAssumed: power?.scaleAssumed ?? true,
|
|
980
|
-
sharedScorerChannel: options.sharedScorerChannel,
|
|
981
|
-
reason: power?.recommendation ?? `need at least ${Math.max(3, minProductiveRuns)} paired runs`
|
|
982
|
-
},
|
|
983
|
-
executionCostUsd,
|
|
984
|
-
totalCostUsd,
|
|
985
|
-
executionDurationMs
|
|
986
|
-
};
|
|
987
|
-
}
|
|
650
|
+
if (options.measurements.length === 0) throw new Error("paired measurement evaluation requires at least one paired cell");
|
|
651
|
+
const additionalCostUsd = options.additionalCostUsd ?? 0;
|
|
652
|
+
if (!Number.isFinite(additionalCostUsd) || additionalCostUsd < 0) throw new Error("paired measurement evaluation additionalCostUsd must be a non-negative number");
|
|
653
|
+
if (typeof options.sharedScorerChannel !== "boolean") throw new Error("paired measurement evaluation sharedScorerChannel must be a boolean");
|
|
654
|
+
const policy = agentCandidateEvaluationPolicySchema.parse(options.policy);
|
|
655
|
+
const measurements = options.measurements.map((measurement, index) => projectPairedMeasurement(measurement, index, options.adapter));
|
|
656
|
+
const cellIds = measurements.map((measurement) => measurement.cellId);
|
|
657
|
+
if (new Set(cellIds).size !== cellIds.length) throw new Error("paired measurement evaluation cell ids must be unique");
|
|
658
|
+
const dimensions = sharedProjectedDimensions(measurements);
|
|
659
|
+
const baselineScores = measurements.map((measurement) => measurement.baseline.score);
|
|
660
|
+
const candidateScores = measurements.map((measurement) => measurement.candidate.score);
|
|
661
|
+
const { confidenceLevel: confidence, resamples, bootstrapSeed, deltaThreshold, minProductiveRuns, budgetUsd, criticalDimensions, regressionTolerance } = policy;
|
|
662
|
+
const significance = heldoutSignificance({
|
|
663
|
+
before: baselineScores,
|
|
664
|
+
after: candidateScores,
|
|
665
|
+
cellIds
|
|
666
|
+
}, {
|
|
667
|
+
confidence,
|
|
668
|
+
resamples,
|
|
669
|
+
seed: bootstrapSeed,
|
|
670
|
+
statistic: "mean",
|
|
671
|
+
deltaThreshold,
|
|
672
|
+
minProductiveRuns
|
|
673
|
+
});
|
|
674
|
+
const overall = measuredEstimate(baselineScores, candidateScores, {
|
|
675
|
+
confidence,
|
|
676
|
+
resamples,
|
|
677
|
+
seed: bootstrapSeed
|
|
678
|
+
});
|
|
679
|
+
const objectives = [{
|
|
680
|
+
kind: "objective",
|
|
681
|
+
name: "benchmark-score",
|
|
682
|
+
direction: "higher-is-better",
|
|
683
|
+
unit: "score",
|
|
684
|
+
availability: "measured",
|
|
685
|
+
...overall
|
|
686
|
+
}, ...dimensions.map((name, index) => ({
|
|
687
|
+
kind: "dimension",
|
|
688
|
+
objective: "benchmark-score",
|
|
689
|
+
name,
|
|
690
|
+
direction: "higher-is-better",
|
|
691
|
+
unit: "score",
|
|
692
|
+
availability: "measured",
|
|
693
|
+
...measuredEstimate(measurements.map((measurement) => dimensionScore(measurement.baseline, name)), measurements.map((measurement) => dimensionScore(measurement.candidate, name)), {
|
|
694
|
+
confidence,
|
|
695
|
+
resamples,
|
|
696
|
+
seed: bootstrapSeed + index + 1
|
|
697
|
+
})
|
|
698
|
+
}))];
|
|
699
|
+
const cost = measuredEstimate(measurements.map((measurement) => measurement.baseline.costUsd), measurements.map((measurement) => measurement.candidate.costUsd), {
|
|
700
|
+
confidence,
|
|
701
|
+
resamples,
|
|
702
|
+
seed: bootstrapSeed + dimensions.length + 1
|
|
703
|
+
});
|
|
704
|
+
const latency = measuredEstimate(measurements.map((measurement) => measurement.baseline.latencyMs), measurements.map((measurement) => measurement.candidate.latencyMs), {
|
|
705
|
+
confidence,
|
|
706
|
+
resamples,
|
|
707
|
+
seed: bootstrapSeed + dimensions.length + 2
|
|
708
|
+
});
|
|
709
|
+
objectives.push({
|
|
710
|
+
kind: "cost",
|
|
711
|
+
name: "cost",
|
|
712
|
+
direction: "lower-is-better",
|
|
713
|
+
unit: "usd",
|
|
714
|
+
availability: "measured",
|
|
715
|
+
...cost
|
|
716
|
+
}, {
|
|
717
|
+
kind: "latency",
|
|
718
|
+
name: "latency",
|
|
719
|
+
direction: "lower-is-better",
|
|
720
|
+
unit: "milliseconds",
|
|
721
|
+
availability: "measured",
|
|
722
|
+
...latency
|
|
723
|
+
});
|
|
724
|
+
const power = baselineScores.length >= 3 ? powerPreflight({
|
|
725
|
+
baselineComposites: baselineScores,
|
|
726
|
+
pairedN: baselineScores.length,
|
|
727
|
+
deltaThreshold,
|
|
728
|
+
confidence,
|
|
729
|
+
sharedScorerChannel: options.sharedScorerChannel
|
|
730
|
+
}) : void 0;
|
|
731
|
+
const powerSufficient = baselineScores.length >= minProductiveRuns && power !== void 0 && !power.underpowered;
|
|
732
|
+
const guardedDimensions = new Set(criticalDimensions);
|
|
733
|
+
const missingCriticalDimensions = criticalDimensions.filter((dimension) => !dimensions.includes(dimension));
|
|
734
|
+
const regressions = objectives.filter((objective) => objective.kind === "dimension" && guardedDimensions.has(objective.name) && objective.availability === "measured" && objective.confidenceInterval.lower < -regressionTolerance);
|
|
735
|
+
const executionCostUsd = measurements.reduce((sum, measurement) => sum + measurement.baseline.costUsd + measurement.candidate.costUsd, 0);
|
|
736
|
+
const executionDurationMs = measurements.reduce((sum, measurement) => sum + measurement.baseline.latencyMs + measurement.candidate.latencyMs, 0);
|
|
737
|
+
const incompleteRuns = measurements.flatMap((measurement) => [measurement.baseline, measurement.candidate]).filter((run) => !run.completed);
|
|
738
|
+
const failedCandidateResults = measurements.filter((measurement) => !measurement.candidate.passed);
|
|
739
|
+
const totalCostUsd = executionCostUsd + additionalCostUsd;
|
|
740
|
+
const budgetPassed = budgetUsd === void 0 || totalCostUsd <= budgetUsd;
|
|
741
|
+
const checks = [
|
|
742
|
+
{
|
|
743
|
+
name: "paired-significance",
|
|
744
|
+
passed: significance.significant
|
|
745
|
+
},
|
|
746
|
+
{
|
|
747
|
+
name: "statistical-power",
|
|
748
|
+
passed: powerSufficient
|
|
749
|
+
},
|
|
750
|
+
{
|
|
751
|
+
name: "all-runs-completed",
|
|
752
|
+
passed: incompleteRuns.length === 0
|
|
753
|
+
},
|
|
754
|
+
{
|
|
755
|
+
name: "candidate-task-pass",
|
|
756
|
+
passed: failedCandidateResults.length === 0
|
|
757
|
+
},
|
|
758
|
+
{
|
|
759
|
+
name: "critical-dimensions",
|
|
760
|
+
passed: regressions.length === 0 && missingCriticalDimensions.length === 0
|
|
761
|
+
},
|
|
762
|
+
{
|
|
763
|
+
name: "budget",
|
|
764
|
+
passed: budgetPassed
|
|
765
|
+
}
|
|
766
|
+
];
|
|
767
|
+
const shipped = checks.every((check) => check.passed);
|
|
768
|
+
const reasons = [
|
|
769
|
+
...significance.significant ? [] : [significance.fewRuns ? `only ${significance.n} paired runs; ${minProductiveRuns} required` : `paired interval lower bound ${significance.bootstrap.low} did not clear ${deltaThreshold}`],
|
|
770
|
+
...powerSufficient ? [] : [power?.recommendation ?? `need at least ${Math.max(3, minProductiveRuns)} paired runs`],
|
|
771
|
+
...regressions.length === 0 ? [] : [`critical dimensions regressed: ${regressions.map((entry) => entry.name).join(", ")}`],
|
|
772
|
+
...missingCriticalDimensions.length === 0 ? [] : [`critical dimensions missing: ${missingCriticalDimensions.join(", ")}`],
|
|
773
|
+
...incompleteRuns.length === 0 ? [] : [`${incompleteRuns.length} benchmark executions did not exit successfully`],
|
|
774
|
+
...failedCandidateResults.length === 0 ? [] : [`candidate failed ${failedCandidateResults.length} benchmark tasks`],
|
|
775
|
+
...budgetPassed ? [] : [`total cost ${totalCostUsd} exceeded budget ${budgetUsd}`]
|
|
776
|
+
];
|
|
777
|
+
return {
|
|
778
|
+
overall: {
|
|
779
|
+
name: "composite",
|
|
780
|
+
direction: "higher-is-better",
|
|
781
|
+
unit: "score",
|
|
782
|
+
...overall
|
|
783
|
+
},
|
|
784
|
+
objectives,
|
|
785
|
+
decision: {
|
|
786
|
+
outcome: shipped ? "ship" : significance.fewRuns || !powerSufficient ? "need_more_work" : "hold",
|
|
787
|
+
reasons: reasons.length > 0 ? reasons : ["all measured checks passed"],
|
|
788
|
+
contributingChecks: checks
|
|
789
|
+
},
|
|
790
|
+
power: {
|
|
791
|
+
sufficient: powerSufficient,
|
|
792
|
+
n: baselineScores.length,
|
|
793
|
+
minimumDetectableDelta: power?.mde ?? 1,
|
|
794
|
+
confidenceLevel: confidence,
|
|
795
|
+
scaleAssumed: power?.scaleAssumed ?? true,
|
|
796
|
+
sharedScorerChannel: options.sharedScorerChannel,
|
|
797
|
+
reason: power?.recommendation ?? `need at least ${Math.max(3, minProductiveRuns)} paired runs`
|
|
798
|
+
},
|
|
799
|
+
executionCostUsd,
|
|
800
|
+
totalCostUsd,
|
|
801
|
+
executionDurationMs
|
|
802
|
+
};
|
|
803
|
+
}
|
|
804
|
+
/** Build the only publishable comparison: paired statistics over Runtime receipts. */
|
|
988
805
|
function measuredComparisonFromCandidateExperiment(options) {
|
|
989
|
-
|
|
990
|
-
|
|
991
|
-
|
|
992
|
-
|
|
993
|
-
|
|
994
|
-
|
|
995
|
-
|
|
996
|
-
|
|
997
|
-
|
|
998
|
-
|
|
999
|
-
|
|
1000
|
-
|
|
1001
|
-
|
|
1002
|
-
|
|
1003
|
-
|
|
1004
|
-
|
|
1005
|
-
|
|
1006
|
-
|
|
1007
|
-
|
|
1008
|
-
|
|
1009
|
-
|
|
1010
|
-
|
|
1011
|
-
|
|
1012
|
-
|
|
1013
|
-
|
|
1014
|
-
|
|
1015
|
-
|
|
1016
|
-
|
|
1017
|
-
|
|
1018
|
-
|
|
1019
|
-
|
|
1020
|
-
|
|
1021
|
-
|
|
1022
|
-
|
|
1023
|
-
|
|
1024
|
-
|
|
1025
|
-
|
|
1026
|
-
|
|
1027
|
-
|
|
1028
|
-
|
|
1029
|
-
|
|
1030
|
-
|
|
1031
|
-
|
|
1032
|
-
|
|
1033
|
-
|
|
1034
|
-
|
|
1035
|
-
|
|
1036
|
-
|
|
1037
|
-
|
|
1038
|
-
|
|
1039
|
-
|
|
1040
|
-
|
|
1041
|
-
|
|
1042
|
-
|
|
1043
|
-
|
|
1044
|
-
|
|
1045
|
-
|
|
1046
|
-
|
|
1047
|
-
|
|
1048
|
-
|
|
1049
|
-
|
|
1050
|
-
|
|
1051
|
-
|
|
1052
|
-
});
|
|
1053
|
-
}
|
|
806
|
+
const experiment = verifyCandidateExperiment(options.experiment);
|
|
807
|
+
const measurements = options.measurements.map((measurement, index) => verifyMeasurement(experiment, measurement, index));
|
|
808
|
+
const expectedN = experiment.benchmark.suite.taskDigests.length * experiment.benchmark.suite.reps;
|
|
809
|
+
if (measurements.length !== expectedN) throw new Error(`candidate experiment is incomplete (${measurements.length}/${expectedN} paired cells)`);
|
|
810
|
+
verifyStableProfileMaterialization(measurements);
|
|
811
|
+
if (!options.runId.trim()) throw new Error("candidate experiment runId is required");
|
|
812
|
+
const searchCostUsd = options.searchCostUsd ?? 0;
|
|
813
|
+
const evaluation = evaluatePairedMeasurements({
|
|
814
|
+
measurements: measurements.map((measurement, index) => ({
|
|
815
|
+
cellId: cellIds(experiment)[index],
|
|
816
|
+
...measurement
|
|
817
|
+
})),
|
|
818
|
+
policy: experiment.policy,
|
|
819
|
+
adapter: candidateExecutionEvidenceAdapter,
|
|
820
|
+
sharedScorerChannel: true,
|
|
821
|
+
additionalCostUsd: searchCostUsd
|
|
822
|
+
});
|
|
823
|
+
const diff = deriveCandidateBundleDiff(experiment);
|
|
824
|
+
const searchDurationMs = options.searchDurationMs ?? 0;
|
|
825
|
+
const totalCostUsd = evaluation.totalCostUsd;
|
|
826
|
+
const durationMs = evaluation.executionDurationMs + searchDurationMs;
|
|
827
|
+
const provisional = agentImprovementMeasuredComparisonSchema.parse({
|
|
828
|
+
kind: "agent-improvement-measured-comparison",
|
|
829
|
+
experiment,
|
|
830
|
+
measurements,
|
|
831
|
+
overall: evaluation.overall,
|
|
832
|
+
objectives: evaluation.objectives,
|
|
833
|
+
...options.candidate ? { candidate: options.candidate } : {},
|
|
834
|
+
decision: evaluation.decision,
|
|
835
|
+
power: evaluation.power,
|
|
836
|
+
provenance: {
|
|
837
|
+
kind: "agent-eval-loop",
|
|
838
|
+
schema: "agent-candidate-experiment",
|
|
839
|
+
runId: options.runId,
|
|
840
|
+
recordDigest: canonicalCandidateDigest({}),
|
|
841
|
+
baselineContentHash: experiment.baseline.digest,
|
|
842
|
+
candidateContentHash: experiment.candidate.digest
|
|
843
|
+
},
|
|
844
|
+
diff,
|
|
845
|
+
evaluation: {
|
|
846
|
+
generationsExplored: options.generationsExplored ?? 0,
|
|
847
|
+
searchDurationMs,
|
|
848
|
+
executionDurationMs: evaluation.executionDurationMs,
|
|
849
|
+
durationMs,
|
|
850
|
+
searchCostUsd,
|
|
851
|
+
executionCostUsd: evaluation.executionCostUsd,
|
|
852
|
+
totalCostUsd
|
|
853
|
+
},
|
|
854
|
+
...options.metadata ? { metadata: options.metadata } : {}
|
|
855
|
+
});
|
|
856
|
+
const { recordDigest: _recordDigest, ...provenance } = provisional.provenance;
|
|
857
|
+
return agentImprovementMeasuredComparisonSchema.parse({
|
|
858
|
+
...provisional,
|
|
859
|
+
provenance: {
|
|
860
|
+
...provenance,
|
|
861
|
+
recordDigest: canonicalCandidateDigest({
|
|
862
|
+
...provisional,
|
|
863
|
+
provenance
|
|
864
|
+
})
|
|
865
|
+
}
|
|
866
|
+
});
|
|
867
|
+
}
|
|
868
|
+
/** Recompute every statistic and decision from the signed experiment receipts. */
|
|
1054
869
|
function verifyCandidateExperimentComparison(input) {
|
|
1055
|
-
|
|
1056
|
-
|
|
1057
|
-
|
|
1058
|
-
|
|
1059
|
-
|
|
1060
|
-
|
|
1061
|
-
|
|
1062
|
-
|
|
1063
|
-
|
|
1064
|
-
|
|
1065
|
-
|
|
1066
|
-
|
|
1067
|
-
throw new Error("candidate experiment comparison does not match its Runtime receipts");
|
|
1068
|
-
}
|
|
1069
|
-
return comparison;
|
|
870
|
+
const comparison = agentImprovementMeasuredComparisonSchema.parse(input);
|
|
871
|
+
if (canonicalCandidateDigest(measuredComparisonFromCandidateExperiment({
|
|
872
|
+
experiment: comparison.experiment,
|
|
873
|
+
measurements: comparison.measurements,
|
|
874
|
+
runId: comparison.provenance.runId,
|
|
875
|
+
...comparison.candidate ? { candidate: comparison.candidate } : {},
|
|
876
|
+
generationsExplored: comparison.evaluation.generationsExplored,
|
|
877
|
+
searchDurationMs: comparison.evaluation.searchDurationMs,
|
|
878
|
+
searchCostUsd: comparison.evaluation.searchCostUsd,
|
|
879
|
+
...comparison.metadata ? { metadata: comparison.metadata } : {}
|
|
880
|
+
})) !== canonicalCandidateDigest(comparison)) throw new Error("candidate experiment comparison does not match its Runtime receipts");
|
|
881
|
+
return comparison;
|
|
1070
882
|
}
|
|
1071
883
|
function deriveCandidateBundleDiff(experiment) {
|
|
1072
|
-
|
|
1073
|
-
|
|
1074
|
-
|
|
1075
|
-
|
|
1076
|
-
|
|
1077
|
-
|
|
1078
|
-
|
|
1079
|
-
|
|
1080
|
-
|
|
1081
|
-
|
|
1082
|
-
|
|
1083
|
-
|
|
1084
|
-
|
|
1085
|
-
|
|
1086
|
-
|
|
1087
|
-
|
|
1088
|
-
|
|
1089
|
-
|
|
1090
|
-
|
|
884
|
+
const changed = [
|
|
885
|
+
"profile",
|
|
886
|
+
"code",
|
|
887
|
+
"execution",
|
|
888
|
+
"knowledge",
|
|
889
|
+
"memory"
|
|
890
|
+
].flatMap((surface) => {
|
|
891
|
+
const baseline = experiment.baseline[surface] ?? null;
|
|
892
|
+
const candidate = experiment.candidate[surface] ?? null;
|
|
893
|
+
const baselineDigest = canonicalCandidateDigest(baseline);
|
|
894
|
+
const candidateDigest = canonicalCandidateDigest(candidate);
|
|
895
|
+
if (baselineDigest === candidateDigest) return [];
|
|
896
|
+
return [[
|
|
897
|
+
`--- baseline/${surface} (${baselineDigest})`,
|
|
898
|
+
`+++ candidate/${surface} (${candidateDigest})`,
|
|
899
|
+
JSON.stringify({
|
|
900
|
+
baseline,
|
|
901
|
+
candidate
|
|
902
|
+
}, null, 2)
|
|
903
|
+
].join("\n")];
|
|
904
|
+
});
|
|
905
|
+
if (changed.length === 0) throw new Error("candidate experiment has no changed candidate surface");
|
|
906
|
+
return changed.join("\n\n");
|
|
1091
907
|
}
|
|
1092
908
|
function verifyCandidateBenchmarkTask(input) {
|
|
1093
|
-
|
|
1094
|
-
|
|
1095
|
-
|
|
909
|
+
const task = agentCandidateBenchmarkTaskSchema.parse(input);
|
|
910
|
+
verifySelfAddressed(task, "candidate benchmark task");
|
|
911
|
+
return task;
|
|
1096
912
|
}
|
|
1097
913
|
function verifyCandidateBenchmarkSuiteInputs(input) {
|
|
1098
|
-
|
|
1099
|
-
|
|
1100
|
-
|
|
1101
|
-
|
|
1102
|
-
|
|
1103
|
-
|
|
1104
|
-
|
|
1105
|
-
|
|
1106
|
-
|
|
1107
|
-
|
|
1108
|
-
|
|
1109
|
-
throw new Error(`candidate benchmark task ${index} does not match the signed suite`);
|
|
1110
|
-
}
|
|
1111
|
-
});
|
|
1112
|
-
return { suite, tasks: candidate.tasks };
|
|
914
|
+
if (input === null || typeof input !== "object" || Array.isArray(input)) throw new Error("candidate benchmark suite inputs must be an object");
|
|
915
|
+
const candidate = input;
|
|
916
|
+
const suite = verifyCandidateBenchmarkSuite(candidate.suite);
|
|
917
|
+
if (!Array.isArray(candidate.tasks) || candidate.tasks.length !== suite.taskDigests.length) throw new Error("candidate benchmark suite task count does not match its signed digests");
|
|
918
|
+
candidate.tasks.forEach((task, index) => {
|
|
919
|
+
if (verifyCandidateBenchmarkTask(task).digest !== suite.taskDigests[index]) throw new Error(`candidate benchmark task ${index} does not match the signed suite`);
|
|
920
|
+
});
|
|
921
|
+
return {
|
|
922
|
+
suite,
|
|
923
|
+
tasks: candidate.tasks
|
|
924
|
+
};
|
|
1113
925
|
}
|
|
1114
926
|
function verifyCandidateBenchmarkSuite(input) {
|
|
1115
|
-
|
|
1116
|
-
|
|
1117
|
-
|
|
927
|
+
const suite = agentCandidateBenchmarkSuiteSchema.parse(input);
|
|
928
|
+
verifySelfAddressed(suite, "candidate benchmark suite");
|
|
929
|
+
return suite;
|
|
1118
930
|
}
|
|
1119
931
|
function verifyBundle(input, label) {
|
|
1120
|
-
|
|
1121
|
-
|
|
1122
|
-
|
|
932
|
+
const bundle = agentCandidateBundleSchema.parse(input);
|
|
933
|
+
verifySelfAddressed(bundle, label);
|
|
934
|
+
return bundle;
|
|
1123
935
|
}
|
|
1124
936
|
function verifyMeasurement(experiment, input, index) {
|
|
1125
|
-
|
|
1126
|
-
|
|
1127
|
-
|
|
1128
|
-
|
|
1129
|
-
|
|
1130
|
-
|
|
1131
|
-
|
|
1132
|
-
|
|
1133
|
-
|
|
1134
|
-
|
|
1135
|
-
|
|
1136
|
-
|
|
1137
|
-
|
|
1138
|
-
|
|
1139
|
-
|
|
1140
|
-
|
|
1141
|
-
|
|
1142
|
-
|
|
1143
|
-
|
|
1144
|
-
|
|
1145
|
-
|
|
1146
|
-
|
|
1147
|
-
|
|
1148
|
-
}
|
|
1149
|
-
const baselinePlan = baseline.materializationReceipt.executionPlan.material;
|
|
1150
|
-
const candidatePlan = candidate.materializationReceipt.executionPlan.material;
|
|
1151
|
-
if (baselinePlan.executionId === candidatePlan.executionId || baselinePlan.runCell.digest === candidatePlan.runCell.digest || baseline.materializationReceipt.digest === candidate.materializationReceipt.digest || baseline.receipt.digest === candidate.receipt.digest || baseline.digest === candidate.digest) {
|
|
1152
|
-
throw new Error(`candidate experiment measurement ${index} reused one execution across arms`);
|
|
1153
|
-
}
|
|
1154
|
-
return { baseline, candidate };
|
|
937
|
+
const suite = experiment.benchmark.suite;
|
|
938
|
+
const taskIndex = Math.floor(index / suite.reps);
|
|
939
|
+
const repetition = index % suite.reps;
|
|
940
|
+
const task = experiment.benchmark.tasks[taskIndex];
|
|
941
|
+
const seed = suite.seeds[index];
|
|
942
|
+
if (!task || seed === void 0) throw new Error(`candidate experiment measurement ${index} is outside the signed suite`);
|
|
943
|
+
const baseline = verifyExecutionEvidence(input.baseline);
|
|
944
|
+
const candidate = verifyExecutionEvidence(input.candidate);
|
|
945
|
+
for (const [arm, evidence] of [["baseline", baseline], ["candidate", candidate]]) {
|
|
946
|
+
const bundle = experiment[arm];
|
|
947
|
+
const materialization = evidence.materializationReceipt;
|
|
948
|
+
const runCell = materialization.executionPlan.material.runCell;
|
|
949
|
+
verifySelfAddressed(runCell, "candidate run cell");
|
|
950
|
+
if (runCell.experimentDigest !== experiment.digest || runCell.arm !== arm || runCell.bundleDigest !== bundle.digest || runCell.suiteDigest !== suite.digest || runCell.taskDigest !== task.digest || runCell.taskIndex !== taskIndex || runCell.repetition !== repetition || runCell.seed !== seed || runCell.attempt > task.attempt.maxAttempts || materialization.bundleDigest !== bundle.digest || materialization.benchmark.suite.digest !== suite.digest || materialization.benchmark.task.digest !== task.digest || materialization.codeKind !== bundle.code.kind || materialization.profileActivation.profilePlan.material.sourceProfileDigest !== canonicalCandidateDigest(bundle.profile) || evidence.receipt.runCellDigest !== runCell.digest || JSON.stringify(materialization.resolvedModel) !== JSON.stringify(task.model)) throw new Error(`candidate experiment measurement ${index} substituted its ${arm} arm`);
|
|
951
|
+
verifyTaskOutcome(task, evidence, index, arm);
|
|
952
|
+
}
|
|
953
|
+
const baselinePlan = baseline.materializationReceipt.executionPlan.material;
|
|
954
|
+
const candidatePlan = candidate.materializationReceipt.executionPlan.material;
|
|
955
|
+
if (baselinePlan.executionId === candidatePlan.executionId || baselinePlan.runCell.digest === candidatePlan.runCell.digest || baseline.materializationReceipt.digest === candidate.materializationReceipt.digest || baseline.receipt.digest === candidate.receipt.digest || baseline.digest === candidate.digest) throw new Error(`candidate experiment measurement ${index} reused one execution across arms`);
|
|
956
|
+
return {
|
|
957
|
+
baseline,
|
|
958
|
+
candidate
|
|
959
|
+
};
|
|
1155
960
|
}
|
|
1156
961
|
function verifyExecutionEvidence(input) {
|
|
1157
|
-
|
|
1158
|
-
|
|
1159
|
-
|
|
1160
|
-
|
|
1161
|
-
|
|
1162
|
-
|
|
1163
|
-
|
|
1164
|
-
|
|
1165
|
-
|
|
1166
|
-
|
|
1167
|
-
|
|
1168
|
-
verifyMaterialAddressed(evidence.materializationReceipt.executionPlan, "candidate execution plan");
|
|
1169
|
-
verifySelfAddressed(evidence.receipt, "candidate run receipt");
|
|
1170
|
-
verifyMaterialAddressed(evidence.receipt.modelSettlement, "candidate model settlement");
|
|
1171
|
-
verifyMaterialAddressed(evidence.receipt.taskOutcome, "candidate task outcome");
|
|
1172
|
-
verifyMaterialAddressed(evidence.receipt.benchmarkResult, "candidate benchmark result");
|
|
1173
|
-
return evidence;
|
|
962
|
+
const evidence = candidateExecutionEvidenceSchema.parse(input);
|
|
963
|
+
verifySelfAddressed(evidence, "candidate execution evidence");
|
|
964
|
+
verifySelfAddressed(evidence.materializationReceipt, "candidate materialization receipt");
|
|
965
|
+
verifySelfAddressed(evidence.materializationReceipt.profileActivation, "candidate profile activation");
|
|
966
|
+
verifyMaterialAddressed(evidence.materializationReceipt.profileActivation.profilePlan, "candidate profile plan");
|
|
967
|
+
verifyMaterialAddressed(evidence.materializationReceipt.executionPlan, "candidate execution plan");
|
|
968
|
+
verifySelfAddressed(evidence.receipt, "candidate run receipt");
|
|
969
|
+
verifyMaterialAddressed(evidence.receipt.modelSettlement, "candidate model settlement");
|
|
970
|
+
verifyMaterialAddressed(evidence.receipt.taskOutcome, "candidate task outcome");
|
|
971
|
+
verifyMaterialAddressed(evidence.receipt.benchmarkResult, "candidate benchmark result");
|
|
972
|
+
return evidence;
|
|
1174
973
|
}
|
|
1175
974
|
function verifySelfAddressed(document, label) {
|
|
1176
|
-
|
|
1177
|
-
throw new Error(`${label} digest is invalid`);
|
|
1178
|
-
}
|
|
975
|
+
if (canonicalCandidateDigest(omitTopLevelDigest(document)) !== document.digest) throw new Error(`${label} digest is invalid`);
|
|
1179
976
|
}
|
|
1180
977
|
function verifyTaskOutcome(task, evidence, index, arm) {
|
|
1181
|
-
|
|
1182
|
-
|
|
1183
|
-
|
|
1184
|
-
|
|
1185
|
-
|
|
1186
|
-
|
|
1187
|
-
|
|
1188
|
-
|
|
1189
|
-
|
|
1190
|
-
|
|
1191
|
-
|
|
1192
|
-
|
|
1193
|
-
|
|
1194
|
-
|
|
1195
|
-
|
|
1196
|
-
|
|
1197
|
-
|
|
1198
|
-
|
|
1199
|
-
|
|
1200
|
-
|
|
1201
|
-
|
|
1202
|
-
|
|
1203
|
-
|
|
1204
|
-
|
|
1205
|
-
|
|
1206
|
-
|
|
1207
|
-
|
|
1208
|
-
|
|
1209
|
-
|
|
1210
|
-
|
|
1211
|
-
|
|
978
|
+
const outcome = evidence.receipt.taskOutcome.material.outcome;
|
|
979
|
+
const result = evidence.receipt.benchmarkResult.material;
|
|
980
|
+
const prefix = `candidate experiment measurement ${index} ${arm}`;
|
|
981
|
+
if (result.evidence.sha256 === task.grader.artifact.sha256) throw new Error(`${prefix} reused grader bytes as grading evidence`);
|
|
982
|
+
const usage = combinedUsage(evidence);
|
|
983
|
+
const usageChecks = [
|
|
984
|
+
[
|
|
985
|
+
usage.modelCalls,
|
|
986
|
+
task.limits.maxModelCalls,
|
|
987
|
+
"model calls"
|
|
988
|
+
],
|
|
989
|
+
[
|
|
990
|
+
usage.inputTokens,
|
|
991
|
+
task.limits.maxInputTokens,
|
|
992
|
+
"input tokens"
|
|
993
|
+
],
|
|
994
|
+
[
|
|
995
|
+
usage.outputTokens,
|
|
996
|
+
task.limits.maxOutputTokens,
|
|
997
|
+
"output tokens"
|
|
998
|
+
],
|
|
999
|
+
[
|
|
1000
|
+
usage.costUsdNanos,
|
|
1001
|
+
Math.round(task.limits.maxCostUsd * 1e9),
|
|
1002
|
+
"cost"
|
|
1003
|
+
]
|
|
1004
|
+
];
|
|
1005
|
+
for (const [actual, maximum, label] of usageChecks) if (actual > maximum) throw new Error(`${prefix} ${label} ${actual} exceeds the signed limit ${maximum}`);
|
|
1006
|
+
if (outcome.kind !== task.outcome.kind) throw new Error(`${prefix} returned an outcome outside the signed task contract`);
|
|
1007
|
+
if (task.outcome.kind === "output") {
|
|
1008
|
+
if (outcome.kind !== "output" || outcome.spec.mediaType !== task.outcome.mediaType || outcome.spec.maxBytes !== task.outcome.maxBytes) throw new Error(`${prefix} changed the signed output contract`);
|
|
1009
|
+
return;
|
|
1010
|
+
}
|
|
1011
|
+
const repository = task.repository;
|
|
1012
|
+
if (outcome.kind !== "workspace" || repository === void 0 || outcome.baseRepository.identity !== repository.identity || outcome.baseRepository.rootIdentity !== repository.rootIdentity || outcome.baseRepository.commit !== repository.baseCommit || outcome.baseRepository.tree !== repository.baseTree) throw new Error(`${prefix} did not start from the signed repository state`);
|
|
1212
1013
|
}
|
|
1213
1014
|
function verifyStableProfileMaterialization(measurements) {
|
|
1214
|
-
|
|
1215
|
-
|
|
1216
|
-
|
|
1217
|
-
|
|
1218
|
-
|
|
1219
|
-
|
|
1220
|
-
|
|
1221
|
-
|
|
1222
|
-
|
|
1223
|
-
|
|
1224
|
-
|
|
1225
|
-
|
|
1226
|
-
|
|
1227
|
-
|
|
1228
|
-
|
|
1229
|
-
);
|
|
1230
|
-
}
|
|
1231
|
-
}
|
|
1232
|
-
}
|
|
1015
|
+
for (const arm of ["baseline", "candidate"]) {
|
|
1016
|
+
const expected = measurements[0]?.[arm].materializationReceipt.profileActivation;
|
|
1017
|
+
if (!expected) throw new Error("candidate experiment contains no profile materialization");
|
|
1018
|
+
const expectedDigest = canonicalCandidateDigest({
|
|
1019
|
+
profilePlanDigest: expected.profilePlan.digest,
|
|
1020
|
+
files: expected.files
|
|
1021
|
+
});
|
|
1022
|
+
for (const [index, measurement] of measurements.entries()) {
|
|
1023
|
+
const activation = measurement[arm].materializationReceipt.profileActivation;
|
|
1024
|
+
if (canonicalCandidateDigest({
|
|
1025
|
+
profilePlanDigest: activation.profilePlan.digest,
|
|
1026
|
+
files: activation.files
|
|
1027
|
+
}) !== expectedDigest) throw new Error(`candidate experiment measurement ${index} ${arm} materialized a different profile`);
|
|
1028
|
+
}
|
|
1029
|
+
}
|
|
1233
1030
|
}
|
|
1234
1031
|
function completedSuccessfully(evidence) {
|
|
1235
|
-
|
|
1236
|
-
|
|
1032
|
+
const termination = evidence.receipt.termination;
|
|
1033
|
+
return termination.kind === "exit" && termination.exitCode === 0;
|
|
1237
1034
|
}
|
|
1238
1035
|
function verifyMaterialAddressed(evidence, label) {
|
|
1239
|
-
|
|
1240
|
-
throw new Error(`${label} digest is invalid`);
|
|
1241
|
-
}
|
|
1036
|
+
if (canonicalCandidateDigest(evidence.material) !== evidence.digest) throw new Error(`${label} digest is invalid`);
|
|
1242
1037
|
}
|
|
1243
1038
|
function projectPairedMeasurement(measurement, index, adapter) {
|
|
1244
|
-
|
|
1245
|
-
|
|
1246
|
-
|
|
1247
|
-
|
|
1248
|
-
|
|
1249
|
-
|
|
1250
|
-
candidate: projectRun(measurement.candidate, adapter, `paired measurement ${index} candidate`)
|
|
1251
|
-
};
|
|
1039
|
+
if (typeof measurement.cellId !== "string" || !measurement.cellId.trim()) throw new Error(`paired measurement ${index} requires a cell id`);
|
|
1040
|
+
return {
|
|
1041
|
+
cellId: measurement.cellId,
|
|
1042
|
+
baseline: projectRun(measurement.baseline, adapter, `paired measurement ${index} baseline`),
|
|
1043
|
+
candidate: projectRun(measurement.candidate, adapter, `paired measurement ${index} candidate`)
|
|
1044
|
+
};
|
|
1252
1045
|
}
|
|
1253
1046
|
function projectRun(run, adapter, label) {
|
|
1254
|
-
|
|
1255
|
-
|
|
1256
|
-
|
|
1257
|
-
|
|
1258
|
-
|
|
1259
|
-
|
|
1260
|
-
|
|
1261
|
-
|
|
1262
|
-
|
|
1263
|
-
|
|
1264
|
-
|
|
1265
|
-
|
|
1266
|
-
|
|
1267
|
-
|
|
1268
|
-
|
|
1269
|
-
|
|
1270
|
-
|
|
1271
|
-
|
|
1272
|
-
|
|
1273
|
-
return {
|
|
1274
|
-
score: finiteMeasurement(adapter.score(run), `${label} score`),
|
|
1275
|
-
dimensions,
|
|
1276
|
-
costUsd: nonNegativeMeasurement(adapter.costUsd(run), `${label} cost`),
|
|
1277
|
-
latencyMs: nonNegativeMeasurement(adapter.latencyMs(run), `${label} latency`),
|
|
1278
|
-
completed,
|
|
1279
|
-
passed
|
|
1280
|
-
};
|
|
1047
|
+
const suppliedDimensions = adapter.dimensions(run);
|
|
1048
|
+
if (!Array.isArray(suppliedDimensions)) throw new Error(`${label} dimensions must be an array`);
|
|
1049
|
+
const dimensions = /* @__PURE__ */ new Map();
|
|
1050
|
+
for (const dimension of suppliedDimensions) {
|
|
1051
|
+
if (typeof dimension.name !== "string" || !dimension.name.trim()) throw new Error(`${label} contains an unnamed dimension`);
|
|
1052
|
+
if (dimensions.has(dimension.name)) throw new Error(`${label} repeats dimension '${dimension.name}'`);
|
|
1053
|
+
dimensions.set(dimension.name, finiteMeasurement(dimension.score, `${label} ${dimension.name}`));
|
|
1054
|
+
}
|
|
1055
|
+
const completed = adapter.completed(run);
|
|
1056
|
+
const passed = adapter.passed(run);
|
|
1057
|
+
if (typeof completed !== "boolean" || typeof passed !== "boolean") throw new Error(`${label} completion and pass values must be booleans`);
|
|
1058
|
+
return {
|
|
1059
|
+
score: finiteMeasurement(adapter.score(run), `${label} score`),
|
|
1060
|
+
dimensions,
|
|
1061
|
+
costUsd: nonNegativeMeasurement(adapter.costUsd(run), `${label} cost`),
|
|
1062
|
+
latencyMs: nonNegativeMeasurement(adapter.latencyMs(run), `${label} latency`),
|
|
1063
|
+
completed,
|
|
1064
|
+
passed
|
|
1065
|
+
};
|
|
1281
1066
|
}
|
|
1282
1067
|
function sharedProjectedDimensions(measurements) {
|
|
1283
|
-
|
|
1284
|
-
|
|
1285
|
-
|
|
1286
|
-
|
|
1287
|
-
|
|
1288
|
-
|
|
1289
|
-
const actual = [...run.dimensions.keys()];
|
|
1290
|
-
if (JSON.stringify(actual) !== JSON.stringify(expected)) {
|
|
1291
|
-
throw new Error(`paired measurement ${index} ${arm} dimensions do not match the suite`);
|
|
1292
|
-
}
|
|
1293
|
-
}
|
|
1294
|
-
}
|
|
1295
|
-
return expected;
|
|
1068
|
+
const expected = [...measurements[0].baseline.dimensions.keys()];
|
|
1069
|
+
for (const [index, measurement] of measurements.entries()) for (const [arm, run] of [["baseline", measurement.baseline], ["candidate", measurement.candidate]]) {
|
|
1070
|
+
const actual = [...run.dimensions.keys()];
|
|
1071
|
+
if (JSON.stringify(actual) !== JSON.stringify(expected)) throw new Error(`paired measurement ${index} ${arm} dimensions do not match the suite`);
|
|
1072
|
+
}
|
|
1073
|
+
return expected;
|
|
1296
1074
|
}
|
|
1297
1075
|
function dimensionScore(run, name) {
|
|
1298
|
-
|
|
1299
|
-
|
|
1300
|
-
|
|
1076
|
+
const value = run.dimensions.get(name);
|
|
1077
|
+
if (value === void 0) throw new Error(`paired measurement is missing dimension '${name}'`);
|
|
1078
|
+
return value;
|
|
1301
1079
|
}
|
|
1302
1080
|
function measuredEstimate(baseline, candidate, options) {
|
|
1303
|
-
|
|
1304
|
-
|
|
1305
|
-
|
|
1306
|
-
|
|
1307
|
-
|
|
1308
|
-
|
|
1309
|
-
|
|
1310
|
-
|
|
1311
|
-
|
|
1312
|
-
|
|
1313
|
-
|
|
1314
|
-
|
|
1315
|
-
|
|
1316
|
-
|
|
1317
|
-
|
|
1318
|
-
|
|
1319
|
-
|
|
1320
|
-
|
|
1321
|
-
|
|
1322
|
-
|
|
1323
|
-
|
|
1324
|
-
|
|
1325
|
-
|
|
1081
|
+
const bootstrap = pairedBootstrap(baseline, candidate, {
|
|
1082
|
+
confidence: options.confidence,
|
|
1083
|
+
resamples: options.resamples,
|
|
1084
|
+
statistic: "mean",
|
|
1085
|
+
seed: options.seed
|
|
1086
|
+
});
|
|
1087
|
+
const baselineMean = mean(baseline);
|
|
1088
|
+
const candidateMean = mean(candidate);
|
|
1089
|
+
const delta = candidateMean - baselineMean;
|
|
1090
|
+
return {
|
|
1091
|
+
baseline: baselineMean,
|
|
1092
|
+
candidate: candidateMean,
|
|
1093
|
+
delta,
|
|
1094
|
+
confidenceInterval: {
|
|
1095
|
+
level: bootstrap.confidence,
|
|
1096
|
+
lower: Math.min(bootstrap.low, delta),
|
|
1097
|
+
upper: Math.max(bootstrap.high, delta),
|
|
1098
|
+
method: "paired-bootstrap",
|
|
1099
|
+
statistic: "mean",
|
|
1100
|
+
resamples: bootstrap.resamples
|
|
1101
|
+
},
|
|
1102
|
+
n: bootstrap.n
|
|
1103
|
+
};
|
|
1326
1104
|
}
|
|
1327
1105
|
function finiteMeasurement(value, label) {
|
|
1328
|
-
|
|
1329
|
-
|
|
1106
|
+
if (!Number.isFinite(value)) throw new Error(`${label} must be finite`);
|
|
1107
|
+
return value;
|
|
1330
1108
|
}
|
|
1331
1109
|
function nonNegativeMeasurement(value, label) {
|
|
1332
|
-
|
|
1333
|
-
|
|
1334
|
-
}
|
|
1335
|
-
|
|
1336
|
-
|
|
1337
|
-
|
|
1338
|
-
|
|
1339
|
-
|
|
1340
|
-
|
|
1341
|
-
|
|
1110
|
+
if (!Number.isFinite(value) || value < 0) throw new Error(`${label} must be non-negative`);
|
|
1111
|
+
return value;
|
|
1112
|
+
}
|
|
1113
|
+
const candidateExecutionEvidenceAdapter = {
|
|
1114
|
+
score: (evidence) => evidence.receipt.benchmarkResult.material.score,
|
|
1115
|
+
dimensions: (evidence) => evidence.receipt.benchmarkResult.material.dimensions,
|
|
1116
|
+
costUsd: costFromEvidence,
|
|
1117
|
+
latencyMs: latencyFromEvidence,
|
|
1118
|
+
completed: completedSuccessfully,
|
|
1119
|
+
passed: (evidence) => evidence.receipt.benchmarkResult.material.passed
|
|
1342
1120
|
};
|
|
1343
1121
|
function costFromEvidence(evidence) {
|
|
1344
|
-
|
|
1122
|
+
return combinedUsage(evidence).costUsdNanos / 1e9;
|
|
1345
1123
|
}
|
|
1346
1124
|
function latencyFromEvidence(evidence) {
|
|
1347
|
-
|
|
1125
|
+
return evidence.receipt.timing.durationMs + evidence.receipt.benchmarkResult.material.grading.timing.durationMs;
|
|
1348
1126
|
}
|
|
1349
1127
|
function combinedUsage(evidence) {
|
|
1350
|
-
|
|
1351
|
-
|
|
1352
|
-
|
|
1353
|
-
|
|
1354
|
-
|
|
1355
|
-
|
|
1356
|
-
|
|
1357
|
-
|
|
1358
|
-
|
|
1359
|
-
|
|
1128
|
+
const candidate = evidence.receipt.modelSettlement.material.usage;
|
|
1129
|
+
const grader = evidence.receipt.benchmarkResult.material.grading.usage;
|
|
1130
|
+
return {
|
|
1131
|
+
inputTokens: candidate.inputTokens + grader.inputTokens,
|
|
1132
|
+
outputTokens: candidate.outputTokens + grader.outputTokens,
|
|
1133
|
+
cachedInputTokens: candidate.cachedInputTokens + grader.cachedInputTokens,
|
|
1134
|
+
reasoningTokens: candidate.reasoningTokens + grader.reasoningTokens,
|
|
1135
|
+
modelCalls: candidate.modelCalls + grader.modelCalls,
|
|
1136
|
+
costUsdNanos: candidate.costUsdNanos + grader.costUsdNanos
|
|
1137
|
+
};
|
|
1360
1138
|
}
|
|
1361
1139
|
function cellIds(experiment) {
|
|
1362
|
-
|
|
1363
|
-
|
|
1364
|
-
|
|
1365
|
-
|
|
1366
|
-
|
|
1367
|
-
|
|
1140
|
+
const { suite, tasks } = experiment.benchmark;
|
|
1141
|
+
return suite.seeds.map((_, index) => {
|
|
1142
|
+
const taskIndex = Math.floor(index / suite.reps);
|
|
1143
|
+
const repetition = index % suite.reps;
|
|
1144
|
+
return `${tasks[taskIndex]?.scenario.id ?? taskIndex}:${repetition}`;
|
|
1145
|
+
});
|
|
1368
1146
|
}
|
|
1369
1147
|
function mean(values) {
|
|
1370
|
-
|
|
1371
|
-
|
|
1148
|
+
if (values.length === 0) throw new Error("candidate experiment requires measured values");
|
|
1149
|
+
return values.reduce((sum, value) => sum + value, 0) / values.length;
|
|
1372
1150
|
}
|
|
1373
1151
|
function abortError(signal) {
|
|
1374
|
-
|
|
1375
|
-
}
|
|
1376
|
-
|
|
1377
|
-
|
|
1378
|
-
|
|
1379
|
-
|
|
1380
|
-
|
|
1381
|
-
|
|
1382
|
-
|
|
1383
|
-
|
|
1384
|
-
|
|
1152
|
+
return signal.reason instanceof Error ? signal.reason : /* @__PURE__ */ new Error("candidate experiment aborted");
|
|
1153
|
+
}
|
|
1154
|
+
//#endregion
|
|
1155
|
+
//#region src/contract/intake/run-record-dir.ts
|
|
1156
|
+
/**
|
|
1157
|
+
* # `intake/run-record-dir` — load a directory or file of `RunRecord`s.
|
|
1158
|
+
*
|
|
1159
|
+
* The on-disk counterpart to the in-memory intake adapters: point it at a
|
|
1160
|
+
* single `.json` (array) / `.jsonl` (one record per line) file or at a
|
|
1161
|
+
* directory of such files, and it returns the substrate-canonical
|
|
1162
|
+
* `RunRecord[]` ready for `analyzeRuns({ runs })`.
|
|
1163
|
+
*
|
|
1164
|
+
* Validation is at the boundary: each parsed object goes through
|
|
1165
|
+
* `parseRunRecordSafe`. By default an invalid record fails loud with its
|
|
1166
|
+
* file + index; pass `onInvalid: 'collect'` to keep the valid records and
|
|
1167
|
+
* receive the rejects as structured diagnostics instead.
|
|
1168
|
+
*/
|
|
1169
|
+
const ANALYSIS_ARTIFACT$1 = "analysis.json";
|
|
1385
1170
|
function defaultInclude(fileName) {
|
|
1386
|
-
|
|
1387
|
-
|
|
1388
|
-
}
|
|
1171
|
+
if (fileName === ANALYSIS_ARTIFACT$1) return false;
|
|
1172
|
+
return fileName.endsWith(".json") || fileName.endsWith(".jsonl");
|
|
1173
|
+
}
|
|
1174
|
+
/**
|
|
1175
|
+
* Resolve a file or directory path into validated `RunRecord[]`.
|
|
1176
|
+
*
|
|
1177
|
+
* A `.json` file must parse to a top-level array; a `.jsonl` file is one
|
|
1178
|
+
* record per non-empty line. Directories are read shallowly by default
|
|
1179
|
+
* (set `recursive` to descend); the `analysis.json` output artifact is
|
|
1180
|
+
* always excluded.
|
|
1181
|
+
*/
|
|
1389
1182
|
async function fromRunRecordDir(path, options = {}) {
|
|
1390
|
-
|
|
1391
|
-
|
|
1392
|
-
|
|
1393
|
-
|
|
1394
|
-
|
|
1395
|
-
|
|
1396
|
-
|
|
1397
|
-
|
|
1398
|
-
|
|
1399
|
-
|
|
1400
|
-
|
|
1401
|
-
|
|
1402
|
-
|
|
1403
|
-
|
|
1404
|
-
|
|
1405
|
-
|
|
1406
|
-
|
|
1407
|
-
|
|
1408
|
-
|
|
1409
|
-
|
|
1410
|
-
|
|
1411
|
-
|
|
1412
|
-
|
|
1413
|
-
|
|
1414
|
-
|
|
1183
|
+
const onInvalid = options.onInvalid ?? "throw";
|
|
1184
|
+
const include = options.include ?? defaultInclude;
|
|
1185
|
+
const filePaths = (await stat(path)).isDirectory() ? await collectFiles(path, include, options.recursive ?? false) : [path];
|
|
1186
|
+
const runs = [];
|
|
1187
|
+
const rejected = [];
|
|
1188
|
+
for (const file of filePaths) {
|
|
1189
|
+
const raw = await parseRecordFile(file);
|
|
1190
|
+
for (const { index, value } of raw) {
|
|
1191
|
+
const parsed = parseRunRecordSafe(value);
|
|
1192
|
+
if (parsed.ok) {
|
|
1193
|
+
runs.push(parsed.value);
|
|
1194
|
+
continue;
|
|
1195
|
+
}
|
|
1196
|
+
const rejection = {
|
|
1197
|
+
file,
|
|
1198
|
+
index,
|
|
1199
|
+
reason: parsed.error.message
|
|
1200
|
+
};
|
|
1201
|
+
if (onInvalid === "throw") throw new Error(`fromRunRecordDir: invalid RunRecord in '${file}' at index ${index}: ${parsed.error.message}`);
|
|
1202
|
+
rejected.push(rejection);
|
|
1203
|
+
}
|
|
1204
|
+
}
|
|
1205
|
+
return {
|
|
1206
|
+
runs,
|
|
1207
|
+
rejected,
|
|
1208
|
+
files: filePaths
|
|
1209
|
+
};
|
|
1210
|
+
}
|
|
1211
|
+
/** Read a single `.json` / `.jsonl` file into `{ index, value }` pairs. A
|
|
1212
|
+
* malformed JSONL line throws with its line number rather than being skipped —
|
|
1213
|
+
* silent line-dropping is how corpora quietly shrink. */
|
|
1415
1214
|
async function parseRecordFile(file) {
|
|
1416
|
-
|
|
1417
|
-
|
|
1418
|
-
|
|
1419
|
-
|
|
1420
|
-
|
|
1421
|
-
|
|
1422
|
-
|
|
1423
|
-
|
|
1424
|
-
|
|
1425
|
-
|
|
1426
|
-
|
|
1427
|
-
|
|
1428
|
-
|
|
1429
|
-
|
|
1430
|
-
|
|
1431
|
-
|
|
1432
|
-
|
|
1433
|
-
|
|
1434
|
-
|
|
1435
|
-
|
|
1436
|
-
|
|
1437
|
-
|
|
1438
|
-
|
|
1439
|
-
|
|
1440
|
-
|
|
1215
|
+
const trimmed = (await readFile(file, "utf8")).trim();
|
|
1216
|
+
if (trimmed.length === 0) return [];
|
|
1217
|
+
if (trimmed.startsWith("[")) {
|
|
1218
|
+
const parsed = JSON.parse(trimmed);
|
|
1219
|
+
if (!Array.isArray(parsed)) throw new Error(`fromRunRecordDir: file '${file}' did not parse to an array`);
|
|
1220
|
+
return parsed.map((value, index) => ({
|
|
1221
|
+
index,
|
|
1222
|
+
value
|
|
1223
|
+
}));
|
|
1224
|
+
}
|
|
1225
|
+
const out = [];
|
|
1226
|
+
const lines = trimmed.split("\n");
|
|
1227
|
+
for (let i = 0; i < lines.length; i++) {
|
|
1228
|
+
const line = lines[i].trim();
|
|
1229
|
+
if (line.length === 0) continue;
|
|
1230
|
+
try {
|
|
1231
|
+
out.push({
|
|
1232
|
+
index: i,
|
|
1233
|
+
value: JSON.parse(line)
|
|
1234
|
+
});
|
|
1235
|
+
} catch (err) {
|
|
1236
|
+
throw new Error(`fromRunRecordDir: file '${file}' line ${i + 1} is not valid JSON: ${err instanceof Error ? err.message : String(err)}`);
|
|
1237
|
+
}
|
|
1238
|
+
}
|
|
1239
|
+
return out;
|
|
1240
|
+
}
|
|
1241
|
+
/** Sorted file list under a directory, filtered by `include`. Sorted so the
|
|
1242
|
+
* resulting `RunRecord` order — and any downstream fingerprint — is stable
|
|
1243
|
+
* across filesystems. */
|
|
1441
1244
|
async function collectFiles(dir, include, recursive) {
|
|
1442
|
-
|
|
1443
|
-
|
|
1444
|
-
|
|
1445
|
-
|
|
1446
|
-
|
|
1447
|
-
|
|
1448
|
-
|
|
1449
|
-
|
|
1450
|
-
|
|
1451
|
-
|
|
1452
|
-
|
|
1453
|
-
|
|
1454
|
-
|
|
1455
|
-
|
|
1456
|
-
|
|
1457
|
-
|
|
1458
|
-
|
|
1459
|
-
|
|
1460
|
-
|
|
1461
|
-
|
|
1245
|
+
const entries = await readdir(dir, { withFileTypes: true });
|
|
1246
|
+
const files = [];
|
|
1247
|
+
const subdirs = [];
|
|
1248
|
+
for (const entry of entries) {
|
|
1249
|
+
if (entry.isDirectory()) {
|
|
1250
|
+
if (recursive) subdirs.push(join(dir, entry.name));
|
|
1251
|
+
continue;
|
|
1252
|
+
}
|
|
1253
|
+
if (include(entry.name)) files.push(join(dir, entry.name));
|
|
1254
|
+
}
|
|
1255
|
+
files.sort();
|
|
1256
|
+
subdirs.sort();
|
|
1257
|
+
for (const sub of subdirs) files.push(...await collectFiles(sub, include, recursive));
|
|
1258
|
+
return files;
|
|
1259
|
+
}
|
|
1260
|
+
//#endregion
|
|
1261
|
+
//#region src/contract/eval-reporting-suite.ts
|
|
1262
|
+
/**
|
|
1263
|
+
* # `evalReportingSuite` — one call from runs (or a run dir) to `analysis.json`.
|
|
1264
|
+
*
|
|
1265
|
+
* A thin wrapper over the analysis primitive (`analyzeRuns`) and the on-disk
|
|
1266
|
+
* intake adapter (`fromRunRecordDir`). It does NOT reimplement any statistics,
|
|
1267
|
+
* distributions, or clustering — it resolves the input into validated
|
|
1268
|
+
* `RunRecord[]`, calls `analyzeRuns` with the options you'd pass it directly,
|
|
1269
|
+
* wraps the result in a small provenance envelope, and (optionally) writes a
|
|
1270
|
+
* single `analysis.json` artifact.
|
|
1271
|
+
*
|
|
1272
|
+
* ```ts
|
|
1273
|
+
* // From a directory of run files, write ./runs/analysis.json:
|
|
1274
|
+
* const suite = await evalReportingSuite('./runs', { write: true })
|
|
1275
|
+
* // From records already in memory, no write:
|
|
1276
|
+
* const suite = await evalReportingSuite(records, { analyze: { decisionThreshold: 0.03 } })
|
|
1277
|
+
* suite.report // the InsightReport — distributions, paired lift, findings rollup
|
|
1278
|
+
* ```
|
|
1279
|
+
*/
|
|
1280
|
+
const ANALYSIS_ARTIFACT = "analysis.json";
|
|
1281
|
+
/**
|
|
1282
|
+
* Resolve runs (or a run dir/file), run `analyzeRuns`, and optionally persist a
|
|
1283
|
+
* single `analysis.json`. The only analysis logic lives in `analyzeRuns`; this
|
|
1284
|
+
* function is composition + I/O.
|
|
1285
|
+
*/
|
|
1462
1286
|
async function evalReportingSuite(input, options = {}) {
|
|
1463
|
-
|
|
1464
|
-
|
|
1465
|
-
|
|
1466
|
-
|
|
1467
|
-
|
|
1468
|
-
|
|
1469
|
-
|
|
1470
|
-
|
|
1471
|
-
|
|
1472
|
-
|
|
1473
|
-
|
|
1474
|
-
|
|
1475
|
-
|
|
1476
|
-
|
|
1477
|
-
|
|
1478
|
-
|
|
1479
|
-
|
|
1480
|
-
|
|
1481
|
-
|
|
1482
|
-
|
|
1483
|
-
|
|
1484
|
-
|
|
1485
|
-
|
|
1486
|
-
|
|
1487
|
-
|
|
1488
|
-
|
|
1489
|
-
|
|
1490
|
-
|
|
1491
|
-
|
|
1492
|
-
|
|
1493
|
-
|
|
1494
|
-
|
|
1495
|
-
|
|
1496
|
-
|
|
1497
|
-
|
|
1498
|
-
|
|
1499
|
-
|
|
1500
|
-
}
|
|
1287
|
+
const fromPath = typeof input === "string";
|
|
1288
|
+
let runs;
|
|
1289
|
+
let files = [];
|
|
1290
|
+
let rejected = [];
|
|
1291
|
+
if (fromPath) {
|
|
1292
|
+
const loaded = await fromRunRecordDir(input, options.load);
|
|
1293
|
+
runs = loaded.runs;
|
|
1294
|
+
files = loaded.files;
|
|
1295
|
+
rejected = loaded.rejected;
|
|
1296
|
+
} else runs = input;
|
|
1297
|
+
if (runs.length === 0) throw new Error(fromPath ? `evalReportingSuite: no RunRecords found at '${input}'` : "evalReportingSuite: no RunRecords to analyze");
|
|
1298
|
+
const result = {
|
|
1299
|
+
report: await analyzeRuns({
|
|
1300
|
+
...options.analyze,
|
|
1301
|
+
runs
|
|
1302
|
+
}),
|
|
1303
|
+
provenance: {
|
|
1304
|
+
generatedAt: (/* @__PURE__ */ new Date()).toISOString(),
|
|
1305
|
+
runCount: runs.length,
|
|
1306
|
+
sourcePath: fromPath ? input : null,
|
|
1307
|
+
files,
|
|
1308
|
+
rejected
|
|
1309
|
+
},
|
|
1310
|
+
writtenTo: null
|
|
1311
|
+
};
|
|
1312
|
+
const target = resolveWriteTarget(options.write, fromPath ? input : null);
|
|
1313
|
+
if (target) {
|
|
1314
|
+
await mkdir(dirname(target), { recursive: true });
|
|
1315
|
+
await writeFile(target, `${JSON.stringify(result, null, 2)}\n`, "utf8");
|
|
1316
|
+
result.writtenTo = target;
|
|
1317
|
+
}
|
|
1318
|
+
return result;
|
|
1319
|
+
}
|
|
1320
|
+
/** Resolve where (if anywhere) to write `analysis.json`. Returns null when
|
|
1321
|
+
* writing is disabled. Throws on `write: true` with in-memory input — there is
|
|
1322
|
+
* no directory to anchor the artifact to, and silently inventing `cwd` would
|
|
1323
|
+
* scatter files. */
|
|
1501
1324
|
function resolveWriteTarget(write, sourcePath) {
|
|
1502
|
-
|
|
1503
|
-
|
|
1504
|
-
|
|
1505
|
-
|
|
1506
|
-
}
|
|
1507
|
-
if (sourcePath === null) {
|
|
1508
|
-
throw new Error(
|
|
1509
|
-
"evalReportingSuite: write:true needs a source path to anchor analysis.json \u2014 pass an explicit output path when analyzing in-memory records"
|
|
1510
|
-
);
|
|
1511
|
-
}
|
|
1512
|
-
const isFile = sourcePath.endsWith(".json") || sourcePath.endsWith(".jsonl");
|
|
1513
|
-
return isFile ? join2(dirname(sourcePath), ANALYSIS_ARTIFACT2) : join2(sourcePath, ANALYSIS_ARTIFACT2);
|
|
1325
|
+
if (!write) return null;
|
|
1326
|
+
if (typeof write === "string") return write.endsWith("/") || !write.endsWith(".json") && !write.endsWith(".jsonl") ? join(write, ANALYSIS_ARTIFACT) : write;
|
|
1327
|
+
if (sourcePath === null) throw new Error("evalReportingSuite: write:true needs a source path to anchor analysis.json — pass an explicit output path when analyzing in-memory records");
|
|
1328
|
+
return sourcePath.endsWith(".json") || sourcePath.endsWith(".jsonl") ? join(dirname(sourcePath), ANALYSIS_ARTIFACT) : join(sourcePath, ANALYSIS_ARTIFACT);
|
|
1514
1329
|
}
|
|
1515
|
-
|
|
1516
|
-
|
|
1330
|
+
//#endregion
|
|
1331
|
+
//#region src/contract/diff.ts
|
|
1517
1332
|
function keyForCell(cell) {
|
|
1518
|
-
|
|
1333
|
+
return JSON.stringify([cell.scenarioId, cell.rep]);
|
|
1519
1334
|
}
|
|
1335
|
+
/** Build the per-dimension delta map for a matched cell. Each judge name +
|
|
1336
|
+
* dimension name encountered on EITHER side appears in the result. */
|
|
1520
1337
|
function diffDimensions(before, after) {
|
|
1521
|
-
|
|
1522
|
-
|
|
1523
|
-
|
|
1524
|
-
|
|
1525
|
-
|
|
1526
|
-
|
|
1527
|
-
|
|
1528
|
-
|
|
1529
|
-
|
|
1530
|
-
|
|
1531
|
-
|
|
1532
|
-
|
|
1533
|
-
|
|
1534
|
-
|
|
1535
|
-
|
|
1536
|
-
|
|
1537
|
-
|
|
1538
|
-
|
|
1539
|
-
|
|
1540
|
-
|
|
1541
|
-
|
|
1542
|
-
}
|
|
1338
|
+
const out = {};
|
|
1339
|
+
const judges = /* @__PURE__ */ new Set([...Object.keys(before), ...Object.keys(after)]);
|
|
1340
|
+
for (const judge of judges) {
|
|
1341
|
+
const beforeDims = before[judge] ?? {};
|
|
1342
|
+
const afterDims = after[judge] ?? {};
|
|
1343
|
+
const dims = /* @__PURE__ */ new Set([...Object.keys(beforeDims), ...Object.keys(afterDims)]);
|
|
1344
|
+
const judgeOut = {};
|
|
1345
|
+
for (const dim of dims) {
|
|
1346
|
+
const rawBefore = beforeDims[dim];
|
|
1347
|
+
const rawAfter = afterDims[dim];
|
|
1348
|
+
const b = typeof rawBefore === "number" && Number.isFinite(rawBefore) ? rawBefore : null;
|
|
1349
|
+
const a = typeof rawAfter === "number" && Number.isFinite(rawAfter) ? rawAfter : null;
|
|
1350
|
+
judgeOut[dim] = {
|
|
1351
|
+
before: b,
|
|
1352
|
+
after: a,
|
|
1353
|
+
delta: b !== null && a !== null ? a - b : null
|
|
1354
|
+
};
|
|
1355
|
+
}
|
|
1356
|
+
out[judge] = judgeOut;
|
|
1357
|
+
}
|
|
1358
|
+
return out;
|
|
1359
|
+
}
|
|
1360
|
+
/**
|
|
1361
|
+
* Diff two generation snapshots. Cells are matched on `(scenarioId, rep)`;
|
|
1362
|
+
* unmatched cells surface in `added` / `removed`. Aggregate fields are
|
|
1363
|
+
* recomputed from the snapshot's stored fields, not re-derived from cells —
|
|
1364
|
+
* this keeps the diff consistent with whatever aggregation the substrate
|
|
1365
|
+
* actually reported.
|
|
1366
|
+
*/
|
|
1543
1367
|
function diffGenerations(before, after) {
|
|
1544
|
-
|
|
1545
|
-
|
|
1546
|
-
|
|
1547
|
-
|
|
1548
|
-
|
|
1549
|
-
|
|
1550
|
-
|
|
1551
|
-
|
|
1552
|
-
|
|
1553
|
-
|
|
1554
|
-
|
|
1555
|
-
|
|
1556
|
-
|
|
1557
|
-
|
|
1558
|
-
|
|
1559
|
-
|
|
1560
|
-
|
|
1561
|
-
|
|
1562
|
-
|
|
1563
|
-
|
|
1564
|
-
|
|
1565
|
-
|
|
1566
|
-
|
|
1567
|
-
|
|
1568
|
-
|
|
1569
|
-
|
|
1570
|
-
|
|
1571
|
-
|
|
1572
|
-
|
|
1573
|
-
|
|
1574
|
-
|
|
1575
|
-
|
|
1576
|
-
|
|
1577
|
-
|
|
1578
|
-
|
|
1579
|
-
|
|
1580
|
-
|
|
1581
|
-
|
|
1582
|
-
|
|
1583
|
-
|
|
1584
|
-
|
|
1585
|
-
|
|
1586
|
-
}
|
|
1368
|
+
const beforeMap = new Map(before.cells.map((c) => [keyForCell(c), c]));
|
|
1369
|
+
const afterMap = new Map(after.cells.map((c) => [keyForCell(c), c]));
|
|
1370
|
+
const matched = [];
|
|
1371
|
+
const removed = [];
|
|
1372
|
+
const added = [];
|
|
1373
|
+
for (const [key, beforeCell] of beforeMap) {
|
|
1374
|
+
const afterCell = afterMap.get(key);
|
|
1375
|
+
if (!afterCell) {
|
|
1376
|
+
removed.push(beforeCell);
|
|
1377
|
+
continue;
|
|
1378
|
+
}
|
|
1379
|
+
matched.push({
|
|
1380
|
+
scenarioId: beforeCell.scenarioId,
|
|
1381
|
+
rep: beforeCell.rep,
|
|
1382
|
+
compositeBefore: beforeCell.compositeMean,
|
|
1383
|
+
compositeAfter: afterCell.compositeMean,
|
|
1384
|
+
compositeDelta: beforeCell.compositeMean === null || afterCell.compositeMean === null ? null : afterCell.compositeMean - beforeCell.compositeMean,
|
|
1385
|
+
dimensions: diffDimensions(beforeCell.dimensions, afterCell.dimensions)
|
|
1386
|
+
});
|
|
1387
|
+
}
|
|
1388
|
+
for (const [key, afterCell] of afterMap) if (!beforeMap.has(key)) added.push(afterCell);
|
|
1389
|
+
return {
|
|
1390
|
+
beforeIndex: before.index,
|
|
1391
|
+
afterIndex: after.index,
|
|
1392
|
+
beforeSurfaceHash: before.surfaceHash,
|
|
1393
|
+
afterSurfaceHash: after.surfaceHash,
|
|
1394
|
+
surfaceChanged: before.surfaceHash !== after.surfaceHash,
|
|
1395
|
+
matched,
|
|
1396
|
+
removed,
|
|
1397
|
+
added,
|
|
1398
|
+
compositeBefore: before.compositeMean,
|
|
1399
|
+
compositeAfter: after.compositeMean,
|
|
1400
|
+
compositeDelta: before.compositeMean === null || after.compositeMean === null ? null : after.compositeMean - before.compositeMean,
|
|
1401
|
+
costUsdBefore: before.costUsd,
|
|
1402
|
+
costUsdAfter: after.costUsd,
|
|
1403
|
+
costUsdDelta: after.costUsd - before.costUsd,
|
|
1404
|
+
durationMsBefore: before.durationMs,
|
|
1405
|
+
durationMsAfter: after.durationMs,
|
|
1406
|
+
durationMsDelta: after.durationMs - before.durationMs
|
|
1407
|
+
};
|
|
1408
|
+
}
|
|
1409
|
+
/** Highest-index generation, or null if the run recorded none. */
|
|
1587
1410
|
function winnerOf(run) {
|
|
1588
|
-
|
|
1589
|
-
|
|
1590
|
-
|
|
1591
|
-
|
|
1592
|
-
|
|
1593
|
-
|
|
1594
|
-
|
|
1411
|
+
if (run.generations.length === 0) return null;
|
|
1412
|
+
let winner = run.generations[0];
|
|
1413
|
+
for (const gen of run.generations) if (gen.index > winner.index) winner = gen;
|
|
1414
|
+
return winner;
|
|
1415
|
+
}
|
|
1416
|
+
/**
|
|
1417
|
+
* Diff two full eval-runs. Produces baseline-vs-baseline and
|
|
1418
|
+
* winner-vs-winner generation diffs when both sides expose them, plus
|
|
1419
|
+
* run-level cost / lift / gate-decision deltas.
|
|
1420
|
+
*/
|
|
1595
1421
|
function diffRuns(before, after) {
|
|
1596
|
-
|
|
1597
|
-
|
|
1598
|
-
|
|
1599
|
-
|
|
1600
|
-
|
|
1601
|
-
|
|
1602
|
-
|
|
1603
|
-
|
|
1604
|
-
|
|
1605
|
-
|
|
1606
|
-
|
|
1607
|
-
|
|
1608
|
-
|
|
1609
|
-
|
|
1610
|
-
|
|
1611
|
-
|
|
1612
|
-
|
|
1613
|
-
|
|
1614
|
-
|
|
1615
|
-
|
|
1616
|
-
|
|
1617
|
-
|
|
1618
|
-
|
|
1619
|
-
|
|
1620
|
-
|
|
1621
|
-
}
|
|
1422
|
+
const beforeWinner = winnerOf(before);
|
|
1423
|
+
const afterWinner = winnerOf(after);
|
|
1424
|
+
const baselineDiff = before.baseline && after.baseline ? diffGenerations(before.baseline, after.baseline) : null;
|
|
1425
|
+
const winnersDiff = beforeWinner && afterWinner ? diffGenerations(beforeWinner, afterWinner) : null;
|
|
1426
|
+
const beforeLift = before.holdoutLift ?? null;
|
|
1427
|
+
const afterLift = after.holdoutLift ?? null;
|
|
1428
|
+
return {
|
|
1429
|
+
beforeRunId: before.runId,
|
|
1430
|
+
afterRunId: after.runId,
|
|
1431
|
+
beforeTimestamp: before.timestamp,
|
|
1432
|
+
afterTimestamp: after.timestamp,
|
|
1433
|
+
beforeGateDecision: before.gateDecision ?? null,
|
|
1434
|
+
afterGateDecision: after.gateDecision ?? null,
|
|
1435
|
+
beforeHoldoutLift: beforeLift,
|
|
1436
|
+
afterHoldoutLift: afterLift,
|
|
1437
|
+
holdoutLiftDelta: beforeLift !== null && afterLift !== null ? afterLift - beforeLift : null,
|
|
1438
|
+
beforeTotalCostUsd: before.totalCostUsd,
|
|
1439
|
+
afterTotalCostUsd: after.totalCostUsd,
|
|
1440
|
+
totalCostUsdDelta: after.totalCostUsd - before.totalCostUsd,
|
|
1441
|
+
beforeTotalDurationMs: before.totalDurationMs,
|
|
1442
|
+
afterTotalDurationMs: after.totalDurationMs,
|
|
1443
|
+
totalDurationMsDelta: after.totalDurationMs - before.totalDurationMs,
|
|
1444
|
+
baselineDiff,
|
|
1445
|
+
winnersDiff
|
|
1446
|
+
};
|
|
1447
|
+
}
|
|
1448
|
+
/**
|
|
1449
|
+
* Within-run baseline → winning-generation diff. The natural "what did the
|
|
1450
|
+
* improvement loop produce" view for a single run. Returns null when the
|
|
1451
|
+
* run never reached a generation past baseline (errored early, or the gate
|
|
1452
|
+
* shipped the baseline as-is).
|
|
1453
|
+
*/
|
|
1622
1454
|
function diffRunBaselineToWinner(run) {
|
|
1623
|
-
|
|
1624
|
-
|
|
1625
|
-
|
|
1626
|
-
|
|
1455
|
+
if (!run.baseline) return null;
|
|
1456
|
+
const winner = winnerOf(run);
|
|
1457
|
+
if (!winner || winner.index === run.baseline.index) return null;
|
|
1458
|
+
return diffGenerations(run.baseline, winner);
|
|
1627
1459
|
}
|
|
1628
|
-
|
|
1629
|
-
|
|
1460
|
+
//#endregion
|
|
1461
|
+
//#region src/contract/intake/agent-trace.ts
|
|
1630
1462
|
function rangeLines(r) {
|
|
1631
|
-
|
|
1463
|
+
return Math.max(0, r.end_line - r.start_line + 1);
|
|
1632
1464
|
}
|
|
1465
|
+
/**
|
|
1466
|
+
* Build a commit → provenance index from Agent Trace records. Multiple records
|
|
1467
|
+
* for the same revision are merged. Records without `vcs.revision` are skipped
|
|
1468
|
+
* (the SHA is the join key — without it there is nothing to correlate against).
|
|
1469
|
+
*/
|
|
1633
1470
|
function parseAgentTrace(records) {
|
|
1634
|
-
|
|
1635
|
-
|
|
1636
|
-
|
|
1637
|
-
|
|
1638
|
-
|
|
1639
|
-
|
|
1640
|
-
|
|
1641
|
-
|
|
1642
|
-
|
|
1643
|
-
|
|
1644
|
-
|
|
1645
|
-
|
|
1646
|
-
|
|
1647
|
-
|
|
1648
|
-
|
|
1649
|
-
|
|
1650
|
-
|
|
1651
|
-
|
|
1652
|
-
|
|
1653
|
-
|
|
1654
|
-
|
|
1655
|
-
|
|
1656
|
-
|
|
1657
|
-
|
|
1658
|
-
|
|
1659
|
-
|
|
1660
|
-
|
|
1661
|
-
|
|
1662
|
-
|
|
1663
|
-
|
|
1664
|
-
|
|
1665
|
-
|
|
1666
|
-
|
|
1667
|
-
|
|
1668
|
-
|
|
1669
|
-
|
|
1670
|
-
|
|
1671
|
-
|
|
1672
|
-
|
|
1673
|
-
|
|
1674
|
-
|
|
1675
|
-
|
|
1676
|
-
|
|
1677
|
-
|
|
1678
|
-
|
|
1679
|
-
|
|
1680
|
-
|
|
1681
|
-
|
|
1682
|
-
|
|
1471
|
+
const acc = /* @__PURE__ */ new Map();
|
|
1472
|
+
for (const record of records) {
|
|
1473
|
+
const sha = record.vcs?.revision;
|
|
1474
|
+
if (!sha) continue;
|
|
1475
|
+
let a = acc.get(sha);
|
|
1476
|
+
if (!a) {
|
|
1477
|
+
a = {
|
|
1478
|
+
models: /* @__PURE__ */ new Set(),
|
|
1479
|
+
tools: /* @__PURE__ */ new Set(),
|
|
1480
|
+
files: /* @__PURE__ */ new Set(),
|
|
1481
|
+
conversationCount: 0,
|
|
1482
|
+
lineCount: 0,
|
|
1483
|
+
humanInvolved: false
|
|
1484
|
+
};
|
|
1485
|
+
acc.set(sha, a);
|
|
1486
|
+
}
|
|
1487
|
+
if (record.tool?.name) a.tools.add(record.tool.name);
|
|
1488
|
+
for (const file of record.files ?? []) {
|
|
1489
|
+
a.files.add(file.path);
|
|
1490
|
+
for (const conv of file.conversations ?? []) {
|
|
1491
|
+
a.conversationCount += 1;
|
|
1492
|
+
for (const range of conv.ranges ?? []) {
|
|
1493
|
+
const contributor = range.contributor ?? conv.contributor;
|
|
1494
|
+
a.lineCount += rangeLines(range);
|
|
1495
|
+
if (!contributor) continue;
|
|
1496
|
+
if (contributor.type === "human" || contributor.type === "mixed") a.humanInvolved = true;
|
|
1497
|
+
if ((contributor.type === "ai" || contributor.type === "mixed") && contributor.model_id) a.models.add(contributor.model_id);
|
|
1498
|
+
}
|
|
1499
|
+
}
|
|
1500
|
+
}
|
|
1501
|
+
}
|
|
1502
|
+
const index = /* @__PURE__ */ new Map();
|
|
1503
|
+
for (const [sha, a] of acc) index.set(sha, {
|
|
1504
|
+
commitSha: sha,
|
|
1505
|
+
aiModels: [...a.models].sort(),
|
|
1506
|
+
tools: [...a.tools].sort(),
|
|
1507
|
+
conversationCount: a.conversationCount,
|
|
1508
|
+
fileCount: a.files.size,
|
|
1509
|
+
lineCount: a.lineCount,
|
|
1510
|
+
humanInvolved: a.humanInvolved
|
|
1511
|
+
});
|
|
1512
|
+
return index;
|
|
1513
|
+
}
|
|
1514
|
+
/**
|
|
1515
|
+
* Partition runs by the AI model(s) that authored the code at each run's
|
|
1516
|
+
* `commitSha`. Feed `byModel.get(modelId)` to `analyzeRuns`, or compare two
|
|
1517
|
+
* model cohorts via `analyzeRuns({ runs: a, baselineRuns: b })` for a lift CI
|
|
1518
|
+
* on "model A's code vs model B's code".
|
|
1519
|
+
*/
|
|
1683
1520
|
function partitionRunsByAuthoringModel(runs, index) {
|
|
1684
|
-
|
|
1685
|
-
|
|
1686
|
-
|
|
1687
|
-
|
|
1688
|
-
|
|
1689
|
-
|
|
1690
|
-
|
|
1691
|
-
|
|
1692
|
-
|
|
1693
|
-
|
|
1694
|
-
|
|
1695
|
-
|
|
1696
|
-
|
|
1697
|
-
|
|
1698
|
-
|
|
1699
|
-
|
|
1700
|
-
|
|
1701
|
-
|
|
1521
|
+
const byModel = /* @__PURE__ */ new Map();
|
|
1522
|
+
const unattributed = [];
|
|
1523
|
+
for (const run of runs) {
|
|
1524
|
+
const provenance = index.get(run.commitSha);
|
|
1525
|
+
if (!provenance || provenance.aiModels.length === 0) {
|
|
1526
|
+
unattributed.push(run);
|
|
1527
|
+
continue;
|
|
1528
|
+
}
|
|
1529
|
+
for (const model of provenance.aiModels) {
|
|
1530
|
+
const cohort = byModel.get(model) ?? [];
|
|
1531
|
+
cohort.push(run);
|
|
1532
|
+
byModel.set(model, cohort);
|
|
1533
|
+
}
|
|
1534
|
+
}
|
|
1535
|
+
return {
|
|
1536
|
+
byModel,
|
|
1537
|
+
unattributed
|
|
1538
|
+
};
|
|
1539
|
+
}
|
|
1540
|
+
//#endregion
|
|
1541
|
+
//#region src/contract/intake/feedback-table.ts
|
|
1702
1542
|
function fromFeedbackTable(opts) {
|
|
1703
|
-
|
|
1704
|
-
|
|
1705
|
-
|
|
1706
|
-
|
|
1707
|
-
|
|
1708
|
-
|
|
1709
|
-
|
|
1710
|
-
|
|
1711
|
-
|
|
1712
|
-
|
|
1713
|
-
|
|
1714
|
-
|
|
1715
|
-
|
|
1716
|
-
|
|
1717
|
-
|
|
1718
|
-
|
|
1719
|
-
|
|
1720
|
-
|
|
1721
|
-
|
|
1722
|
-
|
|
1723
|
-
|
|
1724
|
-
|
|
1725
|
-
|
|
1726
|
-
|
|
1727
|
-
|
|
1728
|
-
|
|
1729
|
-
|
|
1730
|
-
|
|
1731
|
-
|
|
1732
|
-
|
|
1733
|
-
|
|
1734
|
-
|
|
1735
|
-
|
|
1736
|
-
|
|
1737
|
-
|
|
1738
|
-
|
|
1739
|
-
|
|
1740
|
-
|
|
1741
|
-
|
|
1742
|
-
|
|
1743
|
-
|
|
1744
|
-
|
|
1745
|
-
|
|
1746
|
-
|
|
1747
|
-
|
|
1748
|
-
|
|
1749
|
-
|
|
1750
|
-
|
|
1751
|
-
|
|
1752
|
-
|
|
1753
|
-
|
|
1754
|
-
|
|
1755
|
-
|
|
1756
|
-
|
|
1757
|
-
|
|
1758
|
-
|
|
1759
|
-
|
|
1760
|
-
|
|
1761
|
-
|
|
1762
|
-
|
|
1763
|
-
|
|
1764
|
-
|
|
1765
|
-
|
|
1766
|
-
|
|
1767
|
-
|
|
1768
|
-
|
|
1543
|
+
const { ratings, meta = [], scale, emitRaterScores = true } = opts;
|
|
1544
|
+
const metaByRun = new Map(meta.map((m) => [m.runId, m]));
|
|
1545
|
+
const normalise = (rating) => {
|
|
1546
|
+
if (typeof rating === "boolean") return rating ? 1 : 0;
|
|
1547
|
+
if (!Number.isFinite(rating)) return NaN;
|
|
1548
|
+
if (!scale) return rating;
|
|
1549
|
+
const { min, max } = scale;
|
|
1550
|
+
if (max === min) return rating;
|
|
1551
|
+
return (rating - min) / (max - min);
|
|
1552
|
+
};
|
|
1553
|
+
const byRun = /* @__PURE__ */ new Map();
|
|
1554
|
+
for (const row of ratings) {
|
|
1555
|
+
const list = byRun.get(row.runId) ?? [];
|
|
1556
|
+
list.push(row);
|
|
1557
|
+
byRun.set(row.runId, list);
|
|
1558
|
+
}
|
|
1559
|
+
const runs = [];
|
|
1560
|
+
const raterScores = [];
|
|
1561
|
+
for (const [runId, rowsForRun] of byRun) {
|
|
1562
|
+
const normalised = rowsForRun.map((r) => ({
|
|
1563
|
+
rater: r.rater,
|
|
1564
|
+
score: normalise(r.rating)
|
|
1565
|
+
})).filter((r) => Number.isFinite(r.score));
|
|
1566
|
+
if (normalised.length === 0) continue;
|
|
1567
|
+
const meanScore = normalised.reduce((s, r) => s + r.score, 0) / normalised.length;
|
|
1568
|
+
const runMeta = metaByRun.get(runId) ?? { runId };
|
|
1569
|
+
const judgeScores = {
|
|
1570
|
+
perJudge: Object.fromEntries(normalised.map((r) => [r.rater, { rating: r.score }])),
|
|
1571
|
+
perDimMean: { rating: meanScore },
|
|
1572
|
+
composite: meanScore
|
|
1573
|
+
};
|
|
1574
|
+
const splitTag = runMeta.splitTag ?? "holdout";
|
|
1575
|
+
const outcome = {
|
|
1576
|
+
...splitTag === "holdout" ? { holdoutScore: meanScore } : { searchScore: meanScore },
|
|
1577
|
+
raw: Object.fromEntries(normalised.map((r) => [`rater:${r.rater}`, r.score])),
|
|
1578
|
+
judgeScores
|
|
1579
|
+
};
|
|
1580
|
+
const costUsd = runMeta.costUsd ?? null;
|
|
1581
|
+
runs.push({
|
|
1582
|
+
runId,
|
|
1583
|
+
experimentId: runMeta.experimentId ?? "feedback-corpus",
|
|
1584
|
+
candidateId: runMeta.candidateId ?? runId,
|
|
1585
|
+
seed: 0,
|
|
1586
|
+
model: runMeta.model ?? "unknown@unknown",
|
|
1587
|
+
promptHash: runMeta.promptHash ?? "sha256:unknown",
|
|
1588
|
+
configHash: runMeta.configHash ?? "sha256:unknown",
|
|
1589
|
+
commitSha: runMeta.commitSha ?? "unknown",
|
|
1590
|
+
wallMs: runMeta.wallMs ?? 0,
|
|
1591
|
+
costUsd,
|
|
1592
|
+
costProvenance: costUsd === null ? {
|
|
1593
|
+
kind: "uncaptured",
|
|
1594
|
+
usd: null
|
|
1595
|
+
} : {
|
|
1596
|
+
kind: "observed",
|
|
1597
|
+
usd: costUsd
|
|
1598
|
+
},
|
|
1599
|
+
tokenUsage: {
|
|
1600
|
+
input: 0,
|
|
1601
|
+
output: 0
|
|
1602
|
+
},
|
|
1603
|
+
terminalOutcome: "unknown",
|
|
1604
|
+
outcome,
|
|
1605
|
+
splitTag,
|
|
1606
|
+
scenarioId: runMeta.scenarioId ?? runId
|
|
1607
|
+
});
|
|
1608
|
+
if (emitRaterScores) for (const r of normalised) raterScores.push({
|
|
1609
|
+
runId,
|
|
1610
|
+
rater: r.rater,
|
|
1611
|
+
score: r.score
|
|
1612
|
+
});
|
|
1613
|
+
}
|
|
1614
|
+
return {
|
|
1615
|
+
runs,
|
|
1616
|
+
raterScores
|
|
1617
|
+
};
|
|
1618
|
+
}
|
|
1619
|
+
//#endregion
|
|
1620
|
+
//#region src/contract/intake/otel-spans.ts
|
|
1621
|
+
/**
|
|
1622
|
+
* # `intake/otel-spans` — OTel `TraceSpanEvent[]` → `RunRecord[]`.
|
|
1623
|
+
*
|
|
1624
|
+
* Turns an existing observability stream into the substrate-canonical
|
|
1625
|
+
* `RunRecord` shape so consumers with logs but no eval discipline can
|
|
1626
|
+
* call `analyzeRuns()` against their production traffic immediately.
|
|
1627
|
+
*
|
|
1628
|
+
* Pivot rule: spans are grouped by `tangle.runId` (the same attribute the
|
|
1629
|
+
* hosted-tier wire format uses) or, when absent, by `traceId`. One group
|
|
1630
|
+
* becomes one `RunRecord`. The root span (no `parentSpanId`) supplies:
|
|
1631
|
+
*
|
|
1632
|
+
* - `runId` (the group key)
|
|
1633
|
+
* - `wallMs` from `endTimeUnixNano - startTimeUnixNano`
|
|
1634
|
+
* - `model` from `gen_ai.request.model` / `llm.model` / `tangle.model`
|
|
1635
|
+
* - task failure class and detail from explicit `tangle.task.*` attributes
|
|
1636
|
+
* - cost from `cost.usd` / `gen_ai.usage.cost_usd` / `tangle.cost.usd`
|
|
1637
|
+
* - token usage from model-call input, output, cache-read, and cache-write
|
|
1638
|
+
* attributes without double-counting aggregate parent spans
|
|
1639
|
+
* - task quality from an explicit `scoreForRun` callback or a designated
|
|
1640
|
+
* evaluation attribute on a root / `EVALUATOR` span; `outcome.raw`
|
|
1641
|
+
* collects every numeric attribute without promoting it to task quality.
|
|
1642
|
+
*
|
|
1643
|
+
* Errored tool, model, and child-agent spans contribute to execution-error
|
|
1644
|
+
* counts. Root process, guardrail, evaluator, propagated parent, and unknown
|
|
1645
|
+
* errors retain separate counters. Only one failed root can set
|
|
1646
|
+
* `RunRecord.terminalOutcome` and `RunRecord.terminalFailureReason`; a child
|
|
1647
|
+
* error cannot become a task failure.
|
|
1648
|
+
*/
|
|
1649
|
+
const TASK_SCORE_ATTR_KEYS = [
|
|
1650
|
+
"gen_ai.evaluation.score.value",
|
|
1651
|
+
"tangle.task.score",
|
|
1652
|
+
"eval.score",
|
|
1653
|
+
"tangle.score"
|
|
1769
1654
|
];
|
|
1770
|
-
|
|
1771
|
-
|
|
1772
|
-
|
|
1655
|
+
const MODEL_KEYS = [
|
|
1656
|
+
"tangle.model",
|
|
1657
|
+
...LLM_MODEL_ATTR_KEYS,
|
|
1658
|
+
"model"
|
|
1659
|
+
];
|
|
1660
|
+
const PROMPT_HASH_KEYS = ["tangle.prompt_hash", "prompt.hash"];
|
|
1661
|
+
const CONFIG_HASH_KEYS = ["tangle.config_hash", "config.hash"];
|
|
1773
1662
|
function fromOtelSpans(opts) {
|
|
1774
|
-
|
|
1775
|
-
|
|
1776
|
-
|
|
1777
|
-
|
|
1778
|
-
|
|
1779
|
-
|
|
1780
|
-
|
|
1781
|
-
|
|
1782
|
-
|
|
1783
|
-
|
|
1784
|
-
|
|
1785
|
-
|
|
1786
|
-
|
|
1787
|
-
|
|
1788
|
-
|
|
1789
|
-
|
|
1790
|
-
|
|
1791
|
-
|
|
1792
|
-
|
|
1793
|
-
|
|
1794
|
-
|
|
1795
|
-
|
|
1796
|
-
|
|
1797
|
-
|
|
1798
|
-
|
|
1799
|
-
|
|
1800
|
-
|
|
1801
|
-
|
|
1802
|
-
|
|
1803
|
-
|
|
1804
|
-
|
|
1805
|
-
|
|
1806
|
-
|
|
1807
|
-
|
|
1808
|
-
|
|
1809
|
-
|
|
1810
|
-
|
|
1811
|
-
|
|
1812
|
-
|
|
1813
|
-
|
|
1814
|
-
|
|
1815
|
-
|
|
1816
|
-
|
|
1817
|
-
|
|
1818
|
-
|
|
1819
|
-
|
|
1820
|
-
|
|
1821
|
-
|
|
1822
|
-
|
|
1823
|
-
|
|
1824
|
-
|
|
1825
|
-
|
|
1826
|
-
|
|
1827
|
-
|
|
1828
|
-
|
|
1829
|
-
|
|
1830
|
-
|
|
1831
|
-
|
|
1832
|
-
|
|
1833
|
-
|
|
1834
|
-
|
|
1835
|
-
|
|
1836
|
-
|
|
1837
|
-
|
|
1838
|
-
|
|
1839
|
-
|
|
1840
|
-
|
|
1841
|
-
|
|
1842
|
-
|
|
1843
|
-
|
|
1844
|
-
|
|
1845
|
-
|
|
1846
|
-
|
|
1847
|
-
|
|
1848
|
-
|
|
1849
|
-
|
|
1850
|
-
|
|
1851
|
-
|
|
1852
|
-
|
|
1853
|
-
|
|
1854
|
-
|
|
1855
|
-
|
|
1856
|
-
|
|
1857
|
-
outcome,
|
|
1858
|
-
...taskFailure,
|
|
1859
|
-
splitTag: defaultSplit,
|
|
1860
|
-
scenarioId
|
|
1861
|
-
});
|
|
1862
|
-
}
|
|
1863
|
-
return runs;
|
|
1663
|
+
const { spans, defaultSplit = "holdout", experimentId = "otel-corpus" } = opts;
|
|
1664
|
+
const grouped = groupSpans(spans);
|
|
1665
|
+
const runs = [];
|
|
1666
|
+
for (const [groupKey, groupSpans] of grouped) {
|
|
1667
|
+
const root = findRoot(groupSpans);
|
|
1668
|
+
if (!root) continue;
|
|
1669
|
+
const measurements = summarizeExecutionMeasurements(groupSpans.map((span) => ({
|
|
1670
|
+
id: span.spanId,
|
|
1671
|
+
...span.parentSpanId ? { parentId: span.parentSpanId } : {},
|
|
1672
|
+
attributes: span.attributes,
|
|
1673
|
+
modelCall: isExplicitModelCall(span),
|
|
1674
|
+
aggregate: isExplicitAggregate(span)
|
|
1675
|
+
})));
|
|
1676
|
+
const callSpanIds = new Set(measurements.callSpanIds);
|
|
1677
|
+
const callSpans = groupSpans.filter((span) => callSpanIds.has(span.spanId));
|
|
1678
|
+
const wallMs = unixNanoDurationMs(root.startTimeUnixNano, root.endTimeUnixNano);
|
|
1679
|
+
const model = readAttrString(callSpans, MODEL_KEYS) ?? readAttrString(groupSpans, MODEL_KEYS) ?? "unknown@unknown";
|
|
1680
|
+
const capturedCost = (measurements.cost.complete ? measurements.cost.value : void 0) ?? measurements.aggregate?.costUsd;
|
|
1681
|
+
const costUsd = capturedCost ?? null;
|
|
1682
|
+
const scenarioId = readConsistentScenarioId(groupKey, groupSpans) ?? groupKey;
|
|
1683
|
+
const promptHash = readAttrString(groupSpans, PROMPT_HASH_KEYS) ?? "sha256:unknown";
|
|
1684
|
+
const configHash = readAttrString(groupSpans, CONFIG_HASH_KEYS) ?? "sha256:unknown";
|
|
1685
|
+
const score = resolveTaskScore(groupKey, groupSpans, opts.scoreForRun);
|
|
1686
|
+
const taskFailure = readTaskFailureLabels(groupSpans.filter((span) => !span.parentSpanId && isTerminalRootCandidate(span)), `fromOtelSpans: run '${groupKey}'`);
|
|
1687
|
+
const rawNumeric = collectNumericAttrs(groupSpans);
|
|
1688
|
+
const errorSummary = summarizeTraceErrors(groupSpans.map((span) => ({
|
|
1689
|
+
id: spanIdentity(span),
|
|
1690
|
+
...span.parentSpanId ? { parentId: parentIdentity(span) } : {},
|
|
1691
|
+
role: errorRoleForSpan(span),
|
|
1692
|
+
error: span.status?.code === "ERROR",
|
|
1693
|
+
processRoot: !span.parentSpanId && isTerminalRootCandidate(span)
|
|
1694
|
+
})));
|
|
1695
|
+
rawNumeric.error_span_count = errorSummary.total;
|
|
1696
|
+
rawNumeric.execution_error_count = errorSummary.execution;
|
|
1697
|
+
rawNumeric.process_error_count = errorSummary.process;
|
|
1698
|
+
rawNumeric.guardrail_error_count = errorSummary.guardrail;
|
|
1699
|
+
rawNumeric.judge_error_count = errorSummary.evaluation;
|
|
1700
|
+
rawNumeric.propagated_error_count = errorSummary.propagated;
|
|
1701
|
+
rawNumeric.unclassified_error_count = errorSummary.unclassified;
|
|
1702
|
+
rawNumeric.llm_span_count = measurements.modelCallCount;
|
|
1703
|
+
if (measurements.cost.value !== void 0 && !measurements.cost.complete) rawNumeric.partial_observed_cost_usd = measurements.cost.value;
|
|
1704
|
+
recordAggregateMeasurements(rawNumeric, measurements.aggregate);
|
|
1705
|
+
const judgeScores = score !== void 0 ? {
|
|
1706
|
+
perJudge: { "otel-derived": { score } },
|
|
1707
|
+
perDimMean: { score },
|
|
1708
|
+
composite: score
|
|
1709
|
+
} : void 0;
|
|
1710
|
+
const terminalOutcome = terminalOutcomeFromRoots(groupSpans);
|
|
1711
|
+
const failedRoot = terminalOutcome === "failed" ? groupSpans.find((span) => !span.parentSpanId && isTerminalRootCandidate(span) && span.status?.code === "ERROR") : void 0;
|
|
1712
|
+
const outcome = {
|
|
1713
|
+
raw: rawNumeric,
|
|
1714
|
+
...judgeScores ? { judgeScores } : {}
|
|
1715
|
+
};
|
|
1716
|
+
if (score !== void 0) if (defaultSplit === "holdout") outcome.holdoutScore = score;
|
|
1717
|
+
else outcome.searchScore = score;
|
|
1718
|
+
runs.push({
|
|
1719
|
+
runId: groupKey,
|
|
1720
|
+
experimentId,
|
|
1721
|
+
candidateId: root.attributes["tangle.candidateId"] ?? "otel-default",
|
|
1722
|
+
seed: 0,
|
|
1723
|
+
model,
|
|
1724
|
+
promptHash,
|
|
1725
|
+
configHash,
|
|
1726
|
+
commitSha: root.attributes["tangle.commit_sha"] ?? "unknown",
|
|
1727
|
+
wallMs,
|
|
1728
|
+
costUsd,
|
|
1729
|
+
costProvenance: capturedCost === void 0 ? {
|
|
1730
|
+
kind: "uncaptured",
|
|
1731
|
+
usd: null
|
|
1732
|
+
} : {
|
|
1733
|
+
kind: "observed",
|
|
1734
|
+
usd: capturedCost
|
|
1735
|
+
},
|
|
1736
|
+
tokenUsage: measurements.tokenUsage,
|
|
1737
|
+
terminalOutcome,
|
|
1738
|
+
...failedRoot ? { terminalFailureReason: failedRoot.status?.message ?? failedRoot.name } : {},
|
|
1739
|
+
outcome,
|
|
1740
|
+
...taskFailure,
|
|
1741
|
+
splitTag: defaultSplit,
|
|
1742
|
+
scenarioId
|
|
1743
|
+
});
|
|
1744
|
+
}
|
|
1745
|
+
return runs;
|
|
1864
1746
|
}
|
|
1865
1747
|
function terminalOutcomeFromRoots(spans) {
|
|
1866
|
-
|
|
1867
|
-
|
|
1868
|
-
|
|
1869
|
-
|
|
1870
|
-
|
|
1748
|
+
const roots = spans.filter((span) => !span.parentSpanId && isTerminalRootCandidate(span));
|
|
1749
|
+
if (roots.length !== 1) return "unknown";
|
|
1750
|
+
if (roots[0].status?.code === "OK") return "succeeded";
|
|
1751
|
+
if (roots[0].status?.code === "ERROR") return "failed";
|
|
1752
|
+
return "unknown";
|
|
1871
1753
|
}
|
|
1872
1754
|
function isTerminalRootCandidate(span) {
|
|
1873
|
-
|
|
1874
|
-
|
|
1875
|
-
|
|
1876
|
-
}
|
|
1877
|
-
return true;
|
|
1755
|
+
const role = errorRoleForSpan(span);
|
|
1756
|
+
if (role === "LLM" || role === "TOOL" || role === "GUARDRAIL" || role === "EVALUATOR") return false;
|
|
1757
|
+
return true;
|
|
1878
1758
|
}
|
|
1879
1759
|
function readSpanKind(span) {
|
|
1880
|
-
|
|
1760
|
+
return readAttrString([span], [...SPAN_KIND_ATTR_KEYS, "span.kind"])?.toUpperCase();
|
|
1881
1761
|
}
|
|
1882
1762
|
function errorRoleForSpan(span) {
|
|
1883
|
-
|
|
1884
|
-
|
|
1885
|
-
|
|
1886
|
-
|
|
1887
|
-
|
|
1763
|
+
return classifyOtlpSpanRole({
|
|
1764
|
+
kind: readSpanKind(span),
|
|
1765
|
+
name: span.name,
|
|
1766
|
+
attributes: span.attributes
|
|
1767
|
+
});
|
|
1888
1768
|
}
|
|
1889
1769
|
function spanIdentity(span) {
|
|
1890
|
-
|
|
1770
|
+
return `${span.traceId}:${span.spanId}`;
|
|
1891
1771
|
}
|
|
1892
1772
|
function parentIdentity(span) {
|
|
1893
|
-
|
|
1773
|
+
return `${span.traceId}:${span.parentSpanId}`;
|
|
1894
1774
|
}
|
|
1895
1775
|
function isExplicitModelCall(span) {
|
|
1896
|
-
|
|
1897
|
-
|
|
1898
|
-
|
|
1899
|
-
|
|
1900
|
-
|
|
1776
|
+
return isOtlpModelCall({
|
|
1777
|
+
kind: readSpanKind(span),
|
|
1778
|
+
name: span.name,
|
|
1779
|
+
attributes: span.attributes
|
|
1780
|
+
});
|
|
1901
1781
|
}
|
|
1902
1782
|
function isExplicitAggregate(span) {
|
|
1903
|
-
|
|
1904
|
-
|
|
1783
|
+
const kind = readSpanKind(span);
|
|
1784
|
+
return kind !== void 0 && kind !== "LLM";
|
|
1905
1785
|
}
|
|
1906
1786
|
function groupSpans(spans) {
|
|
1907
|
-
|
|
1908
|
-
|
|
1909
|
-
|
|
1910
|
-
|
|
1911
|
-
|
|
1912
|
-
|
|
1913
|
-
|
|
1914
|
-
|
|
1787
|
+
const m = /* @__PURE__ */ new Map();
|
|
1788
|
+
for (const span of spans) {
|
|
1789
|
+
const key = span["tangle.runId"] ?? span.traceId;
|
|
1790
|
+
const list = m.get(key) ?? [];
|
|
1791
|
+
list.push(span);
|
|
1792
|
+
m.set(key, list);
|
|
1793
|
+
}
|
|
1794
|
+
return m;
|
|
1915
1795
|
}
|
|
1916
1796
|
function findRoot(group) {
|
|
1917
|
-
|
|
1918
|
-
|
|
1919
|
-
|
|
1920
|
-
return orderSpans(pool)[0];
|
|
1797
|
+
const structuralRoots = group.filter((span) => !span.parentSpanId);
|
|
1798
|
+
const terminalRoots = structuralRoots.filter(isTerminalRootCandidate);
|
|
1799
|
+
return orderSpans(terminalRoots.length > 0 ? terminalRoots : structuralRoots.length > 0 ? structuralRoots : group)[0];
|
|
1921
1800
|
}
|
|
1922
1801
|
function readAttrString(spans, keys) {
|
|
1923
|
-
|
|
1924
|
-
|
|
1925
|
-
|
|
1926
|
-
|
|
1927
|
-
}
|
|
1928
|
-
}
|
|
1929
|
-
return void 0;
|
|
1802
|
+
for (const span of spans) for (const key of keys) {
|
|
1803
|
+
const v = span.attributes[key];
|
|
1804
|
+
if (typeof v === "string" && v.length > 0) return v;
|
|
1805
|
+
}
|
|
1930
1806
|
}
|
|
1931
1807
|
function readConsistentScenarioId(runId, spans) {
|
|
1932
|
-
|
|
1933
|
-
|
|
1934
|
-
|
|
1935
|
-
|
|
1936
|
-
|
|
1937
|
-
|
|
1938
|
-
|
|
1939
|
-
|
|
1940
|
-
|
|
1941
|
-
`fromOtelSpans: conflicting scenario ids for run '${runId}': ${[...values].sort().join(", ")}`
|
|
1942
|
-
);
|
|
1943
|
-
}
|
|
1944
|
-
return values.values().next().value;
|
|
1808
|
+
const values = /* @__PURE__ */ new Set();
|
|
1809
|
+
for (const span of spans) {
|
|
1810
|
+
const topLevel = span["tangle.scenarioId"];
|
|
1811
|
+
if (typeof topLevel === "string" && topLevel.length > 0) values.add(topLevel);
|
|
1812
|
+
const attribute = span.attributes["tangle.scenarioId"];
|
|
1813
|
+
if (typeof attribute === "string" && attribute.length > 0) values.add(attribute);
|
|
1814
|
+
}
|
|
1815
|
+
if (values.size > 1) throw new ValidationError(`fromOtelSpans: conflicting scenario ids for run '${runId}': ${[...values].sort().join(", ")}`);
|
|
1816
|
+
return values.values().next().value;
|
|
1945
1817
|
}
|
|
1946
1818
|
function resolveTaskScore(runId, spans, scoreForRun) {
|
|
1947
|
-
|
|
1948
|
-
|
|
1949
|
-
|
|
1950
|
-
|
|
1951
|
-
|
|
1952
|
-
|
|
1953
|
-
|
|
1954
|
-
|
|
1955
|
-
|
|
1956
|
-
|
|
1957
|
-
|
|
1958
|
-
|
|
1959
|
-
|
|
1960
|
-
|
|
1961
|
-
|
|
1962
|
-
|
|
1963
|
-
|
|
1964
|
-
|
|
1965
|
-
|
|
1966
|
-
|
|
1967
|
-
|
|
1968
|
-
|
|
1969
|
-
|
|
1970
|
-
|
|
1971
|
-
|
|
1972
|
-
|
|
1973
|
-
|
|
1974
|
-
|
|
1975
|
-
|
|
1976
|
-
const details = sources.map((source) => `${source.label}=${source.value}`).join(", ");
|
|
1977
|
-
throw new ValidationError(
|
|
1978
|
-
`fromOtelSpans: conflicting task-quality scores for run '${runId}': ${details}`
|
|
1979
|
-
);
|
|
1980
|
-
}
|
|
1981
|
-
return score;
|
|
1819
|
+
const orderedSpans = orderSpans(spans);
|
|
1820
|
+
const sources = [];
|
|
1821
|
+
if (scoreForRun) {
|
|
1822
|
+
const supplied = scoreForRun(runId, orderedSpans);
|
|
1823
|
+
if (supplied !== void 0) {
|
|
1824
|
+
if (typeof supplied !== "number" || !Number.isFinite(supplied)) throw new ValidationError(`fromOtelSpans: scoreForRun returned a non-finite number for run '${runId}'`);
|
|
1825
|
+
sources.push({
|
|
1826
|
+
label: "scoreForRun",
|
|
1827
|
+
value: supplied
|
|
1828
|
+
});
|
|
1829
|
+
}
|
|
1830
|
+
}
|
|
1831
|
+
for (const span of orderedSpans) {
|
|
1832
|
+
const role = errorRoleForSpan(span);
|
|
1833
|
+
if (span.parentSpanId && role !== "EVALUATOR") continue;
|
|
1834
|
+
if (role === "EVALUATOR" && span.status?.code === "ERROR") continue;
|
|
1835
|
+
for (const key of TASK_SCORE_ATTR_KEYS) {
|
|
1836
|
+
if (!Object.hasOwn(span.attributes, key)) continue;
|
|
1837
|
+
sources.push({
|
|
1838
|
+
label: `span '${span.spanId}' attribute '${key}'`,
|
|
1839
|
+
value: parseTaskScoreAttribute(runId, span.spanId, key, span.attributes[key])
|
|
1840
|
+
});
|
|
1841
|
+
}
|
|
1842
|
+
}
|
|
1843
|
+
if (sources.length === 0) return void 0;
|
|
1844
|
+
sources.sort((left, right) => left.label.localeCompare(right.label));
|
|
1845
|
+
const score = sources[0].value;
|
|
1846
|
+
if (sources.some((source) => source.value !== score)) throw new ValidationError(`fromOtelSpans: conflicting task-quality scores for run '${runId}': ${sources.map((source) => `${source.label}=${source.value}`).join(", ")}`);
|
|
1847
|
+
return score;
|
|
1982
1848
|
}
|
|
1983
1849
|
function parseTaskScoreAttribute(runId, spanId, key, value) {
|
|
1984
|
-
|
|
1985
|
-
|
|
1986
|
-
|
|
1987
|
-
|
|
1988
|
-
|
|
1989
|
-
|
|
1990
|
-
|
|
1991
|
-
const parsed = Number(value);
|
|
1992
|
-
if (Number.isFinite(parsed)) return parsed;
|
|
1993
|
-
} else if (typeof value === "number" && Number.isFinite(value)) {
|
|
1994
|
-
return value;
|
|
1995
|
-
}
|
|
1996
|
-
throw new ValidationError(
|
|
1997
|
-
`fromOtelSpans: ${source} is not a finite task-quality score for run '${runId}'`
|
|
1998
|
-
);
|
|
1850
|
+
const source = `span '${spanId}' attribute '${key}'`;
|
|
1851
|
+
if (typeof value === "string") {
|
|
1852
|
+
if (value.trim().length === 0) throw new ValidationError(`fromOtelSpans: ${source} is blank for run '${runId}'; task quality must be finite`);
|
|
1853
|
+
const parsed = Number(value);
|
|
1854
|
+
if (Number.isFinite(parsed)) return parsed;
|
|
1855
|
+
} else if (typeof value === "number" && Number.isFinite(value)) return value;
|
|
1856
|
+
throw new ValidationError(`fromOtelSpans: ${source} is not a finite task-quality score for run '${runId}'`);
|
|
1999
1857
|
}
|
|
2000
1858
|
function orderSpans(spans) {
|
|
2001
|
-
|
|
2002
|
-
(left, right) => compareUnixNano(left.startTimeUnixNano, right.startTimeUnixNano) || left.spanId.localeCompare(right.spanId)
|
|
2003
|
-
);
|
|
1859
|
+
return [...spans].sort((left, right) => compareUnixNano(left.startTimeUnixNano, right.startTimeUnixNano) || left.spanId.localeCompare(right.spanId));
|
|
2004
1860
|
}
|
|
2005
1861
|
function parseUnixNano(value) {
|
|
2006
|
-
|
|
2007
|
-
|
|
2008
|
-
|
|
2009
|
-
|
|
2010
|
-
|
|
1862
|
+
try {
|
|
1863
|
+
return BigInt(value);
|
|
1864
|
+
} catch {
|
|
1865
|
+
throw new ValidationError(`fromOtelSpans: invalid Unix nanosecond timestamp '${value}'`);
|
|
1866
|
+
}
|
|
2011
1867
|
}
|
|
2012
1868
|
function compareUnixNano(left, right) {
|
|
2013
|
-
|
|
2014
|
-
|
|
2015
|
-
|
|
1869
|
+
const leftValue = parseUnixNano(left);
|
|
1870
|
+
const rightValue = parseUnixNano(right);
|
|
1871
|
+
return leftValue < rightValue ? -1 : leftValue > rightValue ? 1 : 0;
|
|
2016
1872
|
}
|
|
2017
1873
|
function unixNanoDurationMs(start, end) {
|
|
2018
|
-
|
|
2019
|
-
|
|
2020
|
-
|
|
2021
|
-
|
|
2022
|
-
|
|
2023
|
-
|
|
2024
|
-
|
|
2025
|
-
}
|
|
2026
|
-
return value;
|
|
1874
|
+
const delta = parseUnixNano(end) - parseUnixNano(start);
|
|
1875
|
+
if (delta <= 0n) return 0;
|
|
1876
|
+
const wholeMs = delta / 1000000n;
|
|
1877
|
+
const fractionalMs = delta % 1000000n;
|
|
1878
|
+
const value = Number(wholeMs) + Number(fractionalMs) / 1e6;
|
|
1879
|
+
if (!Number.isSafeInteger(Number(wholeMs))) throw new ValidationError("fromOtelSpans: span duration exceeds the safe millisecond range");
|
|
1880
|
+
return value;
|
|
2027
1881
|
}
|
|
2028
1882
|
function collectNumericAttrs(spans) {
|
|
2029
|
-
|
|
2030
|
-
|
|
2031
|
-
|
|
2032
|
-
|
|
2033
|
-
|
|
2034
|
-
|
|
2035
|
-
|
|
2036
|
-
}
|
|
2037
|
-
export {
|
|
2038
|
-
FileSystemOutcomeStore,
|
|
2039
|
-
InMemoryOutcomeStore,
|
|
2040
|
-
REFERENCE_EQUIVALENCE_INPUT_LIMITS,
|
|
2041
|
-
REFERENCE_EQUIVALENCE_JUDGE_VERSION,
|
|
2042
|
-
SelfImproveRunError,
|
|
2043
|
-
analyzeRuns,
|
|
2044
|
-
buildDefaultAnalystRegistry,
|
|
2045
|
-
buildEvidenceVector,
|
|
2046
|
-
campaignSplitDigest,
|
|
2047
|
-
compareOptimizationMethods,
|
|
2048
|
-
composeGate,
|
|
2049
|
-
createChatClient,
|
|
2050
|
-
createReferenceEquivalenceJudge,
|
|
2051
|
-
defaultProductionGate,
|
|
2052
|
-
defineAgentEval,
|
|
2053
|
-
diffGenerations,
|
|
2054
|
-
diffRunBaselineToWinner,
|
|
2055
|
-
diffRuns,
|
|
2056
|
-
evalReportingSuite,
|
|
2057
|
-
evaluatePairedMeasurements,
|
|
2058
|
-
externalTextOptimizationMethod,
|
|
2059
|
-
fromClaudeCodeSession,
|
|
2060
|
-
fromCodexSession,
|
|
2061
|
-
fromFeedbackTable,
|
|
2062
|
-
fromKimiCodeSession,
|
|
2063
|
-
fromOpenCodeSession,
|
|
2064
|
-
fromOtelSpans,
|
|
2065
|
-
fromPiSession,
|
|
2066
|
-
fromPigraphSession,
|
|
2067
|
-
fromRunRecordDir,
|
|
2068
|
-
fsCampaignStorage,
|
|
2069
|
-
gepaOptimizationMethod,
|
|
2070
|
-
heldOutGate,
|
|
2071
|
-
inMemoryCampaignStorage,
|
|
2072
|
-
llmJudge,
|
|
2073
|
-
measuredComparisonFromCandidateExperiment,
|
|
2074
|
-
observeCodeAgentSession,
|
|
2075
|
-
paretoPolicy,
|
|
2076
|
-
paretoSignificanceGate,
|
|
2077
|
-
parseAgentTrace,
|
|
2078
|
-
parseCodeAgentJsonl,
|
|
2079
|
-
partitionRunsByAuthoringModel,
|
|
2080
|
-
runCampaign,
|
|
2081
|
-
runCandidateExperiment,
|
|
2082
|
-
runEval,
|
|
2083
|
-
runImprovementLoop,
|
|
2084
|
-
runReferenceEquivalenceJudge,
|
|
2085
|
-
sealCandidateBenchmarkSuite,
|
|
2086
|
-
sealCandidateBenchmarkTask,
|
|
2087
|
-
sealCandidateExperiment,
|
|
2088
|
-
selfImprove,
|
|
2089
|
-
skillOptOptimizationMethod,
|
|
2090
|
-
summarizeExecution,
|
|
2091
|
-
verifyCandidateBenchmarkSuite,
|
|
2092
|
-
verifyCandidateBenchmarkSuiteInputs,
|
|
2093
|
-
verifyCandidateBenchmarkTask,
|
|
2094
|
-
verifyCandidateExperiment,
|
|
2095
|
-
verifyCandidateExperimentComparison
|
|
2096
|
-
};
|
|
1883
|
+
const raw = {};
|
|
1884
|
+
for (const span of spans) for (const [k, v] of Object.entries(span.attributes)) if (typeof v === "number" && Number.isFinite(v)) raw[k] = v;
|
|
1885
|
+
return raw;
|
|
1886
|
+
}
|
|
1887
|
+
//#endregion
|
|
1888
|
+
export { FileSystemOutcomeStore, InMemoryOutcomeStore, REFERENCE_EQUIVALENCE_INPUT_LIMITS, REFERENCE_EQUIVALENCE_JUDGE_VERSION, SelfImproveRunError, analyzeRuns, buildDefaultAnalystRegistry, buildEvidenceVector, campaignSplitDigest, compareOptimizationMethods, composeGate, createChatClient, createReferenceEquivalenceJudge, defaultProductionGate, defineAgentEval, diffGenerations, diffRunBaselineToWinner, diffRuns, evalReportingSuite, evaluatePairedMeasurements, externalTextOptimizationMethod, fromClaudeCodeSession, fromCodexSession, fromFeedbackTable, fromKimiCodeSession, fromOpenCodeSession, fromOtelSpans, fromPiSession, fromPigraphSession, fromRunRecordDir, fsCampaignStorage, gepaOptimizationMethod, heldOutGate, inMemoryCampaignStorage, llmJudge, measuredComparisonFromCandidateExperiment, observeCodeAgentSession, paretoPolicy, paretoSignificanceGate, parseAgentTrace, parseCodeAgentJsonl, partitionRunsByAuthoringModel, runCampaign, runCandidateExperiment, runEval, runImprovementLoop, runReferenceEquivalenceJudge, sealCandidateBenchmarkSuite, sealCandidateBenchmarkTask, sealCandidateExperiment, selfImprove, skillOptOptimizationMethod, summarizeExecution, verifyCandidateBenchmarkSuite, verifyCandidateBenchmarkSuiteInputs, verifyCandidateBenchmarkTask, verifyCandidateExperiment, verifyCandidateExperimentComparison };
|
|
1889
|
+
|
|
2097
1890
|
//# sourceMappingURL=index.js.map
|