@tangle-network/agent-eval 0.129.0 → 0.130.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +21 -0
- package/README.md +2 -0
- package/dist/active-curriculum-C4mk67HP.js +214 -0
- package/dist/active-curriculum-C4mk67HP.js.map +1 -0
- package/dist/adversarial-smnADNFS.d.ts +21 -0
- package/dist/adversarial-smnADNFS.d.ts.map +1 -0
- package/dist/analyst/index.d.ts +81 -2872
- package/dist/analyst/index.d.ts.map +1 -0
- package/dist/analyst/index.js +319 -360
- package/dist/analyst/index.js.map +1 -1
- package/dist/analyst-BkTS3C58.d.ts +89 -0
- package/dist/analyst-BkTS3C58.d.ts.map +1 -0
- package/dist/analyst-LsnNpSkm.js +152 -0
- package/dist/analyst-LsnNpSkm.js.map +1 -0
- package/dist/analyze-runs-C1CavBMk.js +1067 -0
- package/dist/analyze-runs-C1CavBMk.js.map +1 -0
- package/dist/analyze-runs-FsgYCinh.d.ts +72 -0
- package/dist/analyze-runs-FsgYCinh.d.ts.map +1 -0
- package/dist/attribute-vocabulary-DLJ6303h.d.ts +54 -0
- package/dist/attribute-vocabulary-DLJ6303h.d.ts.map +1 -0
- package/dist/authenticity/index.d.ts +81 -79
- package/dist/authenticity/index.d.ts.map +1 -0
- package/dist/authenticity/index.js +209 -193
- package/dist/authenticity/index.js.map +1 -1
- package/dist/baseline-HsBvw_dk.js +550 -0
- package/dist/baseline-HsBvw_dk.js.map +1 -0
- package/dist/baseline-hG3K85h4.d.ts +125 -0
- package/dist/baseline-hG3K85h4.d.ts.map +1 -0
- package/dist/belief-state/index.d.ts +448 -1188
- package/dist/belief-state/index.d.ts.map +1 -0
- package/dist/belief-state/index.js +1617 -1709
- package/dist/belief-state/index.js.map +1 -1
- package/dist/benchmarks/index.d.ts +2 -891
- package/dist/benchmarks/index.js +2 -60
- package/dist/benchmarks-BJgDGkAD.js +754 -0
- package/dist/benchmarks-BJgDGkAD.js.map +1 -0
- package/dist/builder-eval/index.d.ts +150 -662
- package/dist/builder-eval/index.d.ts.map +1 -0
- package/dist/builder-eval/index.js +356 -345
- package/dist/builder-eval/index.js.map +1 -1
- package/dist/calibration-CNWWA6K8.js +94 -0
- package/dist/calibration-CNWWA6K8.js.map +1 -0
- package/dist/campaign/index.d.ts +5 -6381
- package/dist/campaign/index.js +3 -213
- package/dist/campaign-aKJt6emI.js +3886 -0
- package/dist/campaign-aKJt6emI.js.map +1 -0
- package/dist/cli.d.ts +1 -1
- package/dist/cli.js +164 -175
- package/dist/cli.js.map +1 -1
- package/dist/client-C97NMzqi.d.ts +581 -0
- package/dist/client-C97NMzqi.d.ts.map +1 -0
- package/dist/client-CYzbdJOZ.js +637 -0
- package/dist/client-CYzbdJOZ.js.map +1 -0
- package/dist/code-agent-session-BjkMTQ7H.js +1390 -0
- package/dist/code-agent-session-BjkMTQ7H.js.map +1 -0
- package/dist/code-agent-session-DqqgOJaz.d.ts +143 -0
- package/dist/code-agent-session-DqqgOJaz.d.ts.map +1 -0
- package/dist/concurrency-DIxRZF_J.js +85 -0
- package/dist/concurrency-DIxRZF_J.js.map +1 -0
- package/dist/contract/index.d.ts +645 -5565
- package/dist/contract/index.d.ts.map +1 -0
- package/dist/contract/index.js +1730 -1938
- package/dist/contract/index.js.map +1 -1
- package/dist/control.d.ts +3 -1030
- package/dist/control.js +2 -33
- package/dist/cost-ledger-DIgQUFZZ.js +801 -0
- package/dist/cost-ledger-DIgQUFZZ.js.map +1 -0
- package/dist/cost-ledger-Dye6jCgg.d.ts +236 -0
- package/dist/cost-ledger-Dye6jCgg.d.ts.map +1 -0
- package/dist/dataset-BvtnC8Dc.d.ts +115 -0
- package/dist/dataset-BvtnC8Dc.d.ts.map +1 -0
- package/dist/default-registry-C-vFCSEc.js +2579 -0
- package/dist/default-registry-C-vFCSEc.js.map +1 -0
- package/dist/default-registry-CNPo-Vsb.d.ts +540 -0
- package/dist/default-registry-CNPo-Vsb.d.ts.map +1 -0
- package/dist/emitter-CPBAhxum.js +266 -0
- package/dist/emitter-CPBAhxum.js.map +1 -0
- package/dist/emitter-DGQGoLyj.d.ts +113 -0
- package/dist/emitter-DGQGoLyj.d.ts.map +1 -0
- package/dist/errors-8YnH8WlF.js +64 -0
- package/dist/errors-8YnH8WlF.js.map +1 -0
- package/dist/errors-CEk209JS.d.ts +76 -0
- package/dist/errors-CEk209JS.d.ts.map +1 -0
- package/dist/eval-campaign-DEm6c8ru.js +349 -0
- package/dist/eval-campaign-DEm6c8ru.js.map +1 -0
- package/dist/execution-tracks-CpgFPpS5.js +93 -0
- package/dist/execution-tracks-CpgFPpS5.js.map +1 -0
- package/dist/exporters-q9iL-2Jf.js +148 -0
- package/dist/exporters-q9iL-2Jf.js.map +1 -0
- package/dist/extract-usage-BrQ8mCLX.js +155 -0
- package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
- package/dist/failure-cluster-CqcvCcdR.d.ts +59 -0
- package/dist/failure-cluster-CqcvCcdR.d.ts.map +1 -0
- package/dist/feedback-trajectory-CVaeREXV.d.ts +340 -0
- package/dist/feedback-trajectory-CVaeREXV.d.ts.map +1 -0
- package/dist/fuzz.d.ts +320 -646
- package/dist/fuzz.d.ts.map +1 -0
- package/dist/fuzz.js +670 -618
- package/dist/fuzz.js.map +1 -1
- package/dist/hf-dataset-DBJXXoY1.js +763 -0
- package/dist/hf-dataset-DBJXXoY1.js.map +1 -0
- package/dist/hosted/index.d.ts +11 -831
- package/dist/hosted/index.d.ts.map +1 -0
- package/dist/hosted/index.js +2 -37
- package/dist/index-2JJSA6-r2.d.ts +926 -0
- package/dist/index-2JJSA6-r2.d.ts.map +1 -0
- package/dist/index-6N0aYmpW.d.ts +217 -0
- package/dist/index-6N0aYmpW.d.ts.map +1 -0
- package/dist/index-BAvgST_9.d.ts +131 -0
- package/dist/index-BAvgST_9.d.ts.map +1 -0
- package/dist/index-BvnJuTGD.d.ts +68 -0
- package/dist/index-BvnJuTGD.d.ts.map +1 -0
- package/dist/index-C61Wi7yg.d.ts +547 -0
- package/dist/index-C61Wi7yg.d.ts.map +1 -0
- package/dist/index-CD_WZ_Xr.d.ts +2250 -0
- package/dist/index-CD_WZ_Xr.d.ts.map +1 -0
- package/dist/index-DSC51roc.d.ts +102 -0
- package/dist/index-DSC51roc.d.ts.map +1 -0
- package/dist/index-Em67JBjs.d.ts +335 -0
- package/dist/index-Em67JBjs.d.ts.map +1 -0
- package/dist/index.d.ts +3755 -15555
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +11182 -11216
- package/dist/index.js.map +1 -1
- package/dist/integrity-BzRbCHzi.js +424 -0
- package/dist/integrity-BzRbCHzi.js.map +1 -0
- package/dist/integrity-rmVhXWA7.d.ts +61 -0
- package/dist/integrity-rmVhXWA7.d.ts.map +1 -0
- package/dist/judge-calibration-DFtEMlde.d.ts +146 -0
- package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
- package/dist/ledger-core/index.d.ts +2 -0
- package/dist/ledger-core/index.js +2 -0
- package/dist/ledger-core-DtZz1RG0.js +388 -0
- package/dist/ledger-core-DtZz1RG0.js.map +1 -0
- package/dist/llm-client--GR4JbZE.js +687 -0
- package/dist/llm-client--GR4JbZE.js.map +1 -0
- package/dist/llm-client-B_nIBlYo.d.ts +290 -0
- package/dist/llm-client-B_nIBlYo.d.ts.map +1 -0
- package/dist/matrix/index.d.ts +3 -155
- package/dist/matrix/index.js +2 -8
- package/dist/matrix-BzQnu2S6.js +270 -0
- package/dist/matrix-BzQnu2S6.js.map +1 -0
- package/dist/meta-eval/index.d.ts +4 -1027
- package/dist/meta-eval/index.js +393 -390
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/metrics-C9YY1OcL.js +239 -0
- package/dist/metrics-C9YY1OcL.js.map +1 -0
- package/dist/mint-yN2M2eh0.js +201 -0
- package/dist/mint-yN2M2eh0.js.map +1 -0
- package/dist/multi-layer-verifier-BHY1gWAc.d.ts +138 -0
- package/dist/multi-layer-verifier-BHY1gWAc.d.ts.map +1 -0
- package/dist/multishot/index.d.ts +273 -480
- package/dist/multishot/index.d.ts.map +1 -0
- package/dist/multishot/index.js +595 -548
- package/dist/multishot/index.js.map +1 -1
- package/dist/off-policy-DvgzvtIx.js +220 -0
- package/dist/off-policy-DvgzvtIx.js.map +1 -0
- package/dist/off-policy-mskQw8Mb.d.ts +153 -0
- package/dist/off-policy-mskQw8Mb.d.ts.map +1 -0
- package/dist/openapi.json +1 -1
- package/dist/opencode-sqlite-BGrHeDu3.js +318 -0
- package/dist/opencode-sqlite-BGrHeDu3.js.map +1 -0
- package/dist/outcome-store-BYHIuO0e.d.ts +65 -0
- package/dist/outcome-store-BYHIuO0e.d.ts.map +1 -0
- package/dist/outcome-store-ChBKlTd_.js +75 -0
- package/dist/outcome-store-ChBKlTd_.js.map +1 -0
- package/dist/paired-arms-D9D0wXj2.js +260 -0
- package/dist/paired-arms-D9D0wXj2.js.map +1 -0
- package/dist/pipelines/index.d.ts +95 -532
- package/dist/pipelines/index.d.ts.map +1 -0
- package/dist/pipelines/index.js +497 -478
- package/dist/pipelines/index.js.map +1 -1
- package/dist/pre-registration-DakwTRXk.js +96 -0
- package/dist/pre-registration-DakwTRXk.js.map +1 -0
- package/dist/propose-review-control-LhIWGzHl.js +1458 -0
- package/dist/propose-review-control-LhIWGzHl.js.map +1 -0
- package/dist/query-CJ_DX8vl.d.ts +28 -0
- package/dist/query-CJ_DX8vl.d.ts.map +1 -0
- package/dist/query-Di7eEQ79.js +83 -0
- package/dist/query-Di7eEQ79.js.map +1 -0
- package/dist/raw-provider-sink-BQd7mzyT.js +201 -0
- package/dist/raw-provider-sink-BQd7mzyT.js.map +1 -0
- package/dist/raw-provider-sink-BU29Sh8h.d.ts +134 -0
- package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +1 -0
- package/dist/redact-7Aq1ukl-.js +107 -0
- package/dist/redact-7Aq1ukl-.js.map +1 -0
- package/dist/release-report-Cz9NKH39.js +603 -0
- package/dist/release-report-Cz9NKH39.js.map +1 -0
- package/dist/release-report-mvB2G4_J.d.ts +244 -0
- package/dist/release-report-mvB2G4_J.d.ts.map +1 -0
- package/dist/replay-GnyotH0J.js +1741 -0
- package/dist/replay-GnyotH0J.js.map +1 -0
- package/dist/replay-OoidtG1E.d.ts +749 -0
- package/dist/replay-OoidtG1E.d.ts.map +1 -0
- package/dist/reporting.d.ts +6 -1312
- package/dist/reporting.js +6 -51
- package/dist/researcher-CwTdwXG1.d.ts +314 -0
- package/dist/researcher-CwTdwXG1.d.ts.map +1 -0
- package/dist/reward-hacking-eAnOsynk.d.ts +224 -0
- package/dist/reward-hacking-eAnOsynk.d.ts.map +1 -0
- package/dist/reward-hacking-qipEpKvY.js +596 -0
- package/dist/reward-hacking-qipEpKvY.js.map +1 -0
- package/dist/reward-nw2xZGZG.js +137 -0
- package/dist/reward-nw2xZGZG.js.map +1 -0
- package/dist/rl.d.ts +760 -4010
- package/dist/rl.d.ts.map +1 -0
- package/dist/rl.js +2325 -1958
- package/dist/rl.js.map +1 -1
- package/dist/rolldown-runtime-8H4AJuhK.js +14 -0
- package/dist/rollout/index.d.ts +3 -2087
- package/dist/rollout/index.js +8 -168
- package/dist/rollout-BOYjemfR.js +624 -0
- package/dist/rollout-BOYjemfR.js.map +1 -0
- package/dist/rubric-predictive-validity-B3xmbmS1.js +141 -0
- package/dist/rubric-predictive-validity-B3xmbmS1.js.map +1 -0
- package/dist/rubric-predictive-validity-Ku_clp_1.d.ts +74 -0
- package/dist/rubric-predictive-validity-Ku_clp_1.d.ts.map +1 -0
- package/dist/run-evidence-ClRX_8A9.d.ts +225 -0
- package/dist/run-evidence-ClRX_8A9.d.ts.map +1 -0
- package/dist/run-record-BuoE80Dq.js +467 -0
- package/dist/{chunk-56TAVBOK.js.map → run-record-BuoE80Dq.js.map} +1 -1
- package/dist/run-record-CnZu_gjl.d.ts +357 -0
- package/dist/run-record-CnZu_gjl.d.ts.map +1 -0
- package/dist/run-score-iEEAWiBY.js +41 -0
- package/dist/run-score-iEEAWiBY.js.map +1 -0
- package/dist/runtime-trajectory-1gyaTOoC.js +93 -0
- package/dist/runtime-trajectory-1gyaTOoC.js.map +1 -0
- package/dist/runtime-trajectory-BvSZcCHD.d.ts +50 -0
- package/dist/runtime-trajectory-BvSZcCHD.d.ts.map +1 -0
- package/dist/schema-BtVldJ3T.d.ts +206 -0
- package/dist/schema-BtVldJ3T.d.ts.map +1 -0
- package/dist/schema-C6DW4ZHR.js +821 -0
- package/dist/schema-C6DW4ZHR.js.map +1 -0
- package/dist/schema-CRhEY1SO.js +69 -0
- package/dist/schema-CRhEY1SO.js.map +1 -0
- package/dist/schema-Cef2cFmb.d.ts +408 -0
- package/dist/schema-Cef2cFmb.d.ts.map +1 -0
- package/dist/semantic-concept-judge-B6cWNJ2K.js +725 -0
- package/dist/semantic-concept-judge-B6cWNJ2K.js.map +1 -0
- package/dist/sequential-Br0mAPHA.js +148 -0
- package/dist/sequential-Br0mAPHA.js.map +1 -0
- package/dist/sequential-CYwq6Ff_.d.ts +141 -0
- package/dist/sequential-CYwq6Ff_.d.ts.map +1 -0
- package/dist/series-convergence-CjO2QdRW.js +43 -0
- package/dist/series-convergence-CjO2QdRW.js.map +1 -0
- package/dist/series-convergence-ofsqPWhs.d.ts +35 -0
- package/dist/series-convergence-ofsqPWhs.d.ts.map +1 -0
- package/dist/server-m5D9cvnG.js +1040 -0
- package/dist/server-m5D9cvnG.js.map +1 -0
- package/dist/skill-usage-C_pXm7lP.d.ts +534 -0
- package/dist/skill-usage-C_pXm7lP.d.ts.map +1 -0
- package/dist/skillopt-optimization-method-CUmHkGbI.js +7718 -0
- package/dist/skillopt-optimization-method-CUmHkGbI.js.map +1 -0
- package/dist/skillopt-optimization-method-CWKVTnks.d.ts +1740 -0
- package/dist/skillopt-optimization-method-CWKVTnks.d.ts.map +1 -0
- package/dist/statistics-Cmj6nynr.d.ts +514 -0
- package/dist/statistics-Cmj6nynr.d.ts.map +1 -0
- package/dist/statistics-CnnxdpOg.js +1437 -0
- package/dist/statistics-CnnxdpOg.js.map +1 -0
- package/dist/store-CT9YIIve.d.ts +117 -0
- package/dist/store-CT9YIIve.d.ts.map +1 -0
- package/dist/store-CxJry_cs.d.ts +229 -0
- package/dist/store-CxJry_cs.d.ts.map +1 -0
- package/dist/storyboard/index.d.ts +84 -203
- package/dist/storyboard/index.d.ts.map +1 -0
- package/dist/storyboard/index.js +609 -542
- package/dist/storyboard/index.js.map +1 -1
- package/dist/summary-report-BNs5nmXI.js +862 -0
- package/dist/summary-report-BNs5nmXI.js.map +1 -0
- package/dist/summary-report-Cj9gdw4i.d.ts +407 -0
- package/dist/summary-report-Cj9gdw4i.d.ts.map +1 -0
- package/dist/supervisor-run/index.d.ts +2 -959
- package/dist/supervisor-run/index.js +2 -65
- package/dist/supervisor-run-_lnTLM3z.js +1679 -0
- package/dist/supervisor-run-_lnTLM3z.js.map +1 -0
- package/dist/task-failure-attributes-CQZlB3et.js +311 -0
- package/dist/task-failure-attributes-CQZlB3et.js.map +1 -0
- package/dist/test-graded-scenario-BsqWLmPt.js +318 -0
- package/dist/test-graded-scenario-BsqWLmPt.js.map +1 -0
- package/dist/test-graded-scenario-D1TaI2va.d.ts +141 -0
- package/dist/test-graded-scenario-D1TaI2va.d.ts.map +1 -0
- package/dist/tools-BmuN627J.js +1085 -0
- package/dist/tools-BmuN627J.js.map +1 -0
- package/dist/trace-attributes.d.ts +2 -52
- package/dist/trace-attributes.js +131 -61
- package/dist/trace-attributes.js.map +1 -1
- package/dist/traces.d.ts +12 -2365
- package/dist/traces.js +12 -252
- package/dist/trajectory-D_7rLrvE.js +56 -0
- package/dist/trajectory-D_7rLrvE.js.map +1 -0
- package/dist/types-DGsxbAEd.d.ts +387 -0
- package/dist/types-DGsxbAEd.d.ts.map +1 -0
- package/dist/types-k9tZGKUg.d.ts +640 -0
- package/dist/types-k9tZGKUg.d.ts.map +1 -0
- package/dist/verdict-Dps8_okt.d.ts +37 -0
- package/dist/verdict-Dps8_okt.d.ts.map +1 -0
- package/dist/wire/index.d.ts +702 -1173
- package/dist/wire/index.d.ts.map +1 -0
- package/dist/wire/index.js +2 -81
- package/docs/campaign-proposers.md +1 -0
- package/package.json +17 -9
- package/dist/benchmarks/index.js.map +0 -1
- package/dist/campaign/index.js.map +0 -1
- package/dist/chunk-2QU3YOPR.js +0 -7374
- package/dist/chunk-2QU3YOPR.js.map +0 -1
- package/dist/chunk-3OCR4R5I.js +0 -728
- package/dist/chunk-3OCR4R5I.js.map +0 -1
- package/dist/chunk-3RF76KTD.js +0 -84
- package/dist/chunk-3RF76KTD.js.map +0 -1
- package/dist/chunk-56TAVBOK.js +0 -698
- package/dist/chunk-5DTSBUL2.js +0 -159
- package/dist/chunk-5DTSBUL2.js.map +0 -1
- package/dist/chunk-7FO3TNPI.js +0 -232
- package/dist/chunk-7FO3TNPI.js.map +0 -1
- package/dist/chunk-7ZZMD7UK.js +0 -386
- package/dist/chunk-7ZZMD7UK.js.map +0 -1
- package/dist/chunk-BOD4O7OF.js +0 -40
- package/dist/chunk-BOD4O7OF.js.map +0 -1
- package/dist/chunk-BSO5JDQH.js +0 -2335
- package/dist/chunk-BSO5JDQH.js.map +0 -1
- package/dist/chunk-C6LXANRU.js +0 -1550
- package/dist/chunk-C6LXANRU.js.map +0 -1
- package/dist/chunk-DODXQREJ.js +0 -752
- package/dist/chunk-DODXQREJ.js.map +0 -1
- package/dist/chunk-DRYIUNWY.js +0 -622
- package/dist/chunk-DRYIUNWY.js.map +0 -1
- package/dist/chunk-E7QXT7SX.js +0 -183
- package/dist/chunk-E7QXT7SX.js.map +0 -1
- package/dist/chunk-EG66UGL4.js +0 -341
- package/dist/chunk-EG66UGL4.js.map +0 -1
- package/dist/chunk-FXTVJPYD.js +0 -576
- package/dist/chunk-FXTVJPYD.js.map +0 -1
- package/dist/chunk-G7MGMCZD.js +0 -153
- package/dist/chunk-G7MGMCZD.js.map +0 -1
- package/dist/chunk-GGE4NNQT.js +0 -65
- package/dist/chunk-GGE4NNQT.js.map +0 -1
- package/dist/chunk-H23X7XKK.js +0 -181
- package/dist/chunk-H23X7XKK.js.map +0 -1
- package/dist/chunk-HHWE3POT.js +0 -94
- package/dist/chunk-HHWE3POT.js.map +0 -1
- package/dist/chunk-HPWUNB47.js +0 -289
- package/dist/chunk-HPWUNB47.js.map +0 -1
- package/dist/chunk-IYCLP2N2.js +0 -766
- package/dist/chunk-IYCLP2N2.js.map +0 -1
- package/dist/chunk-JHCHEVET.js +0 -274
- package/dist/chunk-JHCHEVET.js.map +0 -1
- package/dist/chunk-JQSF5DQT.js +0 -701
- package/dist/chunk-JQSF5DQT.js.map +0 -1
- package/dist/chunk-K4DBDHLK.js +0 -158
- package/dist/chunk-K4DBDHLK.js.map +0 -1
- package/dist/chunk-K6N6XJJX.js +0 -306
- package/dist/chunk-K6N6XJJX.js.map +0 -1
- package/dist/chunk-M4YBQKIJ.js +0 -1040
- package/dist/chunk-M4YBQKIJ.js.map +0 -1
- package/dist/chunk-MA6HLL3S.js +0 -65
- package/dist/chunk-MA6HLL3S.js.map +0 -1
- package/dist/chunk-MAZ26DC7.js +0 -99
- package/dist/chunk-MAZ26DC7.js.map +0 -1
- package/dist/chunk-NPCTHQIO.js +0 -91
- package/dist/chunk-NPCTHQIO.js.map +0 -1
- package/dist/chunk-NY44NC4A.js +0 -1056
- package/dist/chunk-NY44NC4A.js.map +0 -1
- package/dist/chunk-OIUOT4QD.js +0 -44
- package/dist/chunk-OIUOT4QD.js.map +0 -1
- package/dist/chunk-ONWEPEDO.js +0 -57
- package/dist/chunk-ONWEPEDO.js.map +0 -1
- package/dist/chunk-OWN5NPMC.js +0 -152
- package/dist/chunk-OWN5NPMC.js.map +0 -1
- package/dist/chunk-P6FYH6K4.js +0 -1161
- package/dist/chunk-P6FYH6K4.js.map +0 -1
- package/dist/chunk-PC4UYEBM.js +0 -166
- package/dist/chunk-PC4UYEBM.js.map +0 -1
- package/dist/chunk-PC5DOSM7.js +0 -579
- package/dist/chunk-PC5DOSM7.js.map +0 -1
- package/dist/chunk-PXE2VKMX.js +0 -140
- package/dist/chunk-PXE2VKMX.js.map +0 -1
- package/dist/chunk-PZ5AY32C.js +0 -10
- package/dist/chunk-PZ5AY32C.js.map +0 -1
- package/dist/chunk-QB6BDBP2.js +0 -4464
- package/dist/chunk-QB6BDBP2.js.map +0 -1
- package/dist/chunk-RXHCETDZ.js +0 -536
- package/dist/chunk-RXHCETDZ.js.map +0 -1
- package/dist/chunk-RZTMDUO7.js +0 -49
- package/dist/chunk-RZTMDUO7.js.map +0 -1
- package/dist/chunk-SFLLL76A.js +0 -669
- package/dist/chunk-SFLLL76A.js.map +0 -1
- package/dist/chunk-SZLVEKMJ.js +0 -1446
- package/dist/chunk-SZLVEKMJ.js.map +0 -1
- package/dist/chunk-T4SQEITX.js +0 -95
- package/dist/chunk-T4SQEITX.js.map +0 -1
- package/dist/chunk-T6RLYGAD.js +0 -158
- package/dist/chunk-T6RLYGAD.js.map +0 -1
- package/dist/chunk-TJVT4QFF.js +0 -911
- package/dist/chunk-TJVT4QFF.js.map +0 -1
- package/dist/chunk-TQ7LNKZ3.js +0 -136
- package/dist/chunk-TQ7LNKZ3.js.map +0 -1
- package/dist/chunk-U4L7JRPZ.js +0 -1706
- package/dist/chunk-U4L7JRPZ.js.map +0 -1
- package/dist/chunk-U4PHLT2N.js +0 -419
- package/dist/chunk-U4PHLT2N.js.map +0 -1
- package/dist/chunk-VCZ5FQYW.js +0 -928
- package/dist/chunk-VCZ5FQYW.js.map +0 -1
- package/dist/chunk-VI2UW6B6.js +0 -162
- package/dist/chunk-VI2UW6B6.js.map +0 -1
- package/dist/chunk-VQMK5FMP.js +0 -247
- package/dist/chunk-VQMK5FMP.js.map +0 -1
- package/dist/chunk-WGXIEX7P.js +0 -116
- package/dist/chunk-WGXIEX7P.js.map +0 -1
- package/dist/chunk-WVATSFCP.js +0 -1553
- package/dist/chunk-WVATSFCP.js.map +0 -1
- package/dist/chunk-X4YIBDER.js +0 -1662
- package/dist/chunk-X4YIBDER.js.map +0 -1
- package/dist/chunk-YQN4ICPP.js +0 -355
- package/dist/chunk-YQN4ICPP.js.map +0 -1
- package/dist/chunk-ZET2UAYW.js +0 -89
- package/dist/chunk-ZET2UAYW.js.map +0 -1
- package/dist/chunk-ZHTZ4EYI.js +0 -1212
- package/dist/chunk-ZHTZ4EYI.js.map +0 -1
- package/dist/control.js.map +0 -1
- package/dist/hosted/index.js.map +0 -1
- package/dist/matrix/index.js.map +0 -1
- package/dist/reporting.js.map +0 -1
- package/dist/rollout/index.js.map +0 -1
- package/dist/run-campaign-OJJ7CZF4.js +0 -18
- package/dist/run-campaign-OJJ7CZF4.js.map +0 -1
- package/dist/supervisor-run/index.js.map +0 -1
- package/dist/traces.js.map +0 -1
- package/dist/wire/index.js.map +0 -1
package/dist/contract/index.js
CHANGED
|
@@ -1,2098 +1,1890 @@
|
|
|
1
|
-
import {
|
|
2
|
-
|
|
3
|
-
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
} from "../
|
|
11
|
-
import {
|
|
12
|
-
|
|
13
|
-
} from "../
|
|
14
|
-
import {
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
emitLoopProvenance,
|
|
28
|
-
externalTextOptimizationMethod,
|
|
29
|
-
gepaOptimizationMethod,
|
|
30
|
-
heldOutGate,
|
|
31
|
-
heldoutSignificance,
|
|
32
|
-
llmJudge,
|
|
33
|
-
loopProvenanceArgsFromResult,
|
|
34
|
-
paretoPolicy,
|
|
35
|
-
paretoSignificanceGate,
|
|
36
|
-
powerPreflight,
|
|
37
|
-
runEval,
|
|
38
|
-
runImprovementLoop,
|
|
39
|
-
runReferenceEquivalenceJudge,
|
|
40
|
-
skillOptOptimizationMethod,
|
|
41
|
-
surfaceContentHash,
|
|
42
|
-
surfaceHash
|
|
43
|
-
} from "../chunk-2QU3YOPR.js";
|
|
44
|
-
import {
|
|
45
|
-
campaignSplitDigest,
|
|
46
|
-
createRunCostLedger,
|
|
47
|
-
fsCampaignStorage,
|
|
48
|
-
inMemoryCampaignStorage,
|
|
49
|
-
resolveRunDir,
|
|
50
|
-
runCampaign
|
|
51
|
-
} from "../chunk-C6LXANRU.js";
|
|
52
|
-
import {
|
|
53
|
-
buildDefaultAnalystRegistry,
|
|
54
|
-
createChatClient
|
|
55
|
-
} from "../chunk-BSO5JDQH.js";
|
|
56
|
-
import "../chunk-HHWE3POT.js";
|
|
57
|
-
import "../chunk-WGXIEX7P.js";
|
|
58
|
-
import {
|
|
59
|
-
FileSystemOutcomeStore,
|
|
60
|
-
InMemoryOutcomeStore
|
|
61
|
-
} from "../chunk-3RF76KTD.js";
|
|
62
|
-
import "../chunk-EG66UGL4.js";
|
|
63
|
-
import {
|
|
64
|
-
campaignCellExecutionEvidence,
|
|
65
|
-
campaignCellJudgeDimensions,
|
|
66
|
-
campaignCellTaskScore,
|
|
67
|
-
campaignCellToRunRecord
|
|
68
|
-
} from "../chunk-E7QXT7SX.js";
|
|
69
|
-
import "../chunk-SFLLL76A.js";
|
|
70
|
-
import "../chunk-TJVT4QFF.js";
|
|
71
|
-
import "../chunk-7FO3TNPI.js";
|
|
72
|
-
import {
|
|
73
|
-
pairedBootstrap
|
|
74
|
-
} from "../chunk-ZHTZ4EYI.js";
|
|
75
|
-
import "../chunk-VCZ5FQYW.js";
|
|
76
|
-
import "../chunk-VI2UW6B6.js";
|
|
77
|
-
import {
|
|
78
|
-
readTaskFailureLabels,
|
|
79
|
-
recordAggregateMeasurements,
|
|
80
|
-
summarizeExecutionMeasurements,
|
|
81
|
-
summarizeTraceErrors
|
|
82
|
-
} from "../chunk-7ZZMD7UK.js";
|
|
83
|
-
import "../chunk-PXE2VKMX.js";
|
|
84
|
-
import "../chunk-ZET2UAYW.js";
|
|
85
|
-
import "../chunk-GGE4NNQT.js";
|
|
86
|
-
import {
|
|
87
|
-
classifyOtlpSpanRole,
|
|
88
|
-
isOtlpModelCall
|
|
89
|
-
} from "../chunk-P6FYH6K4.js";
|
|
90
|
-
import "../chunk-PC4UYEBM.js";
|
|
91
|
-
import {
|
|
92
|
-
modelHasSnapshot,
|
|
93
|
-
parseRunRecordSafe
|
|
94
|
-
} from "../chunk-56TAVBOK.js";
|
|
95
|
-
import "../chunk-MA6HLL3S.js";
|
|
96
|
-
import "../chunk-OIUOT4QD.js";
|
|
97
|
-
import {
|
|
98
|
-
ValidationError
|
|
99
|
-
} from "../chunk-ONWEPEDO.js";
|
|
100
|
-
import {
|
|
101
|
-
LLM_MODEL_ATTR_KEYS,
|
|
102
|
-
SPAN_KIND_ATTR_KEYS
|
|
103
|
-
} from "../chunk-K4DBDHLK.js";
|
|
104
|
-
import "../chunk-PZ5AY32C.js";
|
|
105
|
-
|
|
106
|
-
// src/contract/self-improve.ts
|
|
1
|
+
import { s as ValidationError } from "../errors-8YnH8WlF.js";
|
|
2
|
+
import { i as parseRunRecordSafe, r as modelHasSnapshot } from "../run-record-BuoE80Dq.js";
|
|
3
|
+
import { L as createChatClient, t as buildDefaultAnalystRegistry } from "../default-registry-C-vFCSEc.js";
|
|
4
|
+
import { LLM_MODEL_ATTR_KEYS, SPAN_KIND_ATTR_KEYS } from "../trace-attributes.js";
|
|
5
|
+
import { b as classifyOtlpSpanRole, x as isOtlpModelCall } from "../tools-BmuN627J.js";
|
|
6
|
+
import { B as surfaceContentHash, Ct as llmJudge, M as compareOptimizationMethods, O as composeGate, Q as inMemoryCampaignStorage, S as defaultProductionGate, T as heldoutSignificance, V as surfaceHash, X as createRunCostLedger, Y as runCampaign, Z as fsCampaignStorage, _ as buildEvidenceVector, a as emitLoopProvenance, b as powerPreflight, ct as campaignSplitDigest, d as runImprovementLoop, dt as REFERENCE_EQUIVALENCE_INPUT_LIMITS, ft as REFERENCE_EQUIVALENCE_JUDGE_VERSION, g as gepaOptimizationMethod, j as assertOptimizationResult, k as externalTextOptimizationMethod, mt as runReferenceEquivalenceJudge, o as loopProvenanceArgsFromResult, p as runEval, pt as createReferenceEquivalenceJudge, rt as resolveRunDir, t as skillOptOptimizationMethod, v as paretoPolicy, x as heldOutGate, y as paretoSignificanceGate } from "../skillopt-optimization-method-CUmHkGbI.js";
|
|
7
|
+
import { v as pairedBootstrap } from "../statistics-CnnxdpOg.js";
|
|
8
|
+
import { i as summarizeTraceErrors, n as recordAggregateMeasurements, r as summarizeExecutionMeasurements, t as readTaskFailureLabels } from "../task-failure-attributes-CQZlB3et.js";
|
|
9
|
+
import { n as summarizeExecution, t as analyzeRuns } from "../analyze-runs-C1CavBMk.js";
|
|
10
|
+
import { a as campaignCellExecutionEvidence, c as campaignCellToRunRecord, o as campaignCellJudgeDimensions, s as campaignCellTaskScore } from "../reward-hacking-qipEpKvY.js";
|
|
11
|
+
import { n as InMemoryOutcomeStore, t as FileSystemOutcomeStore } from "../outcome-store-ChBKlTd_.js";
|
|
12
|
+
import { a as fromPiSession, c as observeCodeAgentSession, i as fromOpenCodeSession, n as fromCodexSession, o as fromPigraphSession, r as fromKimiCodeSession, s as parseCodeAgentJsonl, t as fromClaudeCodeSession } from "../code-agent-session-BjkMTQ7H.js";
|
|
13
|
+
import { t as createHostedClient } from "../client-CYzbdJOZ.js";
|
|
14
|
+
import { dirname, join } from "node:path";
|
|
15
|
+
import { mkdir, readFile, readdir, stat, writeFile } from "node:fs/promises";
|
|
16
|
+
import { agentCandidateBenchmarkSuiteSchema, agentCandidateBenchmarkTaskSchema, agentCandidateBundleSchema, agentCandidateEvaluationPolicySchema, agentCandidateExperimentSchema, agentImprovementMeasuredComparisonSchema, candidateExecutionEvidenceSchema, canonicalCandidateDigest, omitTopLevelDigest } from "@tangle-network/agent-interface";
|
|
17
|
+
//#region src/contract/self-improve.ts
|
|
18
|
+
/**
|
|
19
|
+
* Run one complete improvement job.
|
|
20
|
+
*
|
|
21
|
+
* A caller-owned `proposer` can generate candidates across local generations.
|
|
22
|
+
* An external `method`, such as official GEPA or SkillOpt, owns its complete
|
|
23
|
+
* search and returns one candidate. Both paths remeasure the selected candidate
|
|
24
|
+
* against cases that candidate generation never receives.
|
|
25
|
+
*/
|
|
26
|
+
/** Failed self-improvement run with an immutable receipt snapshot. */
|
|
107
27
|
var SelfImproveRunError = class extends Error {
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
28
|
+
cost;
|
|
29
|
+
receipts;
|
|
30
|
+
constructor(cause, ledger) {
|
|
31
|
+
const original = cause instanceof Error ? cause : new Error(String(cause));
|
|
32
|
+
super(original.message, { cause: original });
|
|
33
|
+
this.name = "SelfImproveRunError";
|
|
34
|
+
this.cost = ledger.summary();
|
|
35
|
+
this.receipts = ledger.list();
|
|
36
|
+
}
|
|
117
37
|
};
|
|
118
38
|
function assertSelfImproveSearchMode(opts) {
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
throw new Error("selfImprove: method must have a trimmed name and optimize(input)");
|
|
130
|
-
}
|
|
131
|
-
const budget = opts.budget;
|
|
132
|
-
if (budget?.generations !== void 0 && budget.generations !== 1) {
|
|
133
|
-
throw new Error("selfImprove: method owns its rounds; budget.generations must be 1 when set");
|
|
134
|
-
}
|
|
135
|
-
if (budget?.populationSize !== void 0 && budget.populationSize !== 1) {
|
|
136
|
-
throw new Error(
|
|
137
|
-
"selfImprove: method owns its candidates; budget.populationSize must be 1 when set"
|
|
138
|
-
);
|
|
139
|
-
}
|
|
140
|
-
if (budget?.candidateConcurrency !== void 0 || budget?.maxImprovementShots !== void 0 || opts.analyzeGeneration !== void 0 || opts.findings !== void 0) {
|
|
141
|
-
throw new Error(
|
|
142
|
-
"selfImprove: candidateConcurrency, maxImprovementShots, analyzeGeneration, and findings apply only to proposer mode"
|
|
143
|
-
);
|
|
144
|
-
}
|
|
39
|
+
if (opts.method && opts.proposer) throw new Error("selfImprove: method and proposer are mutually exclusive");
|
|
40
|
+
if (!opts.method) {
|
|
41
|
+
if (opts.selectionScenarios !== void 0) throw new Error("selfImprove: selectionScenarios requires method");
|
|
42
|
+
return;
|
|
43
|
+
}
|
|
44
|
+
if (typeof opts.method.name !== "string" || !opts.method.name.trim() || opts.method.name.trim() !== opts.method.name || typeof opts.method.optimize !== "function") throw new Error("selfImprove: method must have a trimmed name and optimize(input)");
|
|
45
|
+
const budget = opts.budget;
|
|
46
|
+
if (budget?.generations !== void 0 && budget.generations !== 1) throw new Error("selfImprove: method owns its rounds; budget.generations must be 1 when set");
|
|
47
|
+
if (budget?.populationSize !== void 0 && budget.populationSize !== 1) throw new Error("selfImprove: method owns its candidates; budget.populationSize must be 1 when set");
|
|
48
|
+
if (budget?.candidateConcurrency !== void 0 || budget?.maxImprovementShots !== void 0 || opts.analyzeGeneration !== void 0 || opts.findings !== void 0) throw new Error("selfImprove: candidateConcurrency, maxImprovementShots, analyzeGeneration, and findings apply only to proposer mode");
|
|
145
49
|
}
|
|
146
50
|
function splitMethodPartitions(searchScenarios, explicitSelection, fraction) {
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
throw new Error("selfImprove: method train split is empty");
|
|
176
|
-
}
|
|
177
|
-
return {
|
|
178
|
-
train,
|
|
179
|
-
selection: explicitSelection.map((scenario) => byId.get(scenario.id))
|
|
180
|
-
};
|
|
181
|
-
}
|
|
182
|
-
if (searchScenarios.length < 2) {
|
|
183
|
-
throw new Error("selfImprove: method requires at least two non-final scenarios");
|
|
184
|
-
}
|
|
185
|
-
const sorted = [...searchScenarios].sort(
|
|
186
|
-
(a, b) => stableScenarioHash(a.id) - stableScenarioHash(b.id)
|
|
187
|
-
);
|
|
188
|
-
const count = Math.max(1, Math.min(sorted.length - 1, Math.round(sorted.length * fraction)));
|
|
189
|
-
return {
|
|
190
|
-
selection: sorted.slice(0, count),
|
|
191
|
-
train: sorted.slice(count)
|
|
192
|
-
};
|
|
51
|
+
if (!Number.isFinite(fraction) || fraction <= 0 || fraction >= 1) throw new Error("selfImprove: budget.selectionFraction must be in (0, 1)");
|
|
52
|
+
const byId = /* @__PURE__ */ new Map();
|
|
53
|
+
for (const scenario of searchScenarios) {
|
|
54
|
+
if (byId.has(scenario.id)) throw new Error(`selfImprove: duplicate scenario id '${scenario.id}'`);
|
|
55
|
+
byId.set(scenario.id, scenario);
|
|
56
|
+
}
|
|
57
|
+
if (explicitSelection) {
|
|
58
|
+
if (explicitSelection.length === 0) throw new Error("selfImprove: selectionScenarios must not be empty");
|
|
59
|
+
const selectionIds = /* @__PURE__ */ new Set();
|
|
60
|
+
for (const scenario of explicitSelection) {
|
|
61
|
+
if (!byId.has(scenario.id)) throw new Error(`selfImprove: selection scenario '${scenario.id}' is absent from the non-final cases`);
|
|
62
|
+
if (selectionIds.has(scenario.id)) throw new Error(`selfImprove: duplicate selection scenario id '${scenario.id}'`);
|
|
63
|
+
selectionIds.add(scenario.id);
|
|
64
|
+
}
|
|
65
|
+
const train = searchScenarios.filter((scenario) => !selectionIds.has(scenario.id));
|
|
66
|
+
if (train.length === 0) throw new Error("selfImprove: method train split is empty");
|
|
67
|
+
return {
|
|
68
|
+
train,
|
|
69
|
+
selection: explicitSelection.map((scenario) => byId.get(scenario.id))
|
|
70
|
+
};
|
|
71
|
+
}
|
|
72
|
+
if (searchScenarios.length < 2) throw new Error("selfImprove: method requires at least two non-final scenarios");
|
|
73
|
+
const sorted = [...searchScenarios].sort((a, b) => stableScenarioHash(a.id) - stableScenarioHash(b.id));
|
|
74
|
+
const count = Math.max(1, Math.min(sorted.length - 1, Math.round(sorted.length * fraction)));
|
|
75
|
+
return {
|
|
76
|
+
selection: sorted.slice(0, count),
|
|
77
|
+
train: sorted.slice(count)
|
|
78
|
+
};
|
|
193
79
|
}
|
|
194
80
|
function safeRunComponent(value) {
|
|
195
|
-
|
|
81
|
+
return value.replace(/[^a-zA-Z0-9._-]/g, "_");
|
|
196
82
|
}
|
|
197
83
|
function stableScenarioHash(value) {
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
|
|
202
|
-
|
|
203
|
-
|
|
204
|
-
}
|
|
84
|
+
let hash = 2166136261;
|
|
85
|
+
for (let index = 0; index < value.length; index++) {
|
|
86
|
+
hash ^= value.charCodeAt(index);
|
|
87
|
+
hash = Math.imul(hash, 16777619) >>> 0;
|
|
88
|
+
}
|
|
89
|
+
return hash;
|
|
90
|
+
}
|
|
91
|
+
/**
|
|
92
|
+
* Deterministic train/holdout split by a stable hash of `scenario.id`,
|
|
93
|
+
* so the same scenario set always splits the same way across runs.
|
|
94
|
+
*/
|
|
205
95
|
function splitTrainHoldout(scenarios, fraction) {
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
|
|
211
|
-
|
|
96
|
+
const sorted = [...scenarios].sort((a, b) => stableScenarioHash(a.id) - stableScenarioHash(b.id));
|
|
97
|
+
const nHoldout = Math.max(1, Math.min(sorted.length - 1, Math.round(sorted.length * fraction)));
|
|
98
|
+
return {
|
|
99
|
+
holdout: sorted.slice(0, nHoldout),
|
|
100
|
+
train: sorted.slice(nHoldout)
|
|
101
|
+
};
|
|
212
102
|
}
|
|
213
103
|
function meanComposite(byScenario) {
|
|
214
|
-
|
|
215
|
-
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
|
|
221
|
-
|
|
222
|
-
|
|
223
|
-
|
|
224
|
-
}
|
|
104
|
+
const perScenario = {};
|
|
105
|
+
const values = [];
|
|
106
|
+
for (const [id, agg] of Object.entries(byScenario)) {
|
|
107
|
+
perScenario[id] = agg.meanComposite;
|
|
108
|
+
values.push(agg.meanComposite);
|
|
109
|
+
}
|
|
110
|
+
return {
|
|
111
|
+
compositeMean: values.length === 0 ? 0 : values.reduce((s, v) => s + v, 0) / values.length,
|
|
112
|
+
perScenario
|
|
113
|
+
};
|
|
114
|
+
}
|
|
115
|
+
/**
|
|
116
|
+
* Latest search campaign measured for the winner surface; the baseline search
|
|
117
|
+
* campaign when the winner IS the baseline. Used by the deferred-holdout
|
|
118
|
+
* summary, where no holdout campaign exists to summarize.
|
|
119
|
+
*/
|
|
225
120
|
function winnerSearchCampaign(result) {
|
|
226
|
-
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
|
|
230
|
-
|
|
231
|
-
|
|
232
|
-
|
|
233
|
-
|
|
121
|
+
for (let i = result.generations.length - 1; i >= 0; i--) {
|
|
122
|
+
const measured = result.generations[i]?.surfaces.find((s) => s.surfaceHash === result.winnerSurfaceHash);
|
|
123
|
+
if (measured) return measured.campaign;
|
|
124
|
+
}
|
|
125
|
+
return result.baselineCampaign;
|
|
126
|
+
}
|
|
127
|
+
/**
|
|
128
|
+
* One-shot self-improvement loop. See module docstring for defaults +
|
|
129
|
+
* extension points.
|
|
130
|
+
*
|
|
131
|
+
* @example Minimum:
|
|
132
|
+
*
|
|
133
|
+
* const result = await selfImprove({
|
|
134
|
+
* agent: (surface, scenario, ctx) => myAgent(surface, scenario, ctx.signal),
|
|
135
|
+
* scenarios,
|
|
136
|
+
* judge,
|
|
137
|
+
* baselineSurface: DEFAULT_PROMPT,
|
|
138
|
+
* proposer,
|
|
139
|
+
* })
|
|
140
|
+
* console.log(`lift: ${result.lift.toFixed(3)} (${result.gateDecision})`)
|
|
141
|
+
*
|
|
142
|
+
* @example Distributed (workers in three regions):
|
|
143
|
+
*
|
|
144
|
+
* await selfImprove({
|
|
145
|
+
* agent: httpDispatch({ resolveUrl: ({ placement }) => REGION_URLS[placement!] }),
|
|
146
|
+
* scenarios,
|
|
147
|
+
* judge,
|
|
148
|
+
* baselineSurface: DEFAULT_PROMPT,
|
|
149
|
+
* cellPlacement: ({ scenario }) => scenario.region,
|
|
150
|
+
* budget: { maxConcurrency: 12 },
|
|
151
|
+
* })
|
|
152
|
+
*/
|
|
234
153
|
async function selfImprove(opts) {
|
|
235
|
-
|
|
236
|
-
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
|
|
242
|
-
|
|
243
|
-
|
|
244
|
-
|
|
245
|
-
|
|
246
|
-
|
|
247
|
-
|
|
248
|
-
}
|
|
154
|
+
const startedAt = Date.now();
|
|
155
|
+
const runDir = resolveRunDir(opts.runDir ?? (opts.method ? `.agent-eval/runs/self-improve-${startedAt}` : `mem://selfImprove-${startedAt}`));
|
|
156
|
+
const storage = opts.storage ?? (runDir.startsWith("mem://") ? inMemoryCampaignStorage() : fsCampaignStorage());
|
|
157
|
+
const costLedger = createRunCostLedger({
|
|
158
|
+
storage,
|
|
159
|
+
runDir,
|
|
160
|
+
costCeilingUsd: opts.budget?.dollars
|
|
161
|
+
});
|
|
162
|
+
try {
|
|
163
|
+
return await runSelfImprove(opts, costLedger, startedAt, runDir, storage);
|
|
164
|
+
} catch (error) {
|
|
165
|
+
throw new SelfImproveRunError(error, costLedger);
|
|
166
|
+
}
|
|
249
167
|
}
|
|
250
168
|
async function runSelfImprove(opts, costLedger, startedAt, runDir, storage) {
|
|
251
|
-
|
|
252
|
-
|
|
253
|
-
|
|
254
|
-
|
|
255
|
-
|
|
256
|
-
|
|
257
|
-
|
|
258
|
-
|
|
259
|
-
|
|
260
|
-
|
|
261
|
-
|
|
262
|
-
|
|
263
|
-
|
|
264
|
-
|
|
265
|
-
|
|
266
|
-
|
|
267
|
-
|
|
268
|
-
|
|
269
|
-
|
|
270
|
-
|
|
271
|
-
|
|
272
|
-
|
|
273
|
-
|
|
274
|
-
|
|
275
|
-
|
|
276
|
-
|
|
277
|
-
|
|
278
|
-
|
|
279
|
-
|
|
280
|
-
|
|
281
|
-
|
|
282
|
-
|
|
283
|
-
|
|
284
|
-
|
|
285
|
-
|
|
286
|
-
|
|
287
|
-
|
|
288
|
-
|
|
289
|
-
|
|
290
|
-
|
|
291
|
-
|
|
292
|
-
|
|
293
|
-
|
|
294
|
-
|
|
295
|
-
|
|
296
|
-
|
|
297
|
-
|
|
298
|
-
|
|
299
|
-
|
|
300
|
-
|
|
301
|
-
|
|
302
|
-
|
|
303
|
-
|
|
304
|
-
|
|
305
|
-
|
|
306
|
-
|
|
307
|
-
|
|
308
|
-
|
|
309
|
-
|
|
310
|
-
|
|
311
|
-
|
|
312
|
-
|
|
313
|
-
|
|
314
|
-
|
|
315
|
-
|
|
316
|
-
|
|
317
|
-
|
|
318
|
-
|
|
319
|
-
|
|
320
|
-
|
|
321
|
-
|
|
322
|
-
|
|
323
|
-
|
|
324
|
-
|
|
325
|
-
|
|
326
|
-
|
|
327
|
-
|
|
328
|
-
|
|
329
|
-
|
|
330
|
-
|
|
331
|
-
|
|
332
|
-
|
|
333
|
-
|
|
334
|
-
|
|
335
|
-
|
|
336
|
-
|
|
337
|
-
|
|
338
|
-
|
|
339
|
-
|
|
340
|
-
|
|
341
|
-
|
|
342
|
-
|
|
343
|
-
|
|
344
|
-
|
|
345
|
-
|
|
346
|
-
|
|
347
|
-
|
|
348
|
-
|
|
349
|
-
|
|
350
|
-
|
|
351
|
-
|
|
352
|
-
|
|
353
|
-
|
|
354
|
-
|
|
355
|
-
|
|
356
|
-
|
|
357
|
-
|
|
358
|
-
|
|
359
|
-
|
|
360
|
-
|
|
361
|
-
|
|
362
|
-
|
|
363
|
-
|
|
364
|
-
|
|
365
|
-
|
|
366
|
-
|
|
367
|
-
|
|
368
|
-
|
|
369
|
-
|
|
370
|
-
|
|
371
|
-
|
|
372
|
-
|
|
373
|
-
|
|
374
|
-
|
|
375
|
-
|
|
376
|
-
|
|
377
|
-
|
|
378
|
-
|
|
379
|
-
|
|
380
|
-
|
|
381
|
-
|
|
382
|
-
|
|
383
|
-
|
|
384
|
-
|
|
385
|
-
|
|
386
|
-
|
|
387
|
-
|
|
388
|
-
|
|
389
|
-
|
|
390
|
-
|
|
391
|
-
|
|
392
|
-
|
|
393
|
-
|
|
394
|
-
|
|
395
|
-
|
|
396
|
-
|
|
397
|
-
|
|
398
|
-
|
|
399
|
-
|
|
400
|
-
|
|
401
|
-
|
|
402
|
-
|
|
403
|
-
|
|
404
|
-
|
|
405
|
-
|
|
406
|
-
|
|
407
|
-
|
|
408
|
-
|
|
409
|
-
|
|
410
|
-
|
|
411
|
-
|
|
412
|
-
|
|
413
|
-
|
|
414
|
-
|
|
415
|
-
|
|
416
|
-
|
|
417
|
-
|
|
418
|
-
|
|
419
|
-
|
|
420
|
-
|
|
421
|
-
|
|
422
|
-
|
|
423
|
-
|
|
424
|
-
|
|
425
|
-
|
|
426
|
-
|
|
427
|
-
|
|
428
|
-
|
|
429
|
-
|
|
430
|
-
|
|
431
|
-
|
|
432
|
-
|
|
433
|
-
|
|
434
|
-
|
|
435
|
-
|
|
436
|
-
|
|
437
|
-
|
|
438
|
-
|
|
439
|
-
|
|
440
|
-
|
|
441
|
-
|
|
442
|
-
|
|
443
|
-
|
|
444
|
-
|
|
445
|
-
|
|
446
|
-
}
|
|
447
|
-
} : {},
|
|
448
|
-
storage,
|
|
449
|
-
hostedClient: opts.hostedTenant ? createHostedClient(opts.hostedTenant) : void 0
|
|
450
|
-
});
|
|
451
|
-
if (opts.onProvenance) opts.onProvenance(provenance);
|
|
452
|
-
const summary = {
|
|
453
|
-
baseline,
|
|
454
|
-
winner: {
|
|
455
|
-
...winnerStats,
|
|
456
|
-
surface: result.winnerSurface,
|
|
457
|
-
...result.winnerLabel ? { label: result.winnerLabel } : {},
|
|
458
|
-
...result.winnerRationale ? { rationale: result.winnerRationale } : {}
|
|
459
|
-
},
|
|
460
|
-
...holdoutDeferred ? {} : { lift: winnerStats.compositeMean - baseline.compositeMean },
|
|
461
|
-
diff: result.promotedDiff,
|
|
462
|
-
provenance,
|
|
463
|
-
gateDecision: result.gateResult.decision,
|
|
464
|
-
generationsExplored: result.generations.length,
|
|
465
|
-
durationMs,
|
|
466
|
-
totalCostUsd: totalCost,
|
|
467
|
-
cost,
|
|
468
|
-
receipts: costLedger.list(),
|
|
469
|
-
...optimizationResult ? {
|
|
470
|
-
optimization: {
|
|
471
|
-
name: opts.method.name,
|
|
472
|
-
cost: structuredClone(optimizationResult.cost),
|
|
473
|
-
...optimizationResult.durationMs === void 0 ? {} : { durationMs: optimizationResult.durationMs },
|
|
474
|
-
...optimizationResult.provenance === void 0 ? {} : { provenance: structuredClone(optimizationResult.provenance) }
|
|
475
|
-
}
|
|
476
|
-
} : {},
|
|
477
|
-
insight,
|
|
478
|
-
...power ? { power } : {},
|
|
479
|
-
raw: result
|
|
480
|
-
};
|
|
481
|
-
if (opts.hostedTenant) {
|
|
482
|
-
try {
|
|
483
|
-
await shipEvalRunToHosted(opts.hostedTenant, opts, summary, result, runDir);
|
|
484
|
-
} catch (err) {
|
|
485
|
-
const msg = err instanceof Error ? err.message : String(err);
|
|
486
|
-
console.warn(`[agent-eval] hosted ingest failed (continuing): ${msg}`);
|
|
487
|
-
}
|
|
488
|
-
}
|
|
489
|
-
return summary;
|
|
169
|
+
const budget = opts.budget ?? {};
|
|
170
|
+
assertSelfImproveSearchMode(opts);
|
|
171
|
+
const generations = opts.method ? 1 : budget.generations ?? 3;
|
|
172
|
+
const populationSize = opts.method ? 1 : budget.populationSize ?? 2;
|
|
173
|
+
const maxConcurrency = budget.maxConcurrency ?? 2;
|
|
174
|
+
const holdoutFraction = budget.holdoutFraction ?? .25;
|
|
175
|
+
const holdoutMode = budget.holdout ?? "measured";
|
|
176
|
+
const holdoutDeferred = holdoutMode === "deferred";
|
|
177
|
+
const expectUsage = opts.expectUsage ?? "assert";
|
|
178
|
+
const explicitHoldout = budget.holdoutScenarios;
|
|
179
|
+
const { train, holdout } = explicitHoldout ? {
|
|
180
|
+
train: opts.scenarios.filter((s) => !explicitHoldout.some((h) => h.id === s.id)),
|
|
181
|
+
holdout: explicitHoldout
|
|
182
|
+
} : holdoutDeferred ? {
|
|
183
|
+
train: opts.scenarios,
|
|
184
|
+
holdout: []
|
|
185
|
+
} : splitTrainHoldout(opts.scenarios, holdoutFraction);
|
|
186
|
+
if (train.length === 0) throw new Error("selfImprove: train split is empty. Reduce holdoutFraction or pass more scenarios.");
|
|
187
|
+
if (holdout.length === 0 && !holdoutDeferred) throw new Error("selfImprove: holdout split is empty. Pass more scenarios.");
|
|
188
|
+
if (generations > 0 && !opts.proposer && !opts.method) throw new Error("selfImprove: method or proposer is required when budget.generations is greater than zero");
|
|
189
|
+
let optimizationResult;
|
|
190
|
+
const methodPartitions = opts.method ? splitMethodPartitions(train, opts.selectionScenarios, budget.selectionFraction ?? .25) : void 0;
|
|
191
|
+
const proposer = opts.method ? {
|
|
192
|
+
kind: `method:${opts.method.name}`,
|
|
193
|
+
propose: async (context) => {
|
|
194
|
+
if (context.generation > 0) return [];
|
|
195
|
+
const result = await opts.method.optimize(Object.freeze({
|
|
196
|
+
baselineSurface: structuredClone(context.currentSurface),
|
|
197
|
+
trainScenarios: Object.freeze(methodPartitions.train.map((scenario) => structuredClone(scenario))),
|
|
198
|
+
selectionScenarios: Object.freeze(methodPartitions.selection.map((scenario) => structuredClone(scenario))),
|
|
199
|
+
dispatchWithSurface: opts.agent,
|
|
200
|
+
judges: Object.freeze([opts.judge]),
|
|
201
|
+
runDir: `${runDir}/optimization/${safeRunComponent(opts.method.name)}`,
|
|
202
|
+
seed: 42,
|
|
203
|
+
runOptions: Object.freeze({
|
|
204
|
+
storage,
|
|
205
|
+
maxConcurrency,
|
|
206
|
+
reps: budget.reps,
|
|
207
|
+
dispatchTimeoutMs: opts.dispatchTimeoutMs,
|
|
208
|
+
expectUsage,
|
|
209
|
+
costCeiling: budget.dollars
|
|
210
|
+
}),
|
|
211
|
+
costLedger
|
|
212
|
+
}));
|
|
213
|
+
assertOptimizationResult(opts.method.name, result);
|
|
214
|
+
optimizationResult = structuredClone(result);
|
|
215
|
+
return [{
|
|
216
|
+
surface: structuredClone(result.winnerSurface),
|
|
217
|
+
label: opts.method.name,
|
|
218
|
+
rationale: `${opts.method.name} selected this surface without final cases.`
|
|
219
|
+
}];
|
|
220
|
+
}
|
|
221
|
+
} : opts.proposer ?? {
|
|
222
|
+
kind: "baseline-only",
|
|
223
|
+
propose: async () => []
|
|
224
|
+
};
|
|
225
|
+
const gate = opts.gate ?? defaultProductionGate({
|
|
226
|
+
holdoutScenarios: holdout,
|
|
227
|
+
deltaThreshold: .05
|
|
228
|
+
});
|
|
229
|
+
if (opts.onProgress) opts.onProgress({
|
|
230
|
+
kind: "baseline.started",
|
|
231
|
+
scenarios: opts.scenarios.length
|
|
232
|
+
});
|
|
233
|
+
const result = await runImprovementLoop({
|
|
234
|
+
scenarios: train,
|
|
235
|
+
baselineSurface: opts.baselineSurface,
|
|
236
|
+
premeasuredBaseline: opts.premeasuredBaseline,
|
|
237
|
+
dispatchWithSurface: opts.agent,
|
|
238
|
+
proposer,
|
|
239
|
+
judges: [opts.judge],
|
|
240
|
+
populationSize,
|
|
241
|
+
maxGenerations: generations,
|
|
242
|
+
candidateConcurrency: budget.candidateConcurrency,
|
|
243
|
+
reps: budget.reps,
|
|
244
|
+
maxImprovementShots: budget.maxImprovementShots,
|
|
245
|
+
holdoutScenarios: holdout,
|
|
246
|
+
holdout: holdoutMode,
|
|
247
|
+
gate,
|
|
248
|
+
neutralize: opts.neutralize,
|
|
249
|
+
autoOnPromote: opts.autoOnPromote ?? "none",
|
|
250
|
+
ghOwner: opts.ghOwner,
|
|
251
|
+
ghRepo: opts.ghRepo,
|
|
252
|
+
storage,
|
|
253
|
+
runDir,
|
|
254
|
+
maxConcurrency,
|
|
255
|
+
cellPlacement: opts.cellPlacement,
|
|
256
|
+
dispatchTimeoutMs: opts.dispatchTimeoutMs,
|
|
257
|
+
costLedger,
|
|
258
|
+
expectUsage,
|
|
259
|
+
labeledStore: opts.labeledStore,
|
|
260
|
+
captureSource: opts.captureSource,
|
|
261
|
+
analyzeGeneration: opts.analyzeGeneration,
|
|
262
|
+
findings: opts.findings,
|
|
263
|
+
selectionRankKey: opts.selectionRankKey
|
|
264
|
+
});
|
|
265
|
+
const reportSplit = holdoutDeferred ? "search" : "holdout";
|
|
266
|
+
const reportBaselineCampaign = holdoutDeferred ? result.baselineCampaign : result.baselineOnHoldout;
|
|
267
|
+
const reportWinnerCampaign = holdoutDeferred ? winnerSearchCampaign(result) : result.winnerOnHoldout;
|
|
268
|
+
const baseline = meanComposite(reportBaselineCampaign.aggregates.byScenario);
|
|
269
|
+
const winnerStats = meanComposite(reportWinnerCampaign.aggregates.byScenario);
|
|
270
|
+
let power;
|
|
271
|
+
const baselineHoldoutComposites = result.baselineOnHoldout.cells.filter((cell) => !cell.error).map((cell) => {
|
|
272
|
+
const scores = Object.values(cell.judgeScores);
|
|
273
|
+
return scores.length === 0 ? NaN : scores.reduce((sum, s) => sum + s.composite, 0) / scores.length;
|
|
274
|
+
}).filter((v) => Number.isFinite(v));
|
|
275
|
+
if (baselineHoldoutComposites.length >= 3) {
|
|
276
|
+
power = powerPreflight({
|
|
277
|
+
baselineComposites: baselineHoldoutComposites,
|
|
278
|
+
sharedScorerChannel: true
|
|
279
|
+
});
|
|
280
|
+
if (opts.onProgress) opts.onProgress({
|
|
281
|
+
kind: "power.estimated",
|
|
282
|
+
n: power.n,
|
|
283
|
+
sd: power.sd,
|
|
284
|
+
mde: power.mde,
|
|
285
|
+
underpowered: power.underpowered
|
|
286
|
+
});
|
|
287
|
+
if (power.underpowered && generations > 0) console.warn(`[selfImprove] ${power.recommendation}`);
|
|
288
|
+
}
|
|
289
|
+
if (opts.onProgress) {
|
|
290
|
+
opts.onProgress({
|
|
291
|
+
kind: "baseline.completed",
|
|
292
|
+
compositeMean: baseline.compositeMean,
|
|
293
|
+
durationMs: Date.now() - startedAt
|
|
294
|
+
});
|
|
295
|
+
opts.onProgress({
|
|
296
|
+
kind: "gate.decided",
|
|
297
|
+
decision: result.gateResult.decision,
|
|
298
|
+
...holdoutDeferred ? {} : { lift: winnerStats.compositeMean - baseline.compositeMean }
|
|
299
|
+
});
|
|
300
|
+
}
|
|
301
|
+
const cost = result.cost;
|
|
302
|
+
const totalCost = cost.totalCostUsd;
|
|
303
|
+
const insight = await analyzeRuns({
|
|
304
|
+
runs: [...cellsToRunRecords(reportBaselineCampaign.cells, "baseline", runDir, opts.baselineSurface, reportSplit, opts.model), ...cellsToRunRecords(reportWinnerCampaign.cells, "winner", runDir, result.winnerSurface, reportSplit, opts.model)],
|
|
305
|
+
baselineCandidateId: "baseline",
|
|
306
|
+
candidateCandidateId: "winner"
|
|
307
|
+
});
|
|
308
|
+
const durationMs = Date.now() - startedAt;
|
|
309
|
+
const { record: provenance } = await emitLoopProvenance({
|
|
310
|
+
...loopProvenanceArgsFromResult({
|
|
311
|
+
runId: `${runDir}#${startedAt}`,
|
|
312
|
+
runDir,
|
|
313
|
+
timestamp: new Date(startedAt).toISOString(),
|
|
314
|
+
baselineSurface: opts.baselineSurface,
|
|
315
|
+
result,
|
|
316
|
+
costReceipts: costLedger.list(),
|
|
317
|
+
totalCostUsd: totalCost,
|
|
318
|
+
totalDurationMs: durationMs
|
|
319
|
+
}),
|
|
320
|
+
...optimizationResult ? { optimizationMethod: {
|
|
321
|
+
name: opts.method.name,
|
|
322
|
+
cost: structuredClone(optimizationResult.cost),
|
|
323
|
+
...optimizationResult.durationMs === void 0 ? {} : { durationMs: optimizationResult.durationMs },
|
|
324
|
+
...optimizationResult.provenance === void 0 ? {} : { provenance: structuredClone(optimizationResult.provenance) }
|
|
325
|
+
} } : {},
|
|
326
|
+
storage,
|
|
327
|
+
hostedClient: opts.hostedTenant ? createHostedClient(opts.hostedTenant) : void 0
|
|
328
|
+
});
|
|
329
|
+
if (opts.onProvenance) opts.onProvenance(provenance);
|
|
330
|
+
const summary = {
|
|
331
|
+
baseline,
|
|
332
|
+
winner: {
|
|
333
|
+
...winnerStats,
|
|
334
|
+
surface: result.winnerSurface,
|
|
335
|
+
...result.winnerLabel ? { label: result.winnerLabel } : {},
|
|
336
|
+
...result.winnerRationale ? { rationale: result.winnerRationale } : {}
|
|
337
|
+
},
|
|
338
|
+
...holdoutDeferred ? {} : { lift: winnerStats.compositeMean - baseline.compositeMean },
|
|
339
|
+
diff: result.promotedDiff,
|
|
340
|
+
provenance,
|
|
341
|
+
gateDecision: result.gateResult.decision,
|
|
342
|
+
generationsExplored: result.generations.length,
|
|
343
|
+
durationMs,
|
|
344
|
+
totalCostUsd: totalCost,
|
|
345
|
+
cost,
|
|
346
|
+
receipts: costLedger.list(),
|
|
347
|
+
...optimizationResult ? { optimization: {
|
|
348
|
+
name: opts.method.name,
|
|
349
|
+
cost: structuredClone(optimizationResult.cost),
|
|
350
|
+
...optimizationResult.durationMs === void 0 ? {} : { durationMs: optimizationResult.durationMs },
|
|
351
|
+
...optimizationResult.provenance === void 0 ? {} : { provenance: structuredClone(optimizationResult.provenance) }
|
|
352
|
+
} } : {},
|
|
353
|
+
insight,
|
|
354
|
+
...power ? { power } : {},
|
|
355
|
+
raw: result
|
|
356
|
+
};
|
|
357
|
+
if (opts.hostedTenant) try {
|
|
358
|
+
await shipEvalRunToHosted(opts.hostedTenant, opts, summary, result, runDir);
|
|
359
|
+
} catch (err) {
|
|
360
|
+
const msg = err instanceof Error ? err.message : String(err);
|
|
361
|
+
console.warn(`[agent-eval] hosted ingest failed (continuing): ${msg}`);
|
|
362
|
+
}
|
|
363
|
+
return summary;
|
|
490
364
|
}
|
|
491
365
|
async function shipEvalRunToHosted(tenant, opts, summary, raw, runDir) {
|
|
492
|
-
|
|
493
|
-
|
|
494
|
-
|
|
495
|
-
|
|
496
|
-
|
|
497
|
-
|
|
498
|
-
|
|
499
|
-
|
|
500
|
-
|
|
501
|
-
|
|
502
|
-
|
|
503
|
-
|
|
504
|
-
|
|
505
|
-
|
|
506
|
-
|
|
507
|
-
|
|
508
|
-
|
|
509
|
-
|
|
510
|
-
|
|
511
|
-
|
|
512
|
-
|
|
513
|
-
|
|
514
|
-
|
|
515
|
-
|
|
516
|
-
|
|
517
|
-
|
|
518
|
-
|
|
519
|
-
|
|
520
|
-
|
|
521
|
-
|
|
522
|
-
|
|
523
|
-
|
|
524
|
-
|
|
525
|
-
|
|
526
|
-
|
|
527
|
-
|
|
528
|
-
|
|
529
|
-
|
|
530
|
-
|
|
531
|
-
|
|
532
|
-
|
|
533
|
-
|
|
534
|
-
|
|
535
|
-
|
|
536
|
-
|
|
537
|
-
|
|
538
|
-
|
|
539
|
-
|
|
540
|
-
gateDecision: summary.gateDecision,
|
|
541
|
-
holdoutLift: summary.lift,
|
|
542
|
-
totalCostUsd: summary.totalCostUsd,
|
|
543
|
-
totalDurationMs: summary.durationMs,
|
|
544
|
-
insightReport: summary.insight
|
|
545
|
-
};
|
|
546
|
-
await client.ingestEvalRun(event);
|
|
366
|
+
const client = createHostedClient(tenant);
|
|
367
|
+
function snapshotFromCampaign(index, surface, campaign, durationMs) {
|
|
368
|
+
const cells = campaign.cells.map((cell) => {
|
|
369
|
+
const execution = campaignCellExecutionEvidence(cell);
|
|
370
|
+
return {
|
|
371
|
+
scenarioId: cell.scenarioId,
|
|
372
|
+
rep: cell.rep,
|
|
373
|
+
compositeMean: campaignCellTaskScore(cell) ?? null,
|
|
374
|
+
dimensions: campaignCellJudgeDimensions(cell),
|
|
375
|
+
terminalOutcome: execution.terminalOutcome,
|
|
376
|
+
executionErrorCount: execution.executionErrorCount ?? null,
|
|
377
|
+
errorMessage: cell.error ?? void 0
|
|
378
|
+
};
|
|
379
|
+
});
|
|
380
|
+
const scoredCells = cells.flatMap((cell) => cell.compositeMean === null ? [] : [cell.compositeMean]);
|
|
381
|
+
const compositeMean = scoredCells.length === 0 ? null : scoredCells.reduce((sum, score) => sum + score, 0) / scoredCells.length;
|
|
382
|
+
return {
|
|
383
|
+
index,
|
|
384
|
+
surfaceHash: surfaceHash(surface),
|
|
385
|
+
surface,
|
|
386
|
+
cells,
|
|
387
|
+
compositeMean,
|
|
388
|
+
costUsd: campaign.aggregates.cost.totalCostUsd,
|
|
389
|
+
durationMs
|
|
390
|
+
};
|
|
391
|
+
}
|
|
392
|
+
const generations = [];
|
|
393
|
+
generations.push(snapshotFromCampaign(0, opts.baselineSurface, raw.baselineCampaign, 0));
|
|
394
|
+
for (const gen of raw.generations) {
|
|
395
|
+
const winner = gen.surfaces.reduce((best, s) => s.campaign.aggregates.cellsExecuted > 0 && (best === void 0 || averageComposite(s.campaign) > averageComposite(best.campaign)) ? s : best, gen.surfaces[0]);
|
|
396
|
+
if (!winner) continue;
|
|
397
|
+
generations.push(snapshotFromCampaign(gen.record.generationIndex + 1, winner.surface, winner.campaign, 0));
|
|
398
|
+
}
|
|
399
|
+
const event = {
|
|
400
|
+
runId: `${runDir}#${Date.now()}`,
|
|
401
|
+
runDir,
|
|
402
|
+
timestamp: (/* @__PURE__ */ new Date()).toISOString(),
|
|
403
|
+
status: "finished",
|
|
404
|
+
labels: opts.hostedLabels ?? {},
|
|
405
|
+
baseline: generations[0],
|
|
406
|
+
generations,
|
|
407
|
+
gateDecision: summary.gateDecision,
|
|
408
|
+
holdoutLift: summary.lift,
|
|
409
|
+
totalCostUsd: summary.totalCostUsd,
|
|
410
|
+
totalDurationMs: summary.durationMs,
|
|
411
|
+
insightReport: summary.insight
|
|
412
|
+
};
|
|
413
|
+
await client.ingestEvalRun(event);
|
|
547
414
|
}
|
|
548
415
|
function averageComposite(campaign) {
|
|
549
|
-
|
|
550
|
-
|
|
416
|
+
const aggs = Object.values(campaign.aggregates.byScenario);
|
|
417
|
+
return aggs.length === 0 ? 0 : aggs.reduce((s, a) => s + a.meanComposite, 0) / aggs.length;
|
|
551
418
|
}
|
|
552
419
|
function hashString(s) {
|
|
553
|
-
|
|
554
|
-
|
|
555
|
-
|
|
556
|
-
|
|
557
|
-
|
|
558
|
-
|
|
559
|
-
}
|
|
420
|
+
let h = 2166136261;
|
|
421
|
+
for (let i = 0; i < s.length; i++) {
|
|
422
|
+
h ^= s.charCodeAt(i);
|
|
423
|
+
h = Math.imul(h, 16777619) >>> 0;
|
|
424
|
+
}
|
|
425
|
+
return h.toString(16).padStart(8, "0");
|
|
426
|
+
}
|
|
427
|
+
/**
|
|
428
|
+
* Adapt campaign cells into the `RunRecord` shape `analyzeRuns()` consumes.
|
|
429
|
+
* Each cell becomes one run; `candidateId` is the caller-supplied label so
|
|
430
|
+
* baseline + winner pair cleanly on `(experimentId, scenarioId, seed)`.
|
|
431
|
+
*
|
|
432
|
+
* `promptHash` is the REAL sha256 content hash of the surface this cell ran
|
|
433
|
+
* (baseline vs winner are byte-distinguishable + byte-identical-verifiable);
|
|
434
|
+
* `configHash` is the sha256 of the candidate label so the two candidates'
|
|
435
|
+
* config rows differ. Both were previously the literal `'sha256:cell'`, which
|
|
436
|
+
* made baseline and winner indistinguishable in every downstream record.
|
|
437
|
+
*/
|
|
560
438
|
function cellsToRunRecords(cells, candidateId, runId, surface, splitTag, fallbackModel) {
|
|
561
|
-
|
|
562
|
-
|
|
563
|
-
|
|
564
|
-
|
|
565
|
-
|
|
566
|
-
|
|
567
|
-
|
|
568
|
-
|
|
569
|
-
|
|
570
|
-
|
|
571
|
-
|
|
572
|
-
|
|
573
|
-
|
|
574
|
-
|
|
575
|
-
|
|
576
|
-
|
|
577
|
-
|
|
578
|
-
|
|
579
|
-
|
|
580
|
-
|
|
581
|
-
|
|
582
|
-
|
|
583
|
-
|
|
584
|
-
|
|
585
|
-
|
|
586
|
-
|
|
587
|
-
|
|
588
|
-
|
|
589
|
-
|
|
590
|
-
// src/contract/define-agent-eval.ts
|
|
439
|
+
const promptHash = surfaceContentHash(surface);
|
|
440
|
+
const configHash = surfaceContentHash(candidateId);
|
|
441
|
+
return cells.map((cell) => {
|
|
442
|
+
const model = cell.resolvedModel ?? fallbackModel;
|
|
443
|
+
if (!model) throw new ValidationError(`selfImprove.model is required when cell ${cell.cellId} has no paid-call model receipt`);
|
|
444
|
+
if (!modelHasSnapshot(model)) throw new ValidationError(`selfImprove model "${model}" lacks a snapshot version for cell ${cell.cellId}`);
|
|
445
|
+
return campaignCellToRunRecord(cell, {
|
|
446
|
+
runId: `${runId}::${candidateId}::${cell.cellId}`,
|
|
447
|
+
experimentId: runId,
|
|
448
|
+
candidateId,
|
|
449
|
+
seed: cell.rep * 1e6 + hashString(cell.scenarioId).slice(0, 6).split("").reduce((a, c) => a * 31 + c.charCodeAt(0) >>> 0, 0),
|
|
450
|
+
model,
|
|
451
|
+
promptHash,
|
|
452
|
+
configHash,
|
|
453
|
+
commitSha: "cell",
|
|
454
|
+
splitTag
|
|
455
|
+
});
|
|
456
|
+
});
|
|
457
|
+
}
|
|
458
|
+
//#endregion
|
|
459
|
+
//#region src/contract/define-agent-eval.ts
|
|
460
|
+
/**
|
|
461
|
+
* Define an agent eval once, then either score a surface with `evaluate()` or
|
|
462
|
+
* run the closed loop with `improve()`.
|
|
463
|
+
*
|
|
464
|
+
* This is a DX wrapper only: it delegates to `runEval()` and `selfImprove()` and
|
|
465
|
+
* returns their native result shapes.
|
|
466
|
+
*/
|
|
591
467
|
function defineAgentEval(defaults) {
|
|
592
|
-
|
|
593
|
-
|
|
594
|
-
|
|
595
|
-
|
|
596
|
-
|
|
597
|
-
|
|
598
|
-
|
|
599
|
-
|
|
600
|
-
|
|
601
|
-
|
|
602
|
-
|
|
603
|
-
|
|
604
|
-
|
|
605
|
-
|
|
606
|
-
|
|
607
|
-
|
|
608
|
-
|
|
609
|
-
|
|
610
|
-
|
|
611
|
-
|
|
612
|
-
|
|
613
|
-
|
|
614
|
-
|
|
615
|
-
|
|
616
|
-
|
|
617
|
-
|
|
618
|
-
|
|
619
|
-
|
|
620
|
-
|
|
621
|
-
|
|
622
|
-
|
|
623
|
-
|
|
624
|
-
|
|
625
|
-
|
|
626
|
-
...budget ? { budget } : {},
|
|
627
|
-
...hostedTenant ? { hostedTenant } : {}
|
|
628
|
-
});
|
|
629
|
-
}
|
|
630
|
-
};
|
|
468
|
+
const defaultEvaluateOptions = evaluateDefaults(defaults);
|
|
469
|
+
return {
|
|
470
|
+
scenarios: defaults.scenarios,
|
|
471
|
+
baselineSurface: defaults.baselineSurface,
|
|
472
|
+
async evaluate(opts = {}) {
|
|
473
|
+
const { agent, judge, judges, runDir, scenarios, surface, ...campaignOpts } = opts;
|
|
474
|
+
const selectedAgent = agent ?? defaults.agent;
|
|
475
|
+
const selectedSurface = surface ?? defaults.baselineSurface;
|
|
476
|
+
const selectedRunDir = runDir ?? defaults.runDir ?? `mem://defineAgentEval-${Date.now()}`;
|
|
477
|
+
const selectedStorage = campaignOpts.storage ?? defaultEvaluateOptions.storage ?? (selectedRunDir.startsWith("mem://") ? inMemoryCampaignStorage() : void 0);
|
|
478
|
+
const evalOptions = {
|
|
479
|
+
...defaultEvaluateOptions,
|
|
480
|
+
...campaignOpts,
|
|
481
|
+
...selectedStorage ? { storage: selectedStorage } : {},
|
|
482
|
+
runDir: selectedRunDir,
|
|
483
|
+
scenarios: scenarios ?? defaults.scenarios,
|
|
484
|
+
dispatch: (scenario, ctx) => selectedAgent(selectedSurface, scenario, ctx),
|
|
485
|
+
judges: evaluateJudges(judges, judge ?? defaults.judge)
|
|
486
|
+
};
|
|
487
|
+
if (evalOptions.reps !== void 0) evalOptions.reps = requirePositiveInteger(evalOptions.reps, "reps");
|
|
488
|
+
return runEval(evalOptions);
|
|
489
|
+
},
|
|
490
|
+
async improve(opts = {}) {
|
|
491
|
+
const { budget: budgetOverride, hostedTenant: hostedTenantOverride, ...topLevelOverrides } = opts;
|
|
492
|
+
const merged = mergeDefined(defaults, topLevelOverrides);
|
|
493
|
+
const budget = mergeBudget(defaults.budget, budgetOverride);
|
|
494
|
+
const hostedTenant = mergeHostedTenant(defaults.hostedTenant, hostedTenantOverride);
|
|
495
|
+
return selfImprove({
|
|
496
|
+
...merged,
|
|
497
|
+
...budget ? { budget } : {},
|
|
498
|
+
...hostedTenant ? { hostedTenant } : {}
|
|
499
|
+
});
|
|
500
|
+
}
|
|
501
|
+
};
|
|
631
502
|
}
|
|
632
503
|
function evaluateDefaults(defaults) {
|
|
633
|
-
|
|
634
|
-
|
|
635
|
-
|
|
636
|
-
|
|
637
|
-
|
|
638
|
-
|
|
639
|
-
|
|
640
|
-
|
|
641
|
-
|
|
642
|
-
|
|
643
|
-
out.reps = requirePositiveInteger(defaults.budget.reps, "budget.reps");
|
|
644
|
-
return out;
|
|
504
|
+
const out = {};
|
|
505
|
+
if (defaults.storage) out.storage = defaults.storage;
|
|
506
|
+
if (defaults.labeledStore) out.labeledStore = defaults.labeledStore;
|
|
507
|
+
if (defaults.captureSource) out.captureSource = defaults.captureSource;
|
|
508
|
+
if (defaults.cellPlacement) out.cellPlacement = defaults.cellPlacement;
|
|
509
|
+
if (defaults.expectUsage) out.expectUsage = defaults.expectUsage;
|
|
510
|
+
if (defaults.budget?.dollars !== void 0) out.costCeiling = defaults.budget.dollars;
|
|
511
|
+
if (defaults.budget?.maxConcurrency !== void 0) out.maxConcurrency = defaults.budget.maxConcurrency;
|
|
512
|
+
if (defaults.budget?.reps !== void 0) out.reps = requirePositiveInteger(defaults.budget.reps, "budget.reps");
|
|
513
|
+
return out;
|
|
645
514
|
}
|
|
646
515
|
function mergeBudget(defaults, overrides) {
|
|
647
|
-
|
|
648
|
-
|
|
649
|
-
|
|
516
|
+
const merged = mergeOptionalObject(defaults, overrides);
|
|
517
|
+
if (merged?.reps !== void 0) merged.reps = requirePositiveInteger(merged.reps, "budget.reps");
|
|
518
|
+
return merged;
|
|
650
519
|
}
|
|
651
520
|
function mergeHostedTenant(defaults, overrides) {
|
|
652
|
-
|
|
653
|
-
|
|
654
|
-
|
|
655
|
-
|
|
656
|
-
"defineAgentEval.improve: hostedTenant requires endpoint, apiKey, and tenantId after merging defaults and overrides"
|
|
657
|
-
);
|
|
658
|
-
}
|
|
659
|
-
return merged;
|
|
521
|
+
const merged = mergeOptionalObject(defaults, overrides);
|
|
522
|
+
if (!merged) return void 0;
|
|
523
|
+
if (!merged.endpoint?.trim() || !merged.apiKey?.trim() || !merged.tenantId?.trim()) throw new Error("defineAgentEval.improve: hostedTenant requires endpoint, apiKey, and tenantId after merging defaults and overrides");
|
|
524
|
+
return merged;
|
|
660
525
|
}
|
|
661
526
|
function mergeDefined(defaults, overrides) {
|
|
662
|
-
|
|
663
|
-
|
|
664
|
-
|
|
665
|
-
|
|
666
|
-
}
|
|
667
|
-
return merged;
|
|
527
|
+
if (!overrides) return defaults;
|
|
528
|
+
const merged = { ...defaults };
|
|
529
|
+
for (const [key, value] of Object.entries(overrides)) if (value !== void 0) merged[key] = value;
|
|
530
|
+
return merged;
|
|
668
531
|
}
|
|
669
532
|
function mergeOptionalObject(defaults, overrides) {
|
|
670
|
-
|
|
671
|
-
|
|
533
|
+
if (!defaults && !overrides) return void 0;
|
|
534
|
+
return mergeDefined(defaults ?? {}, overrides);
|
|
672
535
|
}
|
|
673
536
|
function evaluateJudges(judges, defaultJudge) {
|
|
674
|
-
|
|
675
|
-
|
|
676
|
-
|
|
677
|
-
|
|
678
|
-
|
|
679
|
-
}
|
|
680
|
-
return [defaultJudge];
|
|
537
|
+
if (judges !== void 0) {
|
|
538
|
+
if (judges.length === 0) throw new Error("defineAgentEval.evaluate: judges must not be empty");
|
|
539
|
+
return judges;
|
|
540
|
+
}
|
|
541
|
+
return [defaultJudge];
|
|
681
542
|
}
|
|
682
543
|
function requirePositiveInteger(value, field) {
|
|
683
|
-
|
|
684
|
-
|
|
685
|
-
}
|
|
686
|
-
return value;
|
|
544
|
+
if (!Number.isInteger(value) || value < 1) throw new Error(`defineAgentEval: ${field} must be a positive integer`);
|
|
545
|
+
return value;
|
|
687
546
|
}
|
|
688
|
-
|
|
689
|
-
|
|
690
|
-
|
|
691
|
-
agentCandidateBenchmarkSuiteSchema,
|
|
692
|
-
agentCandidateBenchmarkTaskSchema,
|
|
693
|
-
agentCandidateBundleSchema,
|
|
694
|
-
agentCandidateEvaluationPolicySchema,
|
|
695
|
-
agentCandidateExperimentSchema,
|
|
696
|
-
agentImprovementMeasuredComparisonSchema,
|
|
697
|
-
candidateExecutionEvidenceSchema,
|
|
698
|
-
canonicalCandidateDigest,
|
|
699
|
-
omitTopLevelDigest
|
|
700
|
-
} from "@tangle-network/agent-interface";
|
|
547
|
+
//#endregion
|
|
548
|
+
//#region src/contract/measured-comparison.ts
|
|
549
|
+
/** Content-address one task before any measured execution can see it. */
|
|
701
550
|
function sealCandidateBenchmarkTask(material) {
|
|
702
|
-
|
|
703
|
-
|
|
704
|
-
|
|
705
|
-
|
|
551
|
+
return agentCandidateBenchmarkTaskSchema.parse({
|
|
552
|
+
...material,
|
|
553
|
+
digest: canonicalCandidateDigest(material)
|
|
554
|
+
});
|
|
706
555
|
}
|
|
556
|
+
/** Freeze task order, repetitions, and every seed before either arm runs. */
|
|
707
557
|
function sealCandidateBenchmarkSuite(options) {
|
|
708
|
-
|
|
709
|
-
|
|
710
|
-
|
|
711
|
-
|
|
712
|
-
|
|
713
|
-
|
|
714
|
-
|
|
715
|
-
|
|
716
|
-
|
|
717
|
-
|
|
718
|
-
|
|
719
|
-
|
|
720
|
-
|
|
721
|
-
|
|
558
|
+
for (const task of options.tasks) verifyCandidateBenchmarkTask(task);
|
|
559
|
+
const material = {
|
|
560
|
+
kind: "agent-candidate-benchmark-suite",
|
|
561
|
+
digestAlgorithm: "rfc8785-sha256",
|
|
562
|
+
taskDigests: options.tasks.map((task) => task.digest),
|
|
563
|
+
reps: options.reps,
|
|
564
|
+
seeds: options.seeds
|
|
565
|
+
};
|
|
566
|
+
return {
|
|
567
|
+
suite: agentCandidateBenchmarkSuiteSchema.parse({
|
|
568
|
+
...material,
|
|
569
|
+
digest: canonicalCandidateDigest(material)
|
|
570
|
+
}),
|
|
571
|
+
tasks: options.tasks
|
|
572
|
+
};
|
|
573
|
+
}
|
|
574
|
+
/** Freeze both complete agent states and their exact held-out work. */
|
|
722
575
|
function sealCandidateExperiment(material) {
|
|
723
|
-
|
|
724
|
-
|
|
725
|
-
|
|
726
|
-
|
|
727
|
-
return verifyCandidateExperiment(parsed);
|
|
576
|
+
return verifyCandidateExperiment(agentCandidateExperimentSchema.parse({
|
|
577
|
+
...material,
|
|
578
|
+
digest: canonicalCandidateDigest(material)
|
|
579
|
+
}));
|
|
728
580
|
}
|
|
729
581
|
function verifyCandidateExperiment(input) {
|
|
730
|
-
|
|
731
|
-
|
|
732
|
-
|
|
733
|
-
|
|
734
|
-
|
|
735
|
-
|
|
736
|
-
|
|
737
|
-
|
|
738
|
-
|
|
739
|
-
}
|
|
582
|
+
const experiment = agentCandidateExperimentSchema.parse(input);
|
|
583
|
+
verifySelfAddressed(experiment, "candidate experiment");
|
|
584
|
+
verifyBundle(experiment.baseline, "baseline bundle");
|
|
585
|
+
verifyBundle(experiment.candidate, "candidate bundle");
|
|
586
|
+
if (experiment.baseline.digest === experiment.candidate.digest) throw new Error("candidate experiment baseline and candidate bundles are identical");
|
|
587
|
+
verifyCandidateBenchmarkSuiteInputs(experiment.benchmark);
|
|
588
|
+
return experiment;
|
|
589
|
+
}
|
|
590
|
+
/** Execute each signed cell for both arms. The callback is Runtime's one executor. */
|
|
740
591
|
async function runCandidateExperiment(options) {
|
|
741
|
-
|
|
742
|
-
|
|
743
|
-
|
|
744
|
-
|
|
745
|
-
|
|
746
|
-
|
|
747
|
-
|
|
748
|
-
|
|
749
|
-
|
|
750
|
-
|
|
751
|
-
|
|
752
|
-
|
|
753
|
-
|
|
754
|
-
|
|
755
|
-
|
|
756
|
-
|
|
757
|
-
|
|
758
|
-
|
|
759
|
-
|
|
760
|
-
|
|
761
|
-
|
|
762
|
-
|
|
763
|
-
|
|
764
|
-
|
|
765
|
-
|
|
766
|
-
|
|
767
|
-
|
|
768
|
-
|
|
769
|
-
|
|
770
|
-
|
|
771
|
-
|
|
772
|
-
|
|
773
|
-
|
|
774
|
-
|
|
775
|
-
|
|
776
|
-
|
|
777
|
-
|
|
778
|
-
|
|
779
|
-
|
|
780
|
-
|
|
781
|
-
|
|
782
|
-
|
|
783
|
-
|
|
784
|
-
|
|
785
|
-
|
|
786
|
-
|
|
787
|
-
|
|
788
|
-
|
|
789
|
-
|
|
790
|
-
|
|
791
|
-
|
|
792
|
-
|
|
793
|
-
|
|
794
|
-
|
|
795
|
-
|
|
796
|
-
|
|
592
|
+
const experiment = verifyCandidateExperiment(options.experiment);
|
|
593
|
+
const { suite, tasks } = experiment.benchmark;
|
|
594
|
+
const maxConcurrency = options.maxConcurrency ?? 2;
|
|
595
|
+
if (!Number.isSafeInteger(maxConcurrency) || maxConcurrency < 1) throw new Error("candidate experiment maxConcurrency must be a positive integer");
|
|
596
|
+
const measurements = new Array(suite.taskDigests.length * suite.reps);
|
|
597
|
+
let nextIndex = 0;
|
|
598
|
+
const lanes = Array.from({ length: Math.min(maxConcurrency, measurements.length) }, async () => {
|
|
599
|
+
while (true) {
|
|
600
|
+
if (options.signal?.aborted) throw abortError(options.signal);
|
|
601
|
+
const index = nextIndex;
|
|
602
|
+
nextIndex += 1;
|
|
603
|
+
if (index >= measurements.length) return;
|
|
604
|
+
const taskIndex = Math.floor(index / suite.reps);
|
|
605
|
+
const repetition = index % suite.reps;
|
|
606
|
+
const task = tasks[taskIndex];
|
|
607
|
+
const seed = suite.seeds[index];
|
|
608
|
+
if (!task || seed === void 0) throw new Error(`candidate experiment cell ${index} has no signed task or seed`);
|
|
609
|
+
const benchmarkCell = {
|
|
610
|
+
suiteDigest: suite.digest,
|
|
611
|
+
taskIndex,
|
|
612
|
+
repetition
|
|
613
|
+
};
|
|
614
|
+
const [baseline, candidate] = await Promise.all([options.execute({
|
|
615
|
+
experiment,
|
|
616
|
+
arm: "baseline",
|
|
617
|
+
bundle: experiment.baseline,
|
|
618
|
+
task,
|
|
619
|
+
benchmarkCell,
|
|
620
|
+
seed,
|
|
621
|
+
...options.signal ? { signal: options.signal } : {}
|
|
622
|
+
}), options.execute({
|
|
623
|
+
experiment,
|
|
624
|
+
arm: "candidate",
|
|
625
|
+
bundle: experiment.candidate,
|
|
626
|
+
task,
|
|
627
|
+
benchmarkCell,
|
|
628
|
+
seed,
|
|
629
|
+
...options.signal ? { signal: options.signal } : {}
|
|
630
|
+
})]);
|
|
631
|
+
const measurement = {
|
|
632
|
+
baseline,
|
|
633
|
+
candidate
|
|
634
|
+
};
|
|
635
|
+
verifyMeasurement(experiment, measurement, index);
|
|
636
|
+
measurements[index] = measurement;
|
|
637
|
+
}
|
|
638
|
+
});
|
|
639
|
+
await Promise.all(lanes);
|
|
640
|
+
return measurements;
|
|
641
|
+
}
|
|
642
|
+
/**
|
|
643
|
+
* Calculate the shared paired decision from any complete receipt shape.
|
|
644
|
+
*
|
|
645
|
+
* Callers still own sealing their tasks, verifying each receipt against its
|
|
646
|
+
* expected arm and state, and proving every expected cell exists. This function
|
|
647
|
+
* only validates the projected measurements and derives their shared decision.
|
|
648
|
+
*/
|
|
797
649
|
function evaluatePairedMeasurements(options) {
|
|
798
|
-
|
|
799
|
-
|
|
800
|
-
|
|
801
|
-
|
|
802
|
-
|
|
803
|
-
|
|
804
|
-
|
|
805
|
-
|
|
806
|
-
|
|
807
|
-
|
|
808
|
-
|
|
809
|
-
|
|
810
|
-
|
|
811
|
-
|
|
812
|
-
|
|
813
|
-
|
|
814
|
-
|
|
815
|
-
|
|
816
|
-
|
|
817
|
-
|
|
818
|
-
|
|
819
|
-
|
|
820
|
-
|
|
821
|
-
|
|
822
|
-
|
|
823
|
-
|
|
824
|
-
|
|
825
|
-
|
|
826
|
-
|
|
827
|
-
|
|
828
|
-
|
|
829
|
-
|
|
830
|
-
|
|
831
|
-
|
|
832
|
-
|
|
833
|
-
|
|
834
|
-
|
|
835
|
-
|
|
836
|
-
|
|
837
|
-
|
|
838
|
-
|
|
839
|
-
|
|
840
|
-
|
|
841
|
-
|
|
842
|
-
|
|
843
|
-
|
|
844
|
-
|
|
845
|
-
|
|
846
|
-
|
|
847
|
-
|
|
848
|
-
|
|
849
|
-
|
|
850
|
-
|
|
851
|
-
|
|
852
|
-
|
|
853
|
-
|
|
854
|
-
|
|
855
|
-
|
|
856
|
-
|
|
857
|
-
|
|
858
|
-
|
|
859
|
-
|
|
860
|
-
|
|
861
|
-
|
|
862
|
-
|
|
863
|
-
|
|
864
|
-
|
|
865
|
-
|
|
866
|
-
|
|
867
|
-
|
|
868
|
-
|
|
869
|
-
|
|
870
|
-
|
|
871
|
-
|
|
872
|
-
|
|
873
|
-
|
|
874
|
-
|
|
875
|
-
|
|
876
|
-
|
|
877
|
-
|
|
878
|
-
|
|
879
|
-
|
|
880
|
-
|
|
881
|
-
|
|
882
|
-
|
|
883
|
-
|
|
884
|
-
|
|
885
|
-
|
|
886
|
-
|
|
887
|
-
|
|
888
|
-
|
|
889
|
-
|
|
890
|
-
|
|
891
|
-
|
|
892
|
-
|
|
893
|
-
|
|
894
|
-
|
|
895
|
-
|
|
896
|
-
|
|
897
|
-
|
|
898
|
-
|
|
899
|
-
|
|
900
|
-
|
|
901
|
-
|
|
902
|
-
|
|
903
|
-
|
|
904
|
-
|
|
905
|
-
|
|
906
|
-
|
|
907
|
-
|
|
908
|
-
|
|
909
|
-
|
|
910
|
-
|
|
911
|
-
|
|
912
|
-
|
|
913
|
-
|
|
914
|
-
|
|
915
|
-
|
|
916
|
-
|
|
917
|
-
|
|
918
|
-
|
|
919
|
-
|
|
920
|
-
|
|
921
|
-
|
|
922
|
-
|
|
923
|
-
|
|
924
|
-
|
|
925
|
-
|
|
926
|
-
|
|
927
|
-
|
|
928
|
-
|
|
929
|
-
|
|
930
|
-
|
|
931
|
-
|
|
932
|
-
|
|
933
|
-
|
|
934
|
-
|
|
935
|
-
|
|
936
|
-
|
|
937
|
-
|
|
938
|
-
|
|
939
|
-
|
|
940
|
-
|
|
941
|
-
|
|
942
|
-
|
|
943
|
-
|
|
944
|
-
|
|
945
|
-
|
|
946
|
-
|
|
947
|
-
|
|
948
|
-
|
|
949
|
-
|
|
950
|
-
|
|
951
|
-
|
|
952
|
-
|
|
953
|
-
significance.fewRuns ? `only ${significance.n} paired runs; ${minProductiveRuns} required` : `paired interval lower bound ${significance.bootstrap.low} did not clear ${deltaThreshold}`
|
|
954
|
-
],
|
|
955
|
-
...powerSufficient ? [] : [power?.recommendation ?? `need at least ${Math.max(3, minProductiveRuns)} paired runs`],
|
|
956
|
-
...regressions.length === 0 ? [] : [`critical dimensions regressed: ${regressions.map((entry) => entry.name).join(", ")}`],
|
|
957
|
-
...missingCriticalDimensions.length === 0 ? [] : [`critical dimensions missing: ${missingCriticalDimensions.join(", ")}`],
|
|
958
|
-
...incompleteRuns.length === 0 ? [] : [`${incompleteRuns.length} benchmark executions did not exit successfully`],
|
|
959
|
-
...failedCandidateResults.length === 0 ? [] : [`candidate failed ${failedCandidateResults.length} benchmark tasks`],
|
|
960
|
-
...budgetPassed ? [] : [`total cost ${totalCostUsd} exceeded budget ${budgetUsd}`]
|
|
961
|
-
];
|
|
962
|
-
return {
|
|
963
|
-
overall: {
|
|
964
|
-
name: "composite",
|
|
965
|
-
direction: "higher-is-better",
|
|
966
|
-
unit: "score",
|
|
967
|
-
...overall
|
|
968
|
-
},
|
|
969
|
-
objectives,
|
|
970
|
-
decision: {
|
|
971
|
-
outcome: shipped ? "ship" : significance.fewRuns || !powerSufficient ? "need_more_work" : "hold",
|
|
972
|
-
reasons: reasons.length > 0 ? reasons : ["all measured checks passed"],
|
|
973
|
-
contributingChecks: checks
|
|
974
|
-
},
|
|
975
|
-
power: {
|
|
976
|
-
sufficient: powerSufficient,
|
|
977
|
-
n: baselineScores.length,
|
|
978
|
-
minimumDetectableDelta: power?.mde ?? 1,
|
|
979
|
-
confidenceLevel: confidence,
|
|
980
|
-
scaleAssumed: power?.scaleAssumed ?? true,
|
|
981
|
-
sharedScorerChannel: options.sharedScorerChannel,
|
|
982
|
-
reason: power?.recommendation ?? `need at least ${Math.max(3, minProductiveRuns)} paired runs`
|
|
983
|
-
},
|
|
984
|
-
executionCostUsd,
|
|
985
|
-
totalCostUsd,
|
|
986
|
-
executionDurationMs
|
|
987
|
-
};
|
|
988
|
-
}
|
|
650
|
+
if (options.measurements.length === 0) throw new Error("paired measurement evaluation requires at least one paired cell");
|
|
651
|
+
const additionalCostUsd = options.additionalCostUsd ?? 0;
|
|
652
|
+
if (!Number.isFinite(additionalCostUsd) || additionalCostUsd < 0) throw new Error("paired measurement evaluation additionalCostUsd must be a non-negative number");
|
|
653
|
+
if (typeof options.sharedScorerChannel !== "boolean") throw new Error("paired measurement evaluation sharedScorerChannel must be a boolean");
|
|
654
|
+
const policy = agentCandidateEvaluationPolicySchema.parse(options.policy);
|
|
655
|
+
const measurements = options.measurements.map((measurement, index) => projectPairedMeasurement(measurement, index, options.adapter));
|
|
656
|
+
const cellIds = measurements.map((measurement) => measurement.cellId);
|
|
657
|
+
if (new Set(cellIds).size !== cellIds.length) throw new Error("paired measurement evaluation cell ids must be unique");
|
|
658
|
+
const dimensions = sharedProjectedDimensions(measurements);
|
|
659
|
+
const baselineScores = measurements.map((measurement) => measurement.baseline.score);
|
|
660
|
+
const candidateScores = measurements.map((measurement) => measurement.candidate.score);
|
|
661
|
+
const { confidenceLevel: confidence, resamples, bootstrapSeed, deltaThreshold, minProductiveRuns, budgetUsd, criticalDimensions, regressionTolerance } = policy;
|
|
662
|
+
const significance = heldoutSignificance({
|
|
663
|
+
before: baselineScores,
|
|
664
|
+
after: candidateScores,
|
|
665
|
+
cellIds
|
|
666
|
+
}, {
|
|
667
|
+
confidence,
|
|
668
|
+
resamples,
|
|
669
|
+
seed: bootstrapSeed,
|
|
670
|
+
statistic: "mean",
|
|
671
|
+
deltaThreshold,
|
|
672
|
+
minProductiveRuns
|
|
673
|
+
});
|
|
674
|
+
const overall = measuredEstimate(baselineScores, candidateScores, {
|
|
675
|
+
confidence,
|
|
676
|
+
resamples,
|
|
677
|
+
seed: bootstrapSeed
|
|
678
|
+
});
|
|
679
|
+
const objectives = [{
|
|
680
|
+
kind: "objective",
|
|
681
|
+
name: "benchmark-score",
|
|
682
|
+
direction: "higher-is-better",
|
|
683
|
+
unit: "score",
|
|
684
|
+
availability: "measured",
|
|
685
|
+
...overall
|
|
686
|
+
}, ...dimensions.map((name, index) => ({
|
|
687
|
+
kind: "dimension",
|
|
688
|
+
objective: "benchmark-score",
|
|
689
|
+
name,
|
|
690
|
+
direction: "higher-is-better",
|
|
691
|
+
unit: "score",
|
|
692
|
+
availability: "measured",
|
|
693
|
+
...measuredEstimate(measurements.map((measurement) => dimensionScore(measurement.baseline, name)), measurements.map((measurement) => dimensionScore(measurement.candidate, name)), {
|
|
694
|
+
confidence,
|
|
695
|
+
resamples,
|
|
696
|
+
seed: bootstrapSeed + index + 1
|
|
697
|
+
})
|
|
698
|
+
}))];
|
|
699
|
+
const cost = measuredEstimate(measurements.map((measurement) => measurement.baseline.costUsd), measurements.map((measurement) => measurement.candidate.costUsd), {
|
|
700
|
+
confidence,
|
|
701
|
+
resamples,
|
|
702
|
+
seed: bootstrapSeed + dimensions.length + 1
|
|
703
|
+
});
|
|
704
|
+
const latency = measuredEstimate(measurements.map((measurement) => measurement.baseline.latencyMs), measurements.map((measurement) => measurement.candidate.latencyMs), {
|
|
705
|
+
confidence,
|
|
706
|
+
resamples,
|
|
707
|
+
seed: bootstrapSeed + dimensions.length + 2
|
|
708
|
+
});
|
|
709
|
+
objectives.push({
|
|
710
|
+
kind: "cost",
|
|
711
|
+
name: "cost",
|
|
712
|
+
direction: "lower-is-better",
|
|
713
|
+
unit: "usd",
|
|
714
|
+
availability: "measured",
|
|
715
|
+
...cost
|
|
716
|
+
}, {
|
|
717
|
+
kind: "latency",
|
|
718
|
+
name: "latency",
|
|
719
|
+
direction: "lower-is-better",
|
|
720
|
+
unit: "milliseconds",
|
|
721
|
+
availability: "measured",
|
|
722
|
+
...latency
|
|
723
|
+
});
|
|
724
|
+
const power = baselineScores.length >= 3 ? powerPreflight({
|
|
725
|
+
baselineComposites: baselineScores,
|
|
726
|
+
pairedN: baselineScores.length,
|
|
727
|
+
deltaThreshold,
|
|
728
|
+
confidence,
|
|
729
|
+
sharedScorerChannel: options.sharedScorerChannel
|
|
730
|
+
}) : void 0;
|
|
731
|
+
const powerSufficient = baselineScores.length >= minProductiveRuns && power !== void 0 && !power.underpowered;
|
|
732
|
+
const guardedDimensions = new Set(criticalDimensions);
|
|
733
|
+
const missingCriticalDimensions = criticalDimensions.filter((dimension) => !dimensions.includes(dimension));
|
|
734
|
+
const regressions = objectives.filter((objective) => objective.kind === "dimension" && guardedDimensions.has(objective.name) && objective.availability === "measured" && objective.confidenceInterval.lower < -regressionTolerance);
|
|
735
|
+
const executionCostUsd = measurements.reduce((sum, measurement) => sum + measurement.baseline.costUsd + measurement.candidate.costUsd, 0);
|
|
736
|
+
const executionDurationMs = measurements.reduce((sum, measurement) => sum + measurement.baseline.latencyMs + measurement.candidate.latencyMs, 0);
|
|
737
|
+
const incompleteRuns = measurements.flatMap((measurement) => [measurement.baseline, measurement.candidate]).filter((run) => !run.completed);
|
|
738
|
+
const failedCandidateResults = measurements.filter((measurement) => !measurement.candidate.passed);
|
|
739
|
+
const totalCostUsd = executionCostUsd + additionalCostUsd;
|
|
740
|
+
const budgetPassed = budgetUsd === void 0 || totalCostUsd <= budgetUsd;
|
|
741
|
+
const checks = [
|
|
742
|
+
{
|
|
743
|
+
name: "paired-significance",
|
|
744
|
+
passed: significance.significant
|
|
745
|
+
},
|
|
746
|
+
{
|
|
747
|
+
name: "statistical-power",
|
|
748
|
+
passed: powerSufficient
|
|
749
|
+
},
|
|
750
|
+
{
|
|
751
|
+
name: "all-runs-completed",
|
|
752
|
+
passed: incompleteRuns.length === 0
|
|
753
|
+
},
|
|
754
|
+
{
|
|
755
|
+
name: "candidate-task-pass",
|
|
756
|
+
passed: failedCandidateResults.length === 0
|
|
757
|
+
},
|
|
758
|
+
{
|
|
759
|
+
name: "critical-dimensions",
|
|
760
|
+
passed: regressions.length === 0 && missingCriticalDimensions.length === 0
|
|
761
|
+
},
|
|
762
|
+
{
|
|
763
|
+
name: "budget",
|
|
764
|
+
passed: budgetPassed
|
|
765
|
+
}
|
|
766
|
+
];
|
|
767
|
+
const shipped = checks.every((check) => check.passed);
|
|
768
|
+
const reasons = [
|
|
769
|
+
...significance.significant ? [] : [significance.fewRuns ? `only ${significance.n} paired runs; ${minProductiveRuns} required` : `paired interval lower bound ${significance.bootstrap.low} did not clear ${deltaThreshold}`],
|
|
770
|
+
...powerSufficient ? [] : [power?.recommendation ?? `need at least ${Math.max(3, minProductiveRuns)} paired runs`],
|
|
771
|
+
...regressions.length === 0 ? [] : [`critical dimensions regressed: ${regressions.map((entry) => entry.name).join(", ")}`],
|
|
772
|
+
...missingCriticalDimensions.length === 0 ? [] : [`critical dimensions missing: ${missingCriticalDimensions.join(", ")}`],
|
|
773
|
+
...incompleteRuns.length === 0 ? [] : [`${incompleteRuns.length} benchmark executions did not exit successfully`],
|
|
774
|
+
...failedCandidateResults.length === 0 ? [] : [`candidate failed ${failedCandidateResults.length} benchmark tasks`],
|
|
775
|
+
...budgetPassed ? [] : [`total cost ${totalCostUsd} exceeded budget ${budgetUsd}`]
|
|
776
|
+
];
|
|
777
|
+
return {
|
|
778
|
+
overall: {
|
|
779
|
+
name: "composite",
|
|
780
|
+
direction: "higher-is-better",
|
|
781
|
+
unit: "score",
|
|
782
|
+
...overall
|
|
783
|
+
},
|
|
784
|
+
objectives,
|
|
785
|
+
decision: {
|
|
786
|
+
outcome: shipped ? "ship" : significance.fewRuns || !powerSufficient ? "need_more_work" : "hold",
|
|
787
|
+
reasons: reasons.length > 0 ? reasons : ["all measured checks passed"],
|
|
788
|
+
contributingChecks: checks
|
|
789
|
+
},
|
|
790
|
+
power: {
|
|
791
|
+
sufficient: powerSufficient,
|
|
792
|
+
n: baselineScores.length,
|
|
793
|
+
minimumDetectableDelta: power?.mde ?? 1,
|
|
794
|
+
confidenceLevel: confidence,
|
|
795
|
+
scaleAssumed: power?.scaleAssumed ?? true,
|
|
796
|
+
sharedScorerChannel: options.sharedScorerChannel,
|
|
797
|
+
reason: power?.recommendation ?? `need at least ${Math.max(3, minProductiveRuns)} paired runs`
|
|
798
|
+
},
|
|
799
|
+
executionCostUsd,
|
|
800
|
+
totalCostUsd,
|
|
801
|
+
executionDurationMs
|
|
802
|
+
};
|
|
803
|
+
}
|
|
804
|
+
/** Build the only publishable comparison: paired statistics over Runtime receipts. */
|
|
989
805
|
function measuredComparisonFromCandidateExperiment(options) {
|
|
990
|
-
|
|
991
|
-
|
|
992
|
-
|
|
993
|
-
|
|
994
|
-
|
|
995
|
-
|
|
996
|
-
|
|
997
|
-
|
|
998
|
-
|
|
999
|
-
|
|
1000
|
-
|
|
1001
|
-
|
|
1002
|
-
|
|
1003
|
-
|
|
1004
|
-
|
|
1005
|
-
|
|
1006
|
-
|
|
1007
|
-
|
|
1008
|
-
|
|
1009
|
-
|
|
1010
|
-
|
|
1011
|
-
|
|
1012
|
-
|
|
1013
|
-
|
|
1014
|
-
|
|
1015
|
-
|
|
1016
|
-
|
|
1017
|
-
|
|
1018
|
-
|
|
1019
|
-
|
|
1020
|
-
|
|
1021
|
-
|
|
1022
|
-
|
|
1023
|
-
|
|
1024
|
-
|
|
1025
|
-
|
|
1026
|
-
|
|
1027
|
-
|
|
1028
|
-
|
|
1029
|
-
|
|
1030
|
-
|
|
1031
|
-
|
|
1032
|
-
|
|
1033
|
-
|
|
1034
|
-
|
|
1035
|
-
|
|
1036
|
-
|
|
1037
|
-
|
|
1038
|
-
|
|
1039
|
-
|
|
1040
|
-
|
|
1041
|
-
|
|
1042
|
-
|
|
1043
|
-
|
|
1044
|
-
|
|
1045
|
-
|
|
1046
|
-
|
|
1047
|
-
|
|
1048
|
-
|
|
1049
|
-
|
|
1050
|
-
|
|
1051
|
-
|
|
1052
|
-
|
|
1053
|
-
});
|
|
1054
|
-
}
|
|
806
|
+
const experiment = verifyCandidateExperiment(options.experiment);
|
|
807
|
+
const measurements = options.measurements.map((measurement, index) => verifyMeasurement(experiment, measurement, index));
|
|
808
|
+
const expectedN = experiment.benchmark.suite.taskDigests.length * experiment.benchmark.suite.reps;
|
|
809
|
+
if (measurements.length !== expectedN) throw new Error(`candidate experiment is incomplete (${measurements.length}/${expectedN} paired cells)`);
|
|
810
|
+
verifyStableProfileMaterialization(measurements);
|
|
811
|
+
if (!options.runId.trim()) throw new Error("candidate experiment runId is required");
|
|
812
|
+
const searchCostUsd = options.searchCostUsd ?? 0;
|
|
813
|
+
const evaluation = evaluatePairedMeasurements({
|
|
814
|
+
measurements: measurements.map((measurement, index) => ({
|
|
815
|
+
cellId: cellIds(experiment)[index],
|
|
816
|
+
...measurement
|
|
817
|
+
})),
|
|
818
|
+
policy: experiment.policy,
|
|
819
|
+
adapter: candidateExecutionEvidenceAdapter,
|
|
820
|
+
sharedScorerChannel: true,
|
|
821
|
+
additionalCostUsd: searchCostUsd
|
|
822
|
+
});
|
|
823
|
+
const diff = deriveCandidateBundleDiff(experiment);
|
|
824
|
+
const searchDurationMs = options.searchDurationMs ?? 0;
|
|
825
|
+
const totalCostUsd = evaluation.totalCostUsd;
|
|
826
|
+
const durationMs = evaluation.executionDurationMs + searchDurationMs;
|
|
827
|
+
const provisional = agentImprovementMeasuredComparisonSchema.parse({
|
|
828
|
+
kind: "agent-improvement-measured-comparison",
|
|
829
|
+
experiment,
|
|
830
|
+
measurements,
|
|
831
|
+
overall: evaluation.overall,
|
|
832
|
+
objectives: evaluation.objectives,
|
|
833
|
+
...options.candidate ? { candidate: options.candidate } : {},
|
|
834
|
+
decision: evaluation.decision,
|
|
835
|
+
power: evaluation.power,
|
|
836
|
+
provenance: {
|
|
837
|
+
kind: "agent-eval-loop",
|
|
838
|
+
schema: "agent-candidate-experiment",
|
|
839
|
+
runId: options.runId,
|
|
840
|
+
recordDigest: canonicalCandidateDigest({}),
|
|
841
|
+
baselineContentHash: experiment.baseline.digest,
|
|
842
|
+
candidateContentHash: experiment.candidate.digest
|
|
843
|
+
},
|
|
844
|
+
diff,
|
|
845
|
+
evaluation: {
|
|
846
|
+
generationsExplored: options.generationsExplored ?? 0,
|
|
847
|
+
searchDurationMs,
|
|
848
|
+
executionDurationMs: evaluation.executionDurationMs,
|
|
849
|
+
durationMs,
|
|
850
|
+
searchCostUsd,
|
|
851
|
+
executionCostUsd: evaluation.executionCostUsd,
|
|
852
|
+
totalCostUsd
|
|
853
|
+
},
|
|
854
|
+
...options.metadata ? { metadata: options.metadata } : {}
|
|
855
|
+
});
|
|
856
|
+
const { recordDigest: _recordDigest, ...provenance } = provisional.provenance;
|
|
857
|
+
return agentImprovementMeasuredComparisonSchema.parse({
|
|
858
|
+
...provisional,
|
|
859
|
+
provenance: {
|
|
860
|
+
...provenance,
|
|
861
|
+
recordDigest: canonicalCandidateDigest({
|
|
862
|
+
...provisional,
|
|
863
|
+
provenance
|
|
864
|
+
})
|
|
865
|
+
}
|
|
866
|
+
});
|
|
867
|
+
}
|
|
868
|
+
/** Recompute every statistic and decision from the signed experiment receipts. */
|
|
1055
869
|
function verifyCandidateExperimentComparison(input) {
|
|
1056
|
-
|
|
1057
|
-
|
|
1058
|
-
|
|
1059
|
-
|
|
1060
|
-
|
|
1061
|
-
|
|
1062
|
-
|
|
1063
|
-
|
|
1064
|
-
|
|
1065
|
-
|
|
1066
|
-
|
|
1067
|
-
|
|
1068
|
-
throw new Error("candidate experiment comparison does not match its Runtime receipts");
|
|
1069
|
-
}
|
|
1070
|
-
return comparison;
|
|
870
|
+
const comparison = agentImprovementMeasuredComparisonSchema.parse(input);
|
|
871
|
+
if (canonicalCandidateDigest(measuredComparisonFromCandidateExperiment({
|
|
872
|
+
experiment: comparison.experiment,
|
|
873
|
+
measurements: comparison.measurements,
|
|
874
|
+
runId: comparison.provenance.runId,
|
|
875
|
+
...comparison.candidate ? { candidate: comparison.candidate } : {},
|
|
876
|
+
generationsExplored: comparison.evaluation.generationsExplored,
|
|
877
|
+
searchDurationMs: comparison.evaluation.searchDurationMs,
|
|
878
|
+
searchCostUsd: comparison.evaluation.searchCostUsd,
|
|
879
|
+
...comparison.metadata ? { metadata: comparison.metadata } : {}
|
|
880
|
+
})) !== canonicalCandidateDigest(comparison)) throw new Error("candidate experiment comparison does not match its Runtime receipts");
|
|
881
|
+
return comparison;
|
|
1071
882
|
}
|
|
1072
883
|
function deriveCandidateBundleDiff(experiment) {
|
|
1073
|
-
|
|
1074
|
-
|
|
1075
|
-
|
|
1076
|
-
|
|
1077
|
-
|
|
1078
|
-
|
|
1079
|
-
|
|
1080
|
-
|
|
1081
|
-
|
|
1082
|
-
|
|
1083
|
-
|
|
1084
|
-
|
|
1085
|
-
|
|
1086
|
-
|
|
1087
|
-
|
|
1088
|
-
|
|
1089
|
-
|
|
1090
|
-
|
|
1091
|
-
|
|
884
|
+
const changed = [
|
|
885
|
+
"profile",
|
|
886
|
+
"code",
|
|
887
|
+
"execution",
|
|
888
|
+
"knowledge",
|
|
889
|
+
"memory"
|
|
890
|
+
].flatMap((surface) => {
|
|
891
|
+
const baseline = experiment.baseline[surface] ?? null;
|
|
892
|
+
const candidate = experiment.candidate[surface] ?? null;
|
|
893
|
+
const baselineDigest = canonicalCandidateDigest(baseline);
|
|
894
|
+
const candidateDigest = canonicalCandidateDigest(candidate);
|
|
895
|
+
if (baselineDigest === candidateDigest) return [];
|
|
896
|
+
return [[
|
|
897
|
+
`--- baseline/${surface} (${baselineDigest})`,
|
|
898
|
+
`+++ candidate/${surface} (${candidateDigest})`,
|
|
899
|
+
JSON.stringify({
|
|
900
|
+
baseline,
|
|
901
|
+
candidate
|
|
902
|
+
}, null, 2)
|
|
903
|
+
].join("\n")];
|
|
904
|
+
});
|
|
905
|
+
if (changed.length === 0) throw new Error("candidate experiment has no changed candidate surface");
|
|
906
|
+
return changed.join("\n\n");
|
|
1092
907
|
}
|
|
1093
908
|
function verifyCandidateBenchmarkTask(input) {
|
|
1094
|
-
|
|
1095
|
-
|
|
1096
|
-
|
|
909
|
+
const task = agentCandidateBenchmarkTaskSchema.parse(input);
|
|
910
|
+
verifySelfAddressed(task, "candidate benchmark task");
|
|
911
|
+
return task;
|
|
1097
912
|
}
|
|
1098
913
|
function verifyCandidateBenchmarkSuiteInputs(input) {
|
|
1099
|
-
|
|
1100
|
-
|
|
1101
|
-
|
|
1102
|
-
|
|
1103
|
-
|
|
1104
|
-
|
|
1105
|
-
|
|
1106
|
-
|
|
1107
|
-
|
|
1108
|
-
|
|
1109
|
-
|
|
1110
|
-
throw new Error(`candidate benchmark task ${index} does not match the signed suite`);
|
|
1111
|
-
}
|
|
1112
|
-
});
|
|
1113
|
-
return { suite, tasks: candidate.tasks };
|
|
914
|
+
if (input === null || typeof input !== "object" || Array.isArray(input)) throw new Error("candidate benchmark suite inputs must be an object");
|
|
915
|
+
const candidate = input;
|
|
916
|
+
const suite = verifyCandidateBenchmarkSuite(candidate.suite);
|
|
917
|
+
if (!Array.isArray(candidate.tasks) || candidate.tasks.length !== suite.taskDigests.length) throw new Error("candidate benchmark suite task count does not match its signed digests");
|
|
918
|
+
candidate.tasks.forEach((task, index) => {
|
|
919
|
+
if (verifyCandidateBenchmarkTask(task).digest !== suite.taskDigests[index]) throw new Error(`candidate benchmark task ${index} does not match the signed suite`);
|
|
920
|
+
});
|
|
921
|
+
return {
|
|
922
|
+
suite,
|
|
923
|
+
tasks: candidate.tasks
|
|
924
|
+
};
|
|
1114
925
|
}
|
|
1115
926
|
function verifyCandidateBenchmarkSuite(input) {
|
|
1116
|
-
|
|
1117
|
-
|
|
1118
|
-
|
|
927
|
+
const suite = agentCandidateBenchmarkSuiteSchema.parse(input);
|
|
928
|
+
verifySelfAddressed(suite, "candidate benchmark suite");
|
|
929
|
+
return suite;
|
|
1119
930
|
}
|
|
1120
931
|
function verifyBundle(input, label) {
|
|
1121
|
-
|
|
1122
|
-
|
|
1123
|
-
|
|
932
|
+
const bundle = agentCandidateBundleSchema.parse(input);
|
|
933
|
+
verifySelfAddressed(bundle, label);
|
|
934
|
+
return bundle;
|
|
1124
935
|
}
|
|
1125
936
|
function verifyMeasurement(experiment, input, index) {
|
|
1126
|
-
|
|
1127
|
-
|
|
1128
|
-
|
|
1129
|
-
|
|
1130
|
-
|
|
1131
|
-
|
|
1132
|
-
|
|
1133
|
-
|
|
1134
|
-
|
|
1135
|
-
|
|
1136
|
-
|
|
1137
|
-
|
|
1138
|
-
|
|
1139
|
-
|
|
1140
|
-
|
|
1141
|
-
|
|
1142
|
-
|
|
1143
|
-
|
|
1144
|
-
|
|
1145
|
-
|
|
1146
|
-
|
|
1147
|
-
|
|
1148
|
-
|
|
1149
|
-
}
|
|
1150
|
-
const baselinePlan = baseline.materializationReceipt.executionPlan.material;
|
|
1151
|
-
const candidatePlan = candidate.materializationReceipt.executionPlan.material;
|
|
1152
|
-
if (baselinePlan.executionId === candidatePlan.executionId || baselinePlan.runCell.digest === candidatePlan.runCell.digest || baseline.materializationReceipt.digest === candidate.materializationReceipt.digest || baseline.receipt.digest === candidate.receipt.digest || baseline.digest === candidate.digest) {
|
|
1153
|
-
throw new Error(`candidate experiment measurement ${index} reused one execution across arms`);
|
|
1154
|
-
}
|
|
1155
|
-
return { baseline, candidate };
|
|
937
|
+
const suite = experiment.benchmark.suite;
|
|
938
|
+
const taskIndex = Math.floor(index / suite.reps);
|
|
939
|
+
const repetition = index % suite.reps;
|
|
940
|
+
const task = experiment.benchmark.tasks[taskIndex];
|
|
941
|
+
const seed = suite.seeds[index];
|
|
942
|
+
if (!task || seed === void 0) throw new Error(`candidate experiment measurement ${index} is outside the signed suite`);
|
|
943
|
+
const baseline = verifyExecutionEvidence(input.baseline);
|
|
944
|
+
const candidate = verifyExecutionEvidence(input.candidate);
|
|
945
|
+
for (const [arm, evidence] of [["baseline", baseline], ["candidate", candidate]]) {
|
|
946
|
+
const bundle = experiment[arm];
|
|
947
|
+
const materialization = evidence.materializationReceipt;
|
|
948
|
+
const runCell = materialization.executionPlan.material.runCell;
|
|
949
|
+
verifySelfAddressed(runCell, "candidate run cell");
|
|
950
|
+
if (runCell.experimentDigest !== experiment.digest || runCell.arm !== arm || runCell.bundleDigest !== bundle.digest || runCell.suiteDigest !== suite.digest || runCell.taskDigest !== task.digest || runCell.taskIndex !== taskIndex || runCell.repetition !== repetition || runCell.seed !== seed || runCell.attempt > task.attempt.maxAttempts || materialization.bundleDigest !== bundle.digest || materialization.benchmark.suite.digest !== suite.digest || materialization.benchmark.task.digest !== task.digest || materialization.codeKind !== bundle.code.kind || materialization.profileActivation.profilePlan.material.sourceProfileDigest !== canonicalCandidateDigest(bundle.profile) || evidence.receipt.runCellDigest !== runCell.digest || JSON.stringify(materialization.resolvedModel) !== JSON.stringify(task.model)) throw new Error(`candidate experiment measurement ${index} substituted its ${arm} arm`);
|
|
951
|
+
verifyTaskOutcome(task, evidence, index, arm);
|
|
952
|
+
}
|
|
953
|
+
const baselinePlan = baseline.materializationReceipt.executionPlan.material;
|
|
954
|
+
const candidatePlan = candidate.materializationReceipt.executionPlan.material;
|
|
955
|
+
if (baselinePlan.executionId === candidatePlan.executionId || baselinePlan.runCell.digest === candidatePlan.runCell.digest || baseline.materializationReceipt.digest === candidate.materializationReceipt.digest || baseline.receipt.digest === candidate.receipt.digest || baseline.digest === candidate.digest) throw new Error(`candidate experiment measurement ${index} reused one execution across arms`);
|
|
956
|
+
return {
|
|
957
|
+
baseline,
|
|
958
|
+
candidate
|
|
959
|
+
};
|
|
1156
960
|
}
|
|
1157
961
|
function verifyExecutionEvidence(input) {
|
|
1158
|
-
|
|
1159
|
-
|
|
1160
|
-
|
|
1161
|
-
|
|
1162
|
-
|
|
1163
|
-
|
|
1164
|
-
|
|
1165
|
-
|
|
1166
|
-
|
|
1167
|
-
|
|
1168
|
-
|
|
1169
|
-
verifyMaterialAddressed(evidence.materializationReceipt.executionPlan, "candidate execution plan");
|
|
1170
|
-
verifySelfAddressed(evidence.receipt, "candidate run receipt");
|
|
1171
|
-
verifyMaterialAddressed(evidence.receipt.modelSettlement, "candidate model settlement");
|
|
1172
|
-
verifyMaterialAddressed(evidence.receipt.taskOutcome, "candidate task outcome");
|
|
1173
|
-
verifyMaterialAddressed(evidence.receipt.benchmarkResult, "candidate benchmark result");
|
|
1174
|
-
return evidence;
|
|
962
|
+
const evidence = candidateExecutionEvidenceSchema.parse(input);
|
|
963
|
+
verifySelfAddressed(evidence, "candidate execution evidence");
|
|
964
|
+
verifySelfAddressed(evidence.materializationReceipt, "candidate materialization receipt");
|
|
965
|
+
verifySelfAddressed(evidence.materializationReceipt.profileActivation, "candidate profile activation");
|
|
966
|
+
verifyMaterialAddressed(evidence.materializationReceipt.profileActivation.profilePlan, "candidate profile plan");
|
|
967
|
+
verifyMaterialAddressed(evidence.materializationReceipt.executionPlan, "candidate execution plan");
|
|
968
|
+
verifySelfAddressed(evidence.receipt, "candidate run receipt");
|
|
969
|
+
verifyMaterialAddressed(evidence.receipt.modelSettlement, "candidate model settlement");
|
|
970
|
+
verifyMaterialAddressed(evidence.receipt.taskOutcome, "candidate task outcome");
|
|
971
|
+
verifyMaterialAddressed(evidence.receipt.benchmarkResult, "candidate benchmark result");
|
|
972
|
+
return evidence;
|
|
1175
973
|
}
|
|
1176
974
|
function verifySelfAddressed(document, label) {
|
|
1177
|
-
|
|
1178
|
-
throw new Error(`${label} digest is invalid`);
|
|
1179
|
-
}
|
|
975
|
+
if (canonicalCandidateDigest(omitTopLevelDigest(document)) !== document.digest) throw new Error(`${label} digest is invalid`);
|
|
1180
976
|
}
|
|
1181
977
|
function verifyTaskOutcome(task, evidence, index, arm) {
|
|
1182
|
-
|
|
1183
|
-
|
|
1184
|
-
|
|
1185
|
-
|
|
1186
|
-
|
|
1187
|
-
|
|
1188
|
-
|
|
1189
|
-
|
|
1190
|
-
|
|
1191
|
-
|
|
1192
|
-
|
|
1193
|
-
|
|
1194
|
-
|
|
1195
|
-
|
|
1196
|
-
|
|
1197
|
-
|
|
1198
|
-
|
|
1199
|
-
|
|
1200
|
-
|
|
1201
|
-
|
|
1202
|
-
|
|
1203
|
-
|
|
1204
|
-
|
|
1205
|
-
|
|
1206
|
-
|
|
1207
|
-
|
|
1208
|
-
|
|
1209
|
-
|
|
1210
|
-
|
|
1211
|
-
|
|
1212
|
-
|
|
978
|
+
const outcome = evidence.receipt.taskOutcome.material.outcome;
|
|
979
|
+
const result = evidence.receipt.benchmarkResult.material;
|
|
980
|
+
const prefix = `candidate experiment measurement ${index} ${arm}`;
|
|
981
|
+
if (result.evidence.sha256 === task.grader.artifact.sha256) throw new Error(`${prefix} reused grader bytes as grading evidence`);
|
|
982
|
+
const usage = combinedUsage(evidence);
|
|
983
|
+
const usageChecks = [
|
|
984
|
+
[
|
|
985
|
+
usage.modelCalls,
|
|
986
|
+
task.limits.maxModelCalls,
|
|
987
|
+
"model calls"
|
|
988
|
+
],
|
|
989
|
+
[
|
|
990
|
+
usage.inputTokens,
|
|
991
|
+
task.limits.maxInputTokens,
|
|
992
|
+
"input tokens"
|
|
993
|
+
],
|
|
994
|
+
[
|
|
995
|
+
usage.outputTokens,
|
|
996
|
+
task.limits.maxOutputTokens,
|
|
997
|
+
"output tokens"
|
|
998
|
+
],
|
|
999
|
+
[
|
|
1000
|
+
usage.costUsdNanos,
|
|
1001
|
+
Math.round(task.limits.maxCostUsd * 1e9),
|
|
1002
|
+
"cost"
|
|
1003
|
+
]
|
|
1004
|
+
];
|
|
1005
|
+
for (const [actual, maximum, label] of usageChecks) if (actual > maximum) throw new Error(`${prefix} ${label} ${actual} exceeds the signed limit ${maximum}`);
|
|
1006
|
+
if (outcome.kind !== task.outcome.kind) throw new Error(`${prefix} returned an outcome outside the signed task contract`);
|
|
1007
|
+
if (task.outcome.kind === "output") {
|
|
1008
|
+
if (outcome.kind !== "output" || outcome.spec.mediaType !== task.outcome.mediaType || outcome.spec.maxBytes !== task.outcome.maxBytes) throw new Error(`${prefix} changed the signed output contract`);
|
|
1009
|
+
return;
|
|
1010
|
+
}
|
|
1011
|
+
const repository = task.repository;
|
|
1012
|
+
if (outcome.kind !== "workspace" || repository === void 0 || outcome.baseRepository.identity !== repository.identity || outcome.baseRepository.rootIdentity !== repository.rootIdentity || outcome.baseRepository.commit !== repository.baseCommit || outcome.baseRepository.tree !== repository.baseTree) throw new Error(`${prefix} did not start from the signed repository state`);
|
|
1213
1013
|
}
|
|
1214
1014
|
function verifyStableProfileMaterialization(measurements) {
|
|
1215
|
-
|
|
1216
|
-
|
|
1217
|
-
|
|
1218
|
-
|
|
1219
|
-
|
|
1220
|
-
|
|
1221
|
-
|
|
1222
|
-
|
|
1223
|
-
|
|
1224
|
-
|
|
1225
|
-
|
|
1226
|
-
|
|
1227
|
-
|
|
1228
|
-
|
|
1229
|
-
|
|
1230
|
-
);
|
|
1231
|
-
}
|
|
1232
|
-
}
|
|
1233
|
-
}
|
|
1015
|
+
for (const arm of ["baseline", "candidate"]) {
|
|
1016
|
+
const expected = measurements[0]?.[arm].materializationReceipt.profileActivation;
|
|
1017
|
+
if (!expected) throw new Error("candidate experiment contains no profile materialization");
|
|
1018
|
+
const expectedDigest = canonicalCandidateDigest({
|
|
1019
|
+
profilePlanDigest: expected.profilePlan.digest,
|
|
1020
|
+
files: expected.files
|
|
1021
|
+
});
|
|
1022
|
+
for (const [index, measurement] of measurements.entries()) {
|
|
1023
|
+
const activation = measurement[arm].materializationReceipt.profileActivation;
|
|
1024
|
+
if (canonicalCandidateDigest({
|
|
1025
|
+
profilePlanDigest: activation.profilePlan.digest,
|
|
1026
|
+
files: activation.files
|
|
1027
|
+
}) !== expectedDigest) throw new Error(`candidate experiment measurement ${index} ${arm} materialized a different profile`);
|
|
1028
|
+
}
|
|
1029
|
+
}
|
|
1234
1030
|
}
|
|
1235
1031
|
function completedSuccessfully(evidence) {
|
|
1236
|
-
|
|
1237
|
-
|
|
1032
|
+
const termination = evidence.receipt.termination;
|
|
1033
|
+
return termination.kind === "exit" && termination.exitCode === 0;
|
|
1238
1034
|
}
|
|
1239
1035
|
function verifyMaterialAddressed(evidence, label) {
|
|
1240
|
-
|
|
1241
|
-
throw new Error(`${label} digest is invalid`);
|
|
1242
|
-
}
|
|
1036
|
+
if (canonicalCandidateDigest(evidence.material) !== evidence.digest) throw new Error(`${label} digest is invalid`);
|
|
1243
1037
|
}
|
|
1244
1038
|
function projectPairedMeasurement(measurement, index, adapter) {
|
|
1245
|
-
|
|
1246
|
-
|
|
1247
|
-
|
|
1248
|
-
|
|
1249
|
-
|
|
1250
|
-
|
|
1251
|
-
candidate: projectRun(measurement.candidate, adapter, `paired measurement ${index} candidate`)
|
|
1252
|
-
};
|
|
1039
|
+
if (typeof measurement.cellId !== "string" || !measurement.cellId.trim()) throw new Error(`paired measurement ${index} requires a cell id`);
|
|
1040
|
+
return {
|
|
1041
|
+
cellId: measurement.cellId,
|
|
1042
|
+
baseline: projectRun(measurement.baseline, adapter, `paired measurement ${index} baseline`),
|
|
1043
|
+
candidate: projectRun(measurement.candidate, adapter, `paired measurement ${index} candidate`)
|
|
1044
|
+
};
|
|
1253
1045
|
}
|
|
1254
1046
|
function projectRun(run, adapter, label) {
|
|
1255
|
-
|
|
1256
|
-
|
|
1257
|
-
|
|
1258
|
-
|
|
1259
|
-
|
|
1260
|
-
|
|
1261
|
-
|
|
1262
|
-
|
|
1263
|
-
|
|
1264
|
-
|
|
1265
|
-
|
|
1266
|
-
|
|
1267
|
-
|
|
1268
|
-
|
|
1269
|
-
|
|
1270
|
-
|
|
1271
|
-
|
|
1272
|
-
|
|
1273
|
-
|
|
1274
|
-
return {
|
|
1275
|
-
score: finiteMeasurement(adapter.score(run), `${label} score`),
|
|
1276
|
-
dimensions,
|
|
1277
|
-
costUsd: nonNegativeMeasurement(adapter.costUsd(run), `${label} cost`),
|
|
1278
|
-
latencyMs: nonNegativeMeasurement(adapter.latencyMs(run), `${label} latency`),
|
|
1279
|
-
completed,
|
|
1280
|
-
passed
|
|
1281
|
-
};
|
|
1047
|
+
const suppliedDimensions = adapter.dimensions(run);
|
|
1048
|
+
if (!Array.isArray(suppliedDimensions)) throw new Error(`${label} dimensions must be an array`);
|
|
1049
|
+
const dimensions = /* @__PURE__ */ new Map();
|
|
1050
|
+
for (const dimension of suppliedDimensions) {
|
|
1051
|
+
if (typeof dimension.name !== "string" || !dimension.name.trim()) throw new Error(`${label} contains an unnamed dimension`);
|
|
1052
|
+
if (dimensions.has(dimension.name)) throw new Error(`${label} repeats dimension '${dimension.name}'`);
|
|
1053
|
+
dimensions.set(dimension.name, finiteMeasurement(dimension.score, `${label} ${dimension.name}`));
|
|
1054
|
+
}
|
|
1055
|
+
const completed = adapter.completed(run);
|
|
1056
|
+
const passed = adapter.passed(run);
|
|
1057
|
+
if (typeof completed !== "boolean" || typeof passed !== "boolean") throw new Error(`${label} completion and pass values must be booleans`);
|
|
1058
|
+
return {
|
|
1059
|
+
score: finiteMeasurement(adapter.score(run), `${label} score`),
|
|
1060
|
+
dimensions,
|
|
1061
|
+
costUsd: nonNegativeMeasurement(adapter.costUsd(run), `${label} cost`),
|
|
1062
|
+
latencyMs: nonNegativeMeasurement(adapter.latencyMs(run), `${label} latency`),
|
|
1063
|
+
completed,
|
|
1064
|
+
passed
|
|
1065
|
+
};
|
|
1282
1066
|
}
|
|
1283
1067
|
function sharedProjectedDimensions(measurements) {
|
|
1284
|
-
|
|
1285
|
-
|
|
1286
|
-
|
|
1287
|
-
|
|
1288
|
-
|
|
1289
|
-
|
|
1290
|
-
const actual = [...run.dimensions.keys()];
|
|
1291
|
-
if (JSON.stringify(actual) !== JSON.stringify(expected)) {
|
|
1292
|
-
throw new Error(`paired measurement ${index} ${arm} dimensions do not match the suite`);
|
|
1293
|
-
}
|
|
1294
|
-
}
|
|
1295
|
-
}
|
|
1296
|
-
return expected;
|
|
1068
|
+
const expected = [...measurements[0].baseline.dimensions.keys()];
|
|
1069
|
+
for (const [index, measurement] of measurements.entries()) for (const [arm, run] of [["baseline", measurement.baseline], ["candidate", measurement.candidate]]) {
|
|
1070
|
+
const actual = [...run.dimensions.keys()];
|
|
1071
|
+
if (JSON.stringify(actual) !== JSON.stringify(expected)) throw new Error(`paired measurement ${index} ${arm} dimensions do not match the suite`);
|
|
1072
|
+
}
|
|
1073
|
+
return expected;
|
|
1297
1074
|
}
|
|
1298
1075
|
function dimensionScore(run, name) {
|
|
1299
|
-
|
|
1300
|
-
|
|
1301
|
-
|
|
1076
|
+
const value = run.dimensions.get(name);
|
|
1077
|
+
if (value === void 0) throw new Error(`paired measurement is missing dimension '${name}'`);
|
|
1078
|
+
return value;
|
|
1302
1079
|
}
|
|
1303
1080
|
function measuredEstimate(baseline, candidate, options) {
|
|
1304
|
-
|
|
1305
|
-
|
|
1306
|
-
|
|
1307
|
-
|
|
1308
|
-
|
|
1309
|
-
|
|
1310
|
-
|
|
1311
|
-
|
|
1312
|
-
|
|
1313
|
-
|
|
1314
|
-
|
|
1315
|
-
|
|
1316
|
-
|
|
1317
|
-
|
|
1318
|
-
|
|
1319
|
-
|
|
1320
|
-
|
|
1321
|
-
|
|
1322
|
-
|
|
1323
|
-
|
|
1324
|
-
|
|
1325
|
-
|
|
1326
|
-
|
|
1081
|
+
const bootstrap = pairedBootstrap(baseline, candidate, {
|
|
1082
|
+
confidence: options.confidence,
|
|
1083
|
+
resamples: options.resamples,
|
|
1084
|
+
statistic: "mean",
|
|
1085
|
+
seed: options.seed
|
|
1086
|
+
});
|
|
1087
|
+
const baselineMean = mean(baseline);
|
|
1088
|
+
const candidateMean = mean(candidate);
|
|
1089
|
+
const delta = candidateMean - baselineMean;
|
|
1090
|
+
return {
|
|
1091
|
+
baseline: baselineMean,
|
|
1092
|
+
candidate: candidateMean,
|
|
1093
|
+
delta,
|
|
1094
|
+
confidenceInterval: {
|
|
1095
|
+
level: bootstrap.confidence,
|
|
1096
|
+
lower: Math.min(bootstrap.low, delta),
|
|
1097
|
+
upper: Math.max(bootstrap.high, delta),
|
|
1098
|
+
method: "paired-bootstrap",
|
|
1099
|
+
statistic: "mean",
|
|
1100
|
+
resamples: bootstrap.resamples
|
|
1101
|
+
},
|
|
1102
|
+
n: bootstrap.n
|
|
1103
|
+
};
|
|
1327
1104
|
}
|
|
1328
1105
|
function finiteMeasurement(value, label) {
|
|
1329
|
-
|
|
1330
|
-
|
|
1106
|
+
if (!Number.isFinite(value)) throw new Error(`${label} must be finite`);
|
|
1107
|
+
return value;
|
|
1331
1108
|
}
|
|
1332
1109
|
function nonNegativeMeasurement(value, label) {
|
|
1333
|
-
|
|
1334
|
-
|
|
1335
|
-
}
|
|
1336
|
-
|
|
1337
|
-
|
|
1338
|
-
|
|
1339
|
-
|
|
1340
|
-
|
|
1341
|
-
|
|
1342
|
-
|
|
1110
|
+
if (!Number.isFinite(value) || value < 0) throw new Error(`${label} must be non-negative`);
|
|
1111
|
+
return value;
|
|
1112
|
+
}
|
|
1113
|
+
const candidateExecutionEvidenceAdapter = {
|
|
1114
|
+
score: (evidence) => evidence.receipt.benchmarkResult.material.score,
|
|
1115
|
+
dimensions: (evidence) => evidence.receipt.benchmarkResult.material.dimensions,
|
|
1116
|
+
costUsd: costFromEvidence,
|
|
1117
|
+
latencyMs: latencyFromEvidence,
|
|
1118
|
+
completed: completedSuccessfully,
|
|
1119
|
+
passed: (evidence) => evidence.receipt.benchmarkResult.material.passed
|
|
1343
1120
|
};
|
|
1344
1121
|
function costFromEvidence(evidence) {
|
|
1345
|
-
|
|
1122
|
+
return combinedUsage(evidence).costUsdNanos / 1e9;
|
|
1346
1123
|
}
|
|
1347
1124
|
function latencyFromEvidence(evidence) {
|
|
1348
|
-
|
|
1125
|
+
return evidence.receipt.timing.durationMs + evidence.receipt.benchmarkResult.material.grading.timing.durationMs;
|
|
1349
1126
|
}
|
|
1350
1127
|
function combinedUsage(evidence) {
|
|
1351
|
-
|
|
1352
|
-
|
|
1353
|
-
|
|
1354
|
-
|
|
1355
|
-
|
|
1356
|
-
|
|
1357
|
-
|
|
1358
|
-
|
|
1359
|
-
|
|
1360
|
-
|
|
1128
|
+
const candidate = evidence.receipt.modelSettlement.material.usage;
|
|
1129
|
+
const grader = evidence.receipt.benchmarkResult.material.grading.usage;
|
|
1130
|
+
return {
|
|
1131
|
+
inputTokens: candidate.inputTokens + grader.inputTokens,
|
|
1132
|
+
outputTokens: candidate.outputTokens + grader.outputTokens,
|
|
1133
|
+
cachedInputTokens: candidate.cachedInputTokens + grader.cachedInputTokens,
|
|
1134
|
+
reasoningTokens: candidate.reasoningTokens + grader.reasoningTokens,
|
|
1135
|
+
modelCalls: candidate.modelCalls + grader.modelCalls,
|
|
1136
|
+
costUsdNanos: candidate.costUsdNanos + grader.costUsdNanos
|
|
1137
|
+
};
|
|
1361
1138
|
}
|
|
1362
1139
|
function cellIds(experiment) {
|
|
1363
|
-
|
|
1364
|
-
|
|
1365
|
-
|
|
1366
|
-
|
|
1367
|
-
|
|
1368
|
-
|
|
1140
|
+
const { suite, tasks } = experiment.benchmark;
|
|
1141
|
+
return suite.seeds.map((_, index) => {
|
|
1142
|
+
const taskIndex = Math.floor(index / suite.reps);
|
|
1143
|
+
const repetition = index % suite.reps;
|
|
1144
|
+
return `${tasks[taskIndex]?.scenario.id ?? taskIndex}:${repetition}`;
|
|
1145
|
+
});
|
|
1369
1146
|
}
|
|
1370
1147
|
function mean(values) {
|
|
1371
|
-
|
|
1372
|
-
|
|
1148
|
+
if (values.length === 0) throw new Error("candidate experiment requires measured values");
|
|
1149
|
+
return values.reduce((sum, value) => sum + value, 0) / values.length;
|
|
1373
1150
|
}
|
|
1374
1151
|
function abortError(signal) {
|
|
1375
|
-
|
|
1376
|
-
}
|
|
1377
|
-
|
|
1378
|
-
|
|
1379
|
-
|
|
1380
|
-
|
|
1381
|
-
|
|
1382
|
-
|
|
1383
|
-
|
|
1384
|
-
|
|
1385
|
-
|
|
1152
|
+
return signal.reason instanceof Error ? signal.reason : /* @__PURE__ */ new Error("candidate experiment aborted");
|
|
1153
|
+
}
|
|
1154
|
+
//#endregion
|
|
1155
|
+
//#region src/contract/intake/run-record-dir.ts
|
|
1156
|
+
/**
|
|
1157
|
+
* # `intake/run-record-dir` — load a directory or file of `RunRecord`s.
|
|
1158
|
+
*
|
|
1159
|
+
* The on-disk counterpart to the in-memory intake adapters: point it at a
|
|
1160
|
+
* single `.json` (array) / `.jsonl` (one record per line) file or at a
|
|
1161
|
+
* directory of such files, and it returns the substrate-canonical
|
|
1162
|
+
* `RunRecord[]` ready for `analyzeRuns({ runs })`.
|
|
1163
|
+
*
|
|
1164
|
+
* Validation is at the boundary: each parsed object goes through
|
|
1165
|
+
* `parseRunRecordSafe`. By default an invalid record fails loud with its
|
|
1166
|
+
* file + index; pass `onInvalid: 'collect'` to keep the valid records and
|
|
1167
|
+
* receive the rejects as structured diagnostics instead.
|
|
1168
|
+
*/
|
|
1169
|
+
const ANALYSIS_ARTIFACT$1 = "analysis.json";
|
|
1386
1170
|
function defaultInclude(fileName) {
|
|
1387
|
-
|
|
1388
|
-
|
|
1389
|
-
}
|
|
1171
|
+
if (fileName === ANALYSIS_ARTIFACT$1) return false;
|
|
1172
|
+
return fileName.endsWith(".json") || fileName.endsWith(".jsonl");
|
|
1173
|
+
}
|
|
1174
|
+
/**
|
|
1175
|
+
* Resolve a file or directory path into validated `RunRecord[]`.
|
|
1176
|
+
*
|
|
1177
|
+
* A `.json` file must parse to a top-level array; a `.jsonl` file is one
|
|
1178
|
+
* record per non-empty line. Directories are read shallowly by default
|
|
1179
|
+
* (set `recursive` to descend); the `analysis.json` output artifact is
|
|
1180
|
+
* always excluded.
|
|
1181
|
+
*/
|
|
1390
1182
|
async function fromRunRecordDir(path, options = {}) {
|
|
1391
|
-
|
|
1392
|
-
|
|
1393
|
-
|
|
1394
|
-
|
|
1395
|
-
|
|
1396
|
-
|
|
1397
|
-
|
|
1398
|
-
|
|
1399
|
-
|
|
1400
|
-
|
|
1401
|
-
|
|
1402
|
-
|
|
1403
|
-
|
|
1404
|
-
|
|
1405
|
-
|
|
1406
|
-
|
|
1407
|
-
|
|
1408
|
-
|
|
1409
|
-
|
|
1410
|
-
|
|
1411
|
-
|
|
1412
|
-
|
|
1413
|
-
|
|
1414
|
-
|
|
1415
|
-
|
|
1183
|
+
const onInvalid = options.onInvalid ?? "throw";
|
|
1184
|
+
const include = options.include ?? defaultInclude;
|
|
1185
|
+
const filePaths = (await stat(path)).isDirectory() ? await collectFiles(path, include, options.recursive ?? false) : [path];
|
|
1186
|
+
const runs = [];
|
|
1187
|
+
const rejected = [];
|
|
1188
|
+
for (const file of filePaths) {
|
|
1189
|
+
const raw = await parseRecordFile(file);
|
|
1190
|
+
for (const { index, value } of raw) {
|
|
1191
|
+
const parsed = parseRunRecordSafe(value);
|
|
1192
|
+
if (parsed.ok) {
|
|
1193
|
+
runs.push(parsed.value);
|
|
1194
|
+
continue;
|
|
1195
|
+
}
|
|
1196
|
+
const rejection = {
|
|
1197
|
+
file,
|
|
1198
|
+
index,
|
|
1199
|
+
reason: parsed.error.message
|
|
1200
|
+
};
|
|
1201
|
+
if (onInvalid === "throw") throw new Error(`fromRunRecordDir: invalid RunRecord in '${file}' at index ${index}: ${parsed.error.message}`);
|
|
1202
|
+
rejected.push(rejection);
|
|
1203
|
+
}
|
|
1204
|
+
}
|
|
1205
|
+
return {
|
|
1206
|
+
runs,
|
|
1207
|
+
rejected,
|
|
1208
|
+
files: filePaths
|
|
1209
|
+
};
|
|
1210
|
+
}
|
|
1211
|
+
/** Read a single `.json` / `.jsonl` file into `{ index, value }` pairs. A
|
|
1212
|
+
* malformed JSONL line throws with its line number rather than being skipped —
|
|
1213
|
+
* silent line-dropping is how corpora quietly shrink. */
|
|
1416
1214
|
async function parseRecordFile(file) {
|
|
1417
|
-
|
|
1418
|
-
|
|
1419
|
-
|
|
1420
|
-
|
|
1421
|
-
|
|
1422
|
-
|
|
1423
|
-
|
|
1424
|
-
|
|
1425
|
-
|
|
1426
|
-
|
|
1427
|
-
|
|
1428
|
-
|
|
1429
|
-
|
|
1430
|
-
|
|
1431
|
-
|
|
1432
|
-
|
|
1433
|
-
|
|
1434
|
-
|
|
1435
|
-
|
|
1436
|
-
|
|
1437
|
-
|
|
1438
|
-
|
|
1439
|
-
|
|
1440
|
-
|
|
1441
|
-
|
|
1215
|
+
const trimmed = (await readFile(file, "utf8")).trim();
|
|
1216
|
+
if (trimmed.length === 0) return [];
|
|
1217
|
+
if (trimmed.startsWith("[")) {
|
|
1218
|
+
const parsed = JSON.parse(trimmed);
|
|
1219
|
+
if (!Array.isArray(parsed)) throw new Error(`fromRunRecordDir: file '${file}' did not parse to an array`);
|
|
1220
|
+
return parsed.map((value, index) => ({
|
|
1221
|
+
index,
|
|
1222
|
+
value
|
|
1223
|
+
}));
|
|
1224
|
+
}
|
|
1225
|
+
const out = [];
|
|
1226
|
+
const lines = trimmed.split("\n");
|
|
1227
|
+
for (let i = 0; i < lines.length; i++) {
|
|
1228
|
+
const line = lines[i].trim();
|
|
1229
|
+
if (line.length === 0) continue;
|
|
1230
|
+
try {
|
|
1231
|
+
out.push({
|
|
1232
|
+
index: i,
|
|
1233
|
+
value: JSON.parse(line)
|
|
1234
|
+
});
|
|
1235
|
+
} catch (err) {
|
|
1236
|
+
throw new Error(`fromRunRecordDir: file '${file}' line ${i + 1} is not valid JSON: ${err instanceof Error ? err.message : String(err)}`);
|
|
1237
|
+
}
|
|
1238
|
+
}
|
|
1239
|
+
return out;
|
|
1240
|
+
}
|
|
1241
|
+
/** Sorted file list under a directory, filtered by `include`. Sorted so the
|
|
1242
|
+
* resulting `RunRecord` order — and any downstream fingerprint — is stable
|
|
1243
|
+
* across filesystems. */
|
|
1442
1244
|
async function collectFiles(dir, include, recursive) {
|
|
1443
|
-
|
|
1444
|
-
|
|
1445
|
-
|
|
1446
|
-
|
|
1447
|
-
|
|
1448
|
-
|
|
1449
|
-
|
|
1450
|
-
|
|
1451
|
-
|
|
1452
|
-
|
|
1453
|
-
|
|
1454
|
-
|
|
1455
|
-
|
|
1456
|
-
|
|
1457
|
-
|
|
1458
|
-
|
|
1459
|
-
|
|
1460
|
-
|
|
1461
|
-
|
|
1462
|
-
|
|
1245
|
+
const entries = await readdir(dir, { withFileTypes: true });
|
|
1246
|
+
const files = [];
|
|
1247
|
+
const subdirs = [];
|
|
1248
|
+
for (const entry of entries) {
|
|
1249
|
+
if (entry.isDirectory()) {
|
|
1250
|
+
if (recursive) subdirs.push(join(dir, entry.name));
|
|
1251
|
+
continue;
|
|
1252
|
+
}
|
|
1253
|
+
if (include(entry.name)) files.push(join(dir, entry.name));
|
|
1254
|
+
}
|
|
1255
|
+
files.sort();
|
|
1256
|
+
subdirs.sort();
|
|
1257
|
+
for (const sub of subdirs) files.push(...await collectFiles(sub, include, recursive));
|
|
1258
|
+
return files;
|
|
1259
|
+
}
|
|
1260
|
+
//#endregion
|
|
1261
|
+
//#region src/contract/eval-reporting-suite.ts
|
|
1262
|
+
/**
|
|
1263
|
+
* # `evalReportingSuite` — one call from runs (or a run dir) to `analysis.json`.
|
|
1264
|
+
*
|
|
1265
|
+
* A thin wrapper over the analysis primitive (`analyzeRuns`) and the on-disk
|
|
1266
|
+
* intake adapter (`fromRunRecordDir`). It does NOT reimplement any statistics,
|
|
1267
|
+
* distributions, or clustering — it resolves the input into validated
|
|
1268
|
+
* `RunRecord[]`, calls `analyzeRuns` with the options you'd pass it directly,
|
|
1269
|
+
* wraps the result in a small provenance envelope, and (optionally) writes a
|
|
1270
|
+
* single `analysis.json` artifact.
|
|
1271
|
+
*
|
|
1272
|
+
* ```ts
|
|
1273
|
+
* // From a directory of run files, write ./runs/analysis.json:
|
|
1274
|
+
* const suite = await evalReportingSuite('./runs', { write: true })
|
|
1275
|
+
* // From records already in memory, no write:
|
|
1276
|
+
* const suite = await evalReportingSuite(records, { analyze: { decisionThreshold: 0.03 } })
|
|
1277
|
+
* suite.report // the InsightReport — distributions, paired lift, findings rollup
|
|
1278
|
+
* ```
|
|
1279
|
+
*/
|
|
1280
|
+
const ANALYSIS_ARTIFACT = "analysis.json";
|
|
1281
|
+
/**
|
|
1282
|
+
* Resolve runs (or a run dir/file), run `analyzeRuns`, and optionally persist a
|
|
1283
|
+
* single `analysis.json`. The only analysis logic lives in `analyzeRuns`; this
|
|
1284
|
+
* function is composition + I/O.
|
|
1285
|
+
*/
|
|
1463
1286
|
async function evalReportingSuite(input, options = {}) {
|
|
1464
|
-
|
|
1465
|
-
|
|
1466
|
-
|
|
1467
|
-
|
|
1468
|
-
|
|
1469
|
-
|
|
1470
|
-
|
|
1471
|
-
|
|
1472
|
-
|
|
1473
|
-
|
|
1474
|
-
|
|
1475
|
-
|
|
1476
|
-
|
|
1477
|
-
|
|
1478
|
-
|
|
1479
|
-
|
|
1480
|
-
|
|
1481
|
-
|
|
1482
|
-
|
|
1483
|
-
|
|
1484
|
-
|
|
1485
|
-
|
|
1486
|
-
|
|
1487
|
-
|
|
1488
|
-
|
|
1489
|
-
|
|
1490
|
-
|
|
1491
|
-
|
|
1492
|
-
|
|
1493
|
-
|
|
1494
|
-
|
|
1495
|
-
|
|
1496
|
-
|
|
1497
|
-
|
|
1498
|
-
|
|
1499
|
-
|
|
1500
|
-
|
|
1501
|
-
}
|
|
1287
|
+
const fromPath = typeof input === "string";
|
|
1288
|
+
let runs;
|
|
1289
|
+
let files = [];
|
|
1290
|
+
let rejected = [];
|
|
1291
|
+
if (fromPath) {
|
|
1292
|
+
const loaded = await fromRunRecordDir(input, options.load);
|
|
1293
|
+
runs = loaded.runs;
|
|
1294
|
+
files = loaded.files;
|
|
1295
|
+
rejected = loaded.rejected;
|
|
1296
|
+
} else runs = input;
|
|
1297
|
+
if (runs.length === 0) throw new Error(fromPath ? `evalReportingSuite: no RunRecords found at '${input}'` : "evalReportingSuite: no RunRecords to analyze");
|
|
1298
|
+
const result = {
|
|
1299
|
+
report: await analyzeRuns({
|
|
1300
|
+
...options.analyze,
|
|
1301
|
+
runs
|
|
1302
|
+
}),
|
|
1303
|
+
provenance: {
|
|
1304
|
+
generatedAt: (/* @__PURE__ */ new Date()).toISOString(),
|
|
1305
|
+
runCount: runs.length,
|
|
1306
|
+
sourcePath: fromPath ? input : null,
|
|
1307
|
+
files,
|
|
1308
|
+
rejected
|
|
1309
|
+
},
|
|
1310
|
+
writtenTo: null
|
|
1311
|
+
};
|
|
1312
|
+
const target = resolveWriteTarget(options.write, fromPath ? input : null);
|
|
1313
|
+
if (target) {
|
|
1314
|
+
await mkdir(dirname(target), { recursive: true });
|
|
1315
|
+
await writeFile(target, `${JSON.stringify(result, null, 2)}\n`, "utf8");
|
|
1316
|
+
result.writtenTo = target;
|
|
1317
|
+
}
|
|
1318
|
+
return result;
|
|
1319
|
+
}
|
|
1320
|
+
/** Resolve where (if anywhere) to write `analysis.json`. Returns null when
|
|
1321
|
+
* writing is disabled. Throws on `write: true` with in-memory input — there is
|
|
1322
|
+
* no directory to anchor the artifact to, and silently inventing `cwd` would
|
|
1323
|
+
* scatter files. */
|
|
1502
1324
|
function resolveWriteTarget(write, sourcePath) {
|
|
1503
|
-
|
|
1504
|
-
|
|
1505
|
-
|
|
1506
|
-
|
|
1507
|
-
}
|
|
1508
|
-
if (sourcePath === null) {
|
|
1509
|
-
throw new Error(
|
|
1510
|
-
"evalReportingSuite: write:true needs a source path to anchor analysis.json \u2014 pass an explicit output path when analyzing in-memory records"
|
|
1511
|
-
);
|
|
1512
|
-
}
|
|
1513
|
-
const isFile = sourcePath.endsWith(".json") || sourcePath.endsWith(".jsonl");
|
|
1514
|
-
return isFile ? join2(dirname(sourcePath), ANALYSIS_ARTIFACT2) : join2(sourcePath, ANALYSIS_ARTIFACT2);
|
|
1325
|
+
if (!write) return null;
|
|
1326
|
+
if (typeof write === "string") return write.endsWith("/") || !write.endsWith(".json") && !write.endsWith(".jsonl") ? join(write, ANALYSIS_ARTIFACT) : write;
|
|
1327
|
+
if (sourcePath === null) throw new Error("evalReportingSuite: write:true needs a source path to anchor analysis.json — pass an explicit output path when analyzing in-memory records");
|
|
1328
|
+
return sourcePath.endsWith(".json") || sourcePath.endsWith(".jsonl") ? join(dirname(sourcePath), ANALYSIS_ARTIFACT) : join(sourcePath, ANALYSIS_ARTIFACT);
|
|
1515
1329
|
}
|
|
1516
|
-
|
|
1517
|
-
|
|
1330
|
+
//#endregion
|
|
1331
|
+
//#region src/contract/diff.ts
|
|
1518
1332
|
function keyForCell(cell) {
|
|
1519
|
-
|
|
1333
|
+
return JSON.stringify([cell.scenarioId, cell.rep]);
|
|
1520
1334
|
}
|
|
1335
|
+
/** Build the per-dimension delta map for a matched cell. Each judge name +
|
|
1336
|
+
* dimension name encountered on EITHER side appears in the result. */
|
|
1521
1337
|
function diffDimensions(before, after) {
|
|
1522
|
-
|
|
1523
|
-
|
|
1524
|
-
|
|
1525
|
-
|
|
1526
|
-
|
|
1527
|
-
|
|
1528
|
-
|
|
1529
|
-
|
|
1530
|
-
|
|
1531
|
-
|
|
1532
|
-
|
|
1533
|
-
|
|
1534
|
-
|
|
1535
|
-
|
|
1536
|
-
|
|
1537
|
-
|
|
1538
|
-
|
|
1539
|
-
|
|
1540
|
-
|
|
1541
|
-
|
|
1542
|
-
|
|
1543
|
-
}
|
|
1338
|
+
const out = {};
|
|
1339
|
+
const judges = /* @__PURE__ */ new Set([...Object.keys(before), ...Object.keys(after)]);
|
|
1340
|
+
for (const judge of judges) {
|
|
1341
|
+
const beforeDims = before[judge] ?? {};
|
|
1342
|
+
const afterDims = after[judge] ?? {};
|
|
1343
|
+
const dims = /* @__PURE__ */ new Set([...Object.keys(beforeDims), ...Object.keys(afterDims)]);
|
|
1344
|
+
const judgeOut = {};
|
|
1345
|
+
for (const dim of dims) {
|
|
1346
|
+
const rawBefore = beforeDims[dim];
|
|
1347
|
+
const rawAfter = afterDims[dim];
|
|
1348
|
+
const b = typeof rawBefore === "number" && Number.isFinite(rawBefore) ? rawBefore : null;
|
|
1349
|
+
const a = typeof rawAfter === "number" && Number.isFinite(rawAfter) ? rawAfter : null;
|
|
1350
|
+
judgeOut[dim] = {
|
|
1351
|
+
before: b,
|
|
1352
|
+
after: a,
|
|
1353
|
+
delta: b !== null && a !== null ? a - b : null
|
|
1354
|
+
};
|
|
1355
|
+
}
|
|
1356
|
+
out[judge] = judgeOut;
|
|
1357
|
+
}
|
|
1358
|
+
return out;
|
|
1359
|
+
}
|
|
1360
|
+
/**
|
|
1361
|
+
* Diff two generation snapshots. Cells are matched on `(scenarioId, rep)`;
|
|
1362
|
+
* unmatched cells surface in `added` / `removed`. Aggregate fields are
|
|
1363
|
+
* recomputed from the snapshot's stored fields, not re-derived from cells —
|
|
1364
|
+
* this keeps the diff consistent with whatever aggregation the substrate
|
|
1365
|
+
* actually reported.
|
|
1366
|
+
*/
|
|
1544
1367
|
function diffGenerations(before, after) {
|
|
1545
|
-
|
|
1546
|
-
|
|
1547
|
-
|
|
1548
|
-
|
|
1549
|
-
|
|
1550
|
-
|
|
1551
|
-
|
|
1552
|
-
|
|
1553
|
-
|
|
1554
|
-
|
|
1555
|
-
|
|
1556
|
-
|
|
1557
|
-
|
|
1558
|
-
|
|
1559
|
-
|
|
1560
|
-
|
|
1561
|
-
|
|
1562
|
-
|
|
1563
|
-
|
|
1564
|
-
|
|
1565
|
-
|
|
1566
|
-
|
|
1567
|
-
|
|
1568
|
-
|
|
1569
|
-
|
|
1570
|
-
|
|
1571
|
-
|
|
1572
|
-
|
|
1573
|
-
|
|
1574
|
-
|
|
1575
|
-
|
|
1576
|
-
|
|
1577
|
-
|
|
1578
|
-
|
|
1579
|
-
|
|
1580
|
-
|
|
1581
|
-
|
|
1582
|
-
|
|
1583
|
-
|
|
1584
|
-
|
|
1585
|
-
|
|
1586
|
-
|
|
1587
|
-
}
|
|
1368
|
+
const beforeMap = new Map(before.cells.map((c) => [keyForCell(c), c]));
|
|
1369
|
+
const afterMap = new Map(after.cells.map((c) => [keyForCell(c), c]));
|
|
1370
|
+
const matched = [];
|
|
1371
|
+
const removed = [];
|
|
1372
|
+
const added = [];
|
|
1373
|
+
for (const [key, beforeCell] of beforeMap) {
|
|
1374
|
+
const afterCell = afterMap.get(key);
|
|
1375
|
+
if (!afterCell) {
|
|
1376
|
+
removed.push(beforeCell);
|
|
1377
|
+
continue;
|
|
1378
|
+
}
|
|
1379
|
+
matched.push({
|
|
1380
|
+
scenarioId: beforeCell.scenarioId,
|
|
1381
|
+
rep: beforeCell.rep,
|
|
1382
|
+
compositeBefore: beforeCell.compositeMean,
|
|
1383
|
+
compositeAfter: afterCell.compositeMean,
|
|
1384
|
+
compositeDelta: beforeCell.compositeMean === null || afterCell.compositeMean === null ? null : afterCell.compositeMean - beforeCell.compositeMean,
|
|
1385
|
+
dimensions: diffDimensions(beforeCell.dimensions, afterCell.dimensions)
|
|
1386
|
+
});
|
|
1387
|
+
}
|
|
1388
|
+
for (const [key, afterCell] of afterMap) if (!beforeMap.has(key)) added.push(afterCell);
|
|
1389
|
+
return {
|
|
1390
|
+
beforeIndex: before.index,
|
|
1391
|
+
afterIndex: after.index,
|
|
1392
|
+
beforeSurfaceHash: before.surfaceHash,
|
|
1393
|
+
afterSurfaceHash: after.surfaceHash,
|
|
1394
|
+
surfaceChanged: before.surfaceHash !== after.surfaceHash,
|
|
1395
|
+
matched,
|
|
1396
|
+
removed,
|
|
1397
|
+
added,
|
|
1398
|
+
compositeBefore: before.compositeMean,
|
|
1399
|
+
compositeAfter: after.compositeMean,
|
|
1400
|
+
compositeDelta: before.compositeMean === null || after.compositeMean === null ? null : after.compositeMean - before.compositeMean,
|
|
1401
|
+
costUsdBefore: before.costUsd,
|
|
1402
|
+
costUsdAfter: after.costUsd,
|
|
1403
|
+
costUsdDelta: after.costUsd - before.costUsd,
|
|
1404
|
+
durationMsBefore: before.durationMs,
|
|
1405
|
+
durationMsAfter: after.durationMs,
|
|
1406
|
+
durationMsDelta: after.durationMs - before.durationMs
|
|
1407
|
+
};
|
|
1408
|
+
}
|
|
1409
|
+
/** Highest-index generation, or null if the run recorded none. */
|
|
1588
1410
|
function winnerOf(run) {
|
|
1589
|
-
|
|
1590
|
-
|
|
1591
|
-
|
|
1592
|
-
|
|
1593
|
-
|
|
1594
|
-
|
|
1595
|
-
|
|
1411
|
+
if (run.generations.length === 0) return null;
|
|
1412
|
+
let winner = run.generations[0];
|
|
1413
|
+
for (const gen of run.generations) if (gen.index > winner.index) winner = gen;
|
|
1414
|
+
return winner;
|
|
1415
|
+
}
|
|
1416
|
+
/**
|
|
1417
|
+
* Diff two full eval-runs. Produces baseline-vs-baseline and
|
|
1418
|
+
* winner-vs-winner generation diffs when both sides expose them, plus
|
|
1419
|
+
* run-level cost / lift / gate-decision deltas.
|
|
1420
|
+
*/
|
|
1596
1421
|
function diffRuns(before, after) {
|
|
1597
|
-
|
|
1598
|
-
|
|
1599
|
-
|
|
1600
|
-
|
|
1601
|
-
|
|
1602
|
-
|
|
1603
|
-
|
|
1604
|
-
|
|
1605
|
-
|
|
1606
|
-
|
|
1607
|
-
|
|
1608
|
-
|
|
1609
|
-
|
|
1610
|
-
|
|
1611
|
-
|
|
1612
|
-
|
|
1613
|
-
|
|
1614
|
-
|
|
1615
|
-
|
|
1616
|
-
|
|
1617
|
-
|
|
1618
|
-
|
|
1619
|
-
|
|
1620
|
-
|
|
1621
|
-
|
|
1622
|
-
}
|
|
1422
|
+
const beforeWinner = winnerOf(before);
|
|
1423
|
+
const afterWinner = winnerOf(after);
|
|
1424
|
+
const baselineDiff = before.baseline && after.baseline ? diffGenerations(before.baseline, after.baseline) : null;
|
|
1425
|
+
const winnersDiff = beforeWinner && afterWinner ? diffGenerations(beforeWinner, afterWinner) : null;
|
|
1426
|
+
const beforeLift = before.holdoutLift ?? null;
|
|
1427
|
+
const afterLift = after.holdoutLift ?? null;
|
|
1428
|
+
return {
|
|
1429
|
+
beforeRunId: before.runId,
|
|
1430
|
+
afterRunId: after.runId,
|
|
1431
|
+
beforeTimestamp: before.timestamp,
|
|
1432
|
+
afterTimestamp: after.timestamp,
|
|
1433
|
+
beforeGateDecision: before.gateDecision ?? null,
|
|
1434
|
+
afterGateDecision: after.gateDecision ?? null,
|
|
1435
|
+
beforeHoldoutLift: beforeLift,
|
|
1436
|
+
afterHoldoutLift: afterLift,
|
|
1437
|
+
holdoutLiftDelta: beforeLift !== null && afterLift !== null ? afterLift - beforeLift : null,
|
|
1438
|
+
beforeTotalCostUsd: before.totalCostUsd,
|
|
1439
|
+
afterTotalCostUsd: after.totalCostUsd,
|
|
1440
|
+
totalCostUsdDelta: after.totalCostUsd - before.totalCostUsd,
|
|
1441
|
+
beforeTotalDurationMs: before.totalDurationMs,
|
|
1442
|
+
afterTotalDurationMs: after.totalDurationMs,
|
|
1443
|
+
totalDurationMsDelta: after.totalDurationMs - before.totalDurationMs,
|
|
1444
|
+
baselineDiff,
|
|
1445
|
+
winnersDiff
|
|
1446
|
+
};
|
|
1447
|
+
}
|
|
1448
|
+
/**
|
|
1449
|
+
* Within-run baseline → winning-generation diff. The natural "what did the
|
|
1450
|
+
* improvement loop produce" view for a single run. Returns null when the
|
|
1451
|
+
* run never reached a generation past baseline (errored early, or the gate
|
|
1452
|
+
* shipped the baseline as-is).
|
|
1453
|
+
*/
|
|
1623
1454
|
function diffRunBaselineToWinner(run) {
|
|
1624
|
-
|
|
1625
|
-
|
|
1626
|
-
|
|
1627
|
-
|
|
1455
|
+
if (!run.baseline) return null;
|
|
1456
|
+
const winner = winnerOf(run);
|
|
1457
|
+
if (!winner || winner.index === run.baseline.index) return null;
|
|
1458
|
+
return diffGenerations(run.baseline, winner);
|
|
1628
1459
|
}
|
|
1629
|
-
|
|
1630
|
-
|
|
1460
|
+
//#endregion
|
|
1461
|
+
//#region src/contract/intake/agent-trace.ts
|
|
1631
1462
|
function rangeLines(r) {
|
|
1632
|
-
|
|
1463
|
+
return Math.max(0, r.end_line - r.start_line + 1);
|
|
1633
1464
|
}
|
|
1465
|
+
/**
|
|
1466
|
+
* Build a commit → provenance index from Agent Trace records. Multiple records
|
|
1467
|
+
* for the same revision are merged. Records without `vcs.revision` are skipped
|
|
1468
|
+
* (the SHA is the join key — without it there is nothing to correlate against).
|
|
1469
|
+
*/
|
|
1634
1470
|
function parseAgentTrace(records) {
|
|
1635
|
-
|
|
1636
|
-
|
|
1637
|
-
|
|
1638
|
-
|
|
1639
|
-
|
|
1640
|
-
|
|
1641
|
-
|
|
1642
|
-
|
|
1643
|
-
|
|
1644
|
-
|
|
1645
|
-
|
|
1646
|
-
|
|
1647
|
-
|
|
1648
|
-
|
|
1649
|
-
|
|
1650
|
-
|
|
1651
|
-
|
|
1652
|
-
|
|
1653
|
-
|
|
1654
|
-
|
|
1655
|
-
|
|
1656
|
-
|
|
1657
|
-
|
|
1658
|
-
|
|
1659
|
-
|
|
1660
|
-
|
|
1661
|
-
|
|
1662
|
-
|
|
1663
|
-
|
|
1664
|
-
|
|
1665
|
-
|
|
1666
|
-
|
|
1667
|
-
|
|
1668
|
-
|
|
1669
|
-
|
|
1670
|
-
|
|
1671
|
-
|
|
1672
|
-
|
|
1673
|
-
|
|
1674
|
-
|
|
1675
|
-
|
|
1676
|
-
|
|
1677
|
-
|
|
1678
|
-
|
|
1679
|
-
|
|
1680
|
-
|
|
1681
|
-
|
|
1682
|
-
|
|
1683
|
-
|
|
1471
|
+
const acc = /* @__PURE__ */ new Map();
|
|
1472
|
+
for (const record of records) {
|
|
1473
|
+
const sha = record.vcs?.revision;
|
|
1474
|
+
if (!sha) continue;
|
|
1475
|
+
let a = acc.get(sha);
|
|
1476
|
+
if (!a) {
|
|
1477
|
+
a = {
|
|
1478
|
+
models: /* @__PURE__ */ new Set(),
|
|
1479
|
+
tools: /* @__PURE__ */ new Set(),
|
|
1480
|
+
files: /* @__PURE__ */ new Set(),
|
|
1481
|
+
conversationCount: 0,
|
|
1482
|
+
lineCount: 0,
|
|
1483
|
+
humanInvolved: false
|
|
1484
|
+
};
|
|
1485
|
+
acc.set(sha, a);
|
|
1486
|
+
}
|
|
1487
|
+
if (record.tool?.name) a.tools.add(record.tool.name);
|
|
1488
|
+
for (const file of record.files ?? []) {
|
|
1489
|
+
a.files.add(file.path);
|
|
1490
|
+
for (const conv of file.conversations ?? []) {
|
|
1491
|
+
a.conversationCount += 1;
|
|
1492
|
+
for (const range of conv.ranges ?? []) {
|
|
1493
|
+
const contributor = range.contributor ?? conv.contributor;
|
|
1494
|
+
a.lineCount += rangeLines(range);
|
|
1495
|
+
if (!contributor) continue;
|
|
1496
|
+
if (contributor.type === "human" || contributor.type === "mixed") a.humanInvolved = true;
|
|
1497
|
+
if ((contributor.type === "ai" || contributor.type === "mixed") && contributor.model_id) a.models.add(contributor.model_id);
|
|
1498
|
+
}
|
|
1499
|
+
}
|
|
1500
|
+
}
|
|
1501
|
+
}
|
|
1502
|
+
const index = /* @__PURE__ */ new Map();
|
|
1503
|
+
for (const [sha, a] of acc) index.set(sha, {
|
|
1504
|
+
commitSha: sha,
|
|
1505
|
+
aiModels: [...a.models].sort(),
|
|
1506
|
+
tools: [...a.tools].sort(),
|
|
1507
|
+
conversationCount: a.conversationCount,
|
|
1508
|
+
fileCount: a.files.size,
|
|
1509
|
+
lineCount: a.lineCount,
|
|
1510
|
+
humanInvolved: a.humanInvolved
|
|
1511
|
+
});
|
|
1512
|
+
return index;
|
|
1513
|
+
}
|
|
1514
|
+
/**
|
|
1515
|
+
* Partition runs by the AI model(s) that authored the code at each run's
|
|
1516
|
+
* `commitSha`. Feed `byModel.get(modelId)` to `analyzeRuns`, or compare two
|
|
1517
|
+
* model cohorts via `analyzeRuns({ runs: a, baselineRuns: b })` for a lift CI
|
|
1518
|
+
* on "model A's code vs model B's code".
|
|
1519
|
+
*/
|
|
1684
1520
|
function partitionRunsByAuthoringModel(runs, index) {
|
|
1685
|
-
|
|
1686
|
-
|
|
1687
|
-
|
|
1688
|
-
|
|
1689
|
-
|
|
1690
|
-
|
|
1691
|
-
|
|
1692
|
-
|
|
1693
|
-
|
|
1694
|
-
|
|
1695
|
-
|
|
1696
|
-
|
|
1697
|
-
|
|
1698
|
-
|
|
1699
|
-
|
|
1700
|
-
|
|
1701
|
-
|
|
1702
|
-
|
|
1521
|
+
const byModel = /* @__PURE__ */ new Map();
|
|
1522
|
+
const unattributed = [];
|
|
1523
|
+
for (const run of runs) {
|
|
1524
|
+
const provenance = index.get(run.commitSha);
|
|
1525
|
+
if (!provenance || provenance.aiModels.length === 0) {
|
|
1526
|
+
unattributed.push(run);
|
|
1527
|
+
continue;
|
|
1528
|
+
}
|
|
1529
|
+
for (const model of provenance.aiModels) {
|
|
1530
|
+
const cohort = byModel.get(model) ?? [];
|
|
1531
|
+
cohort.push(run);
|
|
1532
|
+
byModel.set(model, cohort);
|
|
1533
|
+
}
|
|
1534
|
+
}
|
|
1535
|
+
return {
|
|
1536
|
+
byModel,
|
|
1537
|
+
unattributed
|
|
1538
|
+
};
|
|
1539
|
+
}
|
|
1540
|
+
//#endregion
|
|
1541
|
+
//#region src/contract/intake/feedback-table.ts
|
|
1703
1542
|
function fromFeedbackTable(opts) {
|
|
1704
|
-
|
|
1705
|
-
|
|
1706
|
-
|
|
1707
|
-
|
|
1708
|
-
|
|
1709
|
-
|
|
1710
|
-
|
|
1711
|
-
|
|
1712
|
-
|
|
1713
|
-
|
|
1714
|
-
|
|
1715
|
-
|
|
1716
|
-
|
|
1717
|
-
|
|
1718
|
-
|
|
1719
|
-
|
|
1720
|
-
|
|
1721
|
-
|
|
1722
|
-
|
|
1723
|
-
|
|
1724
|
-
|
|
1725
|
-
|
|
1726
|
-
|
|
1727
|
-
|
|
1728
|
-
|
|
1729
|
-
|
|
1730
|
-
|
|
1731
|
-
|
|
1732
|
-
|
|
1733
|
-
|
|
1734
|
-
|
|
1735
|
-
|
|
1736
|
-
|
|
1737
|
-
|
|
1738
|
-
|
|
1739
|
-
|
|
1740
|
-
|
|
1741
|
-
|
|
1742
|
-
|
|
1743
|
-
|
|
1744
|
-
|
|
1745
|
-
|
|
1746
|
-
|
|
1747
|
-
|
|
1748
|
-
|
|
1749
|
-
|
|
1750
|
-
|
|
1751
|
-
|
|
1752
|
-
|
|
1753
|
-
|
|
1754
|
-
|
|
1755
|
-
|
|
1756
|
-
|
|
1757
|
-
|
|
1758
|
-
|
|
1759
|
-
|
|
1760
|
-
|
|
1761
|
-
|
|
1762
|
-
|
|
1763
|
-
|
|
1764
|
-
|
|
1765
|
-
|
|
1766
|
-
|
|
1767
|
-
|
|
1768
|
-
|
|
1769
|
-
|
|
1543
|
+
const { ratings, meta = [], scale, emitRaterScores = true } = opts;
|
|
1544
|
+
const metaByRun = new Map(meta.map((m) => [m.runId, m]));
|
|
1545
|
+
const normalise = (rating) => {
|
|
1546
|
+
if (typeof rating === "boolean") return rating ? 1 : 0;
|
|
1547
|
+
if (!Number.isFinite(rating)) return NaN;
|
|
1548
|
+
if (!scale) return rating;
|
|
1549
|
+
const { min, max } = scale;
|
|
1550
|
+
if (max === min) return rating;
|
|
1551
|
+
return (rating - min) / (max - min);
|
|
1552
|
+
};
|
|
1553
|
+
const byRun = /* @__PURE__ */ new Map();
|
|
1554
|
+
for (const row of ratings) {
|
|
1555
|
+
const list = byRun.get(row.runId) ?? [];
|
|
1556
|
+
list.push(row);
|
|
1557
|
+
byRun.set(row.runId, list);
|
|
1558
|
+
}
|
|
1559
|
+
const runs = [];
|
|
1560
|
+
const raterScores = [];
|
|
1561
|
+
for (const [runId, rowsForRun] of byRun) {
|
|
1562
|
+
const normalised = rowsForRun.map((r) => ({
|
|
1563
|
+
rater: r.rater,
|
|
1564
|
+
score: normalise(r.rating)
|
|
1565
|
+
})).filter((r) => Number.isFinite(r.score));
|
|
1566
|
+
if (normalised.length === 0) continue;
|
|
1567
|
+
const meanScore = normalised.reduce((s, r) => s + r.score, 0) / normalised.length;
|
|
1568
|
+
const runMeta = metaByRun.get(runId) ?? { runId };
|
|
1569
|
+
const judgeScores = {
|
|
1570
|
+
perJudge: Object.fromEntries(normalised.map((r) => [r.rater, { rating: r.score }])),
|
|
1571
|
+
perDimMean: { rating: meanScore },
|
|
1572
|
+
composite: meanScore
|
|
1573
|
+
};
|
|
1574
|
+
const splitTag = runMeta.splitTag ?? "holdout";
|
|
1575
|
+
const outcome = {
|
|
1576
|
+
...splitTag === "holdout" ? { holdoutScore: meanScore } : { searchScore: meanScore },
|
|
1577
|
+
raw: Object.fromEntries(normalised.map((r) => [`rater:${r.rater}`, r.score])),
|
|
1578
|
+
judgeScores
|
|
1579
|
+
};
|
|
1580
|
+
const costUsd = runMeta.costUsd ?? null;
|
|
1581
|
+
runs.push({
|
|
1582
|
+
runId,
|
|
1583
|
+
experimentId: runMeta.experimentId ?? "feedback-corpus",
|
|
1584
|
+
candidateId: runMeta.candidateId ?? runId,
|
|
1585
|
+
seed: 0,
|
|
1586
|
+
model: runMeta.model ?? "unknown@unknown",
|
|
1587
|
+
promptHash: runMeta.promptHash ?? "sha256:unknown",
|
|
1588
|
+
configHash: runMeta.configHash ?? "sha256:unknown",
|
|
1589
|
+
commitSha: runMeta.commitSha ?? "unknown",
|
|
1590
|
+
wallMs: runMeta.wallMs ?? 0,
|
|
1591
|
+
costUsd,
|
|
1592
|
+
costProvenance: costUsd === null ? {
|
|
1593
|
+
kind: "uncaptured",
|
|
1594
|
+
usd: null
|
|
1595
|
+
} : {
|
|
1596
|
+
kind: "observed",
|
|
1597
|
+
usd: costUsd
|
|
1598
|
+
},
|
|
1599
|
+
tokenUsage: {
|
|
1600
|
+
input: 0,
|
|
1601
|
+
output: 0
|
|
1602
|
+
},
|
|
1603
|
+
terminalOutcome: "unknown",
|
|
1604
|
+
outcome,
|
|
1605
|
+
splitTag,
|
|
1606
|
+
scenarioId: runMeta.scenarioId ?? runId
|
|
1607
|
+
});
|
|
1608
|
+
if (emitRaterScores) for (const r of normalised) raterScores.push({
|
|
1609
|
+
runId,
|
|
1610
|
+
rater: r.rater,
|
|
1611
|
+
score: r.score
|
|
1612
|
+
});
|
|
1613
|
+
}
|
|
1614
|
+
return {
|
|
1615
|
+
runs,
|
|
1616
|
+
raterScores
|
|
1617
|
+
};
|
|
1618
|
+
}
|
|
1619
|
+
//#endregion
|
|
1620
|
+
//#region src/contract/intake/otel-spans.ts
|
|
1621
|
+
/**
|
|
1622
|
+
* # `intake/otel-spans` — OTel `TraceSpanEvent[]` → `RunRecord[]`.
|
|
1623
|
+
*
|
|
1624
|
+
* Turns an existing observability stream into the substrate-canonical
|
|
1625
|
+
* `RunRecord` shape so consumers with logs but no eval discipline can
|
|
1626
|
+
* call `analyzeRuns()` against their production traffic immediately.
|
|
1627
|
+
*
|
|
1628
|
+
* Pivot rule: spans are grouped by `tangle.runId` (the same attribute the
|
|
1629
|
+
* hosted-tier wire format uses) or, when absent, by `traceId`. One group
|
|
1630
|
+
* becomes one `RunRecord`. The root span (no `parentSpanId`) supplies:
|
|
1631
|
+
*
|
|
1632
|
+
* - `runId` (the group key)
|
|
1633
|
+
* - `wallMs` from `endTimeUnixNano - startTimeUnixNano`
|
|
1634
|
+
* - `model` from `gen_ai.request.model` / `llm.model` / `tangle.model`
|
|
1635
|
+
* - task failure class and detail from explicit `tangle.task.*` attributes
|
|
1636
|
+
* - cost from `cost.usd` / `gen_ai.usage.cost_usd` / `tangle.cost.usd`
|
|
1637
|
+
* - token usage from model-call input, output, cache-read, and cache-write
|
|
1638
|
+
* attributes without double-counting aggregate parent spans
|
|
1639
|
+
* - task quality from an explicit `scoreForRun` callback or a designated
|
|
1640
|
+
* evaluation attribute on a root / `EVALUATOR` span; `outcome.raw`
|
|
1641
|
+
* collects every numeric attribute without promoting it to task quality.
|
|
1642
|
+
*
|
|
1643
|
+
* Errored tool, model, and child-agent spans contribute to execution-error
|
|
1644
|
+
* counts. Root process, guardrail, evaluator, propagated parent, and unknown
|
|
1645
|
+
* errors retain separate counters. Only one failed root can set
|
|
1646
|
+
* `RunRecord.terminalOutcome` and `RunRecord.terminalFailureReason`; a child
|
|
1647
|
+
* error cannot become a task failure.
|
|
1648
|
+
*/
|
|
1649
|
+
const TASK_SCORE_ATTR_KEYS = [
|
|
1650
|
+
"gen_ai.evaluation.score.value",
|
|
1651
|
+
"tangle.task.score",
|
|
1652
|
+
"eval.score",
|
|
1653
|
+
"tangle.score"
|
|
1770
1654
|
];
|
|
1771
|
-
|
|
1772
|
-
|
|
1773
|
-
|
|
1655
|
+
const MODEL_KEYS = [
|
|
1656
|
+
"tangle.model",
|
|
1657
|
+
...LLM_MODEL_ATTR_KEYS,
|
|
1658
|
+
"model"
|
|
1659
|
+
];
|
|
1660
|
+
const PROMPT_HASH_KEYS = ["tangle.prompt_hash", "prompt.hash"];
|
|
1661
|
+
const CONFIG_HASH_KEYS = ["tangle.config_hash", "config.hash"];
|
|
1774
1662
|
function fromOtelSpans(opts) {
|
|
1775
|
-
|
|
1776
|
-
|
|
1777
|
-
|
|
1778
|
-
|
|
1779
|
-
|
|
1780
|
-
|
|
1781
|
-
|
|
1782
|
-
|
|
1783
|
-
|
|
1784
|
-
|
|
1785
|
-
|
|
1786
|
-
|
|
1787
|
-
|
|
1788
|
-
|
|
1789
|
-
|
|
1790
|
-
|
|
1791
|
-
|
|
1792
|
-
|
|
1793
|
-
|
|
1794
|
-
|
|
1795
|
-
|
|
1796
|
-
|
|
1797
|
-
|
|
1798
|
-
|
|
1799
|
-
|
|
1800
|
-
|
|
1801
|
-
|
|
1802
|
-
|
|
1803
|
-
|
|
1804
|
-
|
|
1805
|
-
|
|
1806
|
-
|
|
1807
|
-
|
|
1808
|
-
|
|
1809
|
-
|
|
1810
|
-
|
|
1811
|
-
|
|
1812
|
-
|
|
1813
|
-
|
|
1814
|
-
|
|
1815
|
-
|
|
1816
|
-
|
|
1817
|
-
|
|
1818
|
-
|
|
1819
|
-
|
|
1820
|
-
|
|
1821
|
-
|
|
1822
|
-
|
|
1823
|
-
|
|
1824
|
-
|
|
1825
|
-
|
|
1826
|
-
|
|
1827
|
-
|
|
1828
|
-
|
|
1829
|
-
|
|
1830
|
-
|
|
1831
|
-
|
|
1832
|
-
|
|
1833
|
-
|
|
1834
|
-
|
|
1835
|
-
|
|
1836
|
-
|
|
1837
|
-
|
|
1838
|
-
|
|
1839
|
-
|
|
1840
|
-
|
|
1841
|
-
|
|
1842
|
-
|
|
1843
|
-
|
|
1844
|
-
|
|
1845
|
-
|
|
1846
|
-
|
|
1847
|
-
|
|
1848
|
-
|
|
1849
|
-
|
|
1850
|
-
|
|
1851
|
-
|
|
1852
|
-
|
|
1853
|
-
|
|
1854
|
-
|
|
1855
|
-
|
|
1856
|
-
|
|
1857
|
-
|
|
1858
|
-
outcome,
|
|
1859
|
-
...taskFailure,
|
|
1860
|
-
splitTag: defaultSplit,
|
|
1861
|
-
scenarioId
|
|
1862
|
-
});
|
|
1863
|
-
}
|
|
1864
|
-
return runs;
|
|
1663
|
+
const { spans, defaultSplit = "holdout", experimentId = "otel-corpus" } = opts;
|
|
1664
|
+
const grouped = groupSpans(spans);
|
|
1665
|
+
const runs = [];
|
|
1666
|
+
for (const [groupKey, groupSpans] of grouped) {
|
|
1667
|
+
const root = findRoot(groupSpans);
|
|
1668
|
+
if (!root) continue;
|
|
1669
|
+
const measurements = summarizeExecutionMeasurements(groupSpans.map((span) => ({
|
|
1670
|
+
id: span.spanId,
|
|
1671
|
+
...span.parentSpanId ? { parentId: span.parentSpanId } : {},
|
|
1672
|
+
attributes: span.attributes,
|
|
1673
|
+
modelCall: isExplicitModelCall(span),
|
|
1674
|
+
aggregate: isExplicitAggregate(span)
|
|
1675
|
+
})));
|
|
1676
|
+
const callSpanIds = new Set(measurements.callSpanIds);
|
|
1677
|
+
const callSpans = groupSpans.filter((span) => callSpanIds.has(span.spanId));
|
|
1678
|
+
const wallMs = unixNanoDurationMs(root.startTimeUnixNano, root.endTimeUnixNano);
|
|
1679
|
+
const model = readAttrString(callSpans, MODEL_KEYS) ?? readAttrString(groupSpans, MODEL_KEYS) ?? "unknown@unknown";
|
|
1680
|
+
const capturedCost = (measurements.cost.complete ? measurements.cost.value : void 0) ?? measurements.aggregate?.costUsd;
|
|
1681
|
+
const costUsd = capturedCost ?? null;
|
|
1682
|
+
const scenarioId = readConsistentScenarioId(groupKey, groupSpans) ?? groupKey;
|
|
1683
|
+
const promptHash = readAttrString(groupSpans, PROMPT_HASH_KEYS) ?? "sha256:unknown";
|
|
1684
|
+
const configHash = readAttrString(groupSpans, CONFIG_HASH_KEYS) ?? "sha256:unknown";
|
|
1685
|
+
const score = resolveTaskScore(groupKey, groupSpans, opts.scoreForRun);
|
|
1686
|
+
const taskFailure = readTaskFailureLabels(groupSpans.filter((span) => !span.parentSpanId && isTerminalRootCandidate(span)), `fromOtelSpans: run '${groupKey}'`);
|
|
1687
|
+
const rawNumeric = collectNumericAttrs(groupSpans);
|
|
1688
|
+
const errorSummary = summarizeTraceErrors(groupSpans.map((span) => ({
|
|
1689
|
+
id: spanIdentity(span),
|
|
1690
|
+
...span.parentSpanId ? { parentId: parentIdentity(span) } : {},
|
|
1691
|
+
role: errorRoleForSpan(span),
|
|
1692
|
+
error: span.status?.code === "ERROR",
|
|
1693
|
+
processRoot: !span.parentSpanId && isTerminalRootCandidate(span)
|
|
1694
|
+
})));
|
|
1695
|
+
rawNumeric.error_span_count = errorSummary.total;
|
|
1696
|
+
rawNumeric.execution_error_count = errorSummary.execution;
|
|
1697
|
+
rawNumeric.process_error_count = errorSummary.process;
|
|
1698
|
+
rawNumeric.guardrail_error_count = errorSummary.guardrail;
|
|
1699
|
+
rawNumeric.judge_error_count = errorSummary.evaluation;
|
|
1700
|
+
rawNumeric.propagated_error_count = errorSummary.propagated;
|
|
1701
|
+
rawNumeric.unclassified_error_count = errorSummary.unclassified;
|
|
1702
|
+
rawNumeric.llm_span_count = measurements.modelCallCount;
|
|
1703
|
+
if (measurements.cost.value !== void 0 && !measurements.cost.complete) rawNumeric.partial_observed_cost_usd = measurements.cost.value;
|
|
1704
|
+
recordAggregateMeasurements(rawNumeric, measurements.aggregate);
|
|
1705
|
+
const judgeScores = score !== void 0 ? {
|
|
1706
|
+
perJudge: { "otel-derived": { score } },
|
|
1707
|
+
perDimMean: { score },
|
|
1708
|
+
composite: score
|
|
1709
|
+
} : void 0;
|
|
1710
|
+
const terminalOutcome = terminalOutcomeFromRoots(groupSpans);
|
|
1711
|
+
const failedRoot = terminalOutcome === "failed" ? groupSpans.find((span) => !span.parentSpanId && isTerminalRootCandidate(span) && span.status?.code === "ERROR") : void 0;
|
|
1712
|
+
const outcome = {
|
|
1713
|
+
raw: rawNumeric,
|
|
1714
|
+
...judgeScores ? { judgeScores } : {}
|
|
1715
|
+
};
|
|
1716
|
+
if (score !== void 0) if (defaultSplit === "holdout") outcome.holdoutScore = score;
|
|
1717
|
+
else outcome.searchScore = score;
|
|
1718
|
+
runs.push({
|
|
1719
|
+
runId: groupKey,
|
|
1720
|
+
experimentId,
|
|
1721
|
+
candidateId: root.attributes["tangle.candidateId"] ?? "otel-default",
|
|
1722
|
+
seed: 0,
|
|
1723
|
+
model,
|
|
1724
|
+
promptHash,
|
|
1725
|
+
configHash,
|
|
1726
|
+
commitSha: root.attributes["tangle.commit_sha"] ?? "unknown",
|
|
1727
|
+
wallMs,
|
|
1728
|
+
costUsd,
|
|
1729
|
+
costProvenance: capturedCost === void 0 ? {
|
|
1730
|
+
kind: "uncaptured",
|
|
1731
|
+
usd: null
|
|
1732
|
+
} : {
|
|
1733
|
+
kind: "observed",
|
|
1734
|
+
usd: capturedCost
|
|
1735
|
+
},
|
|
1736
|
+
tokenUsage: measurements.tokenUsage,
|
|
1737
|
+
terminalOutcome,
|
|
1738
|
+
...failedRoot ? { terminalFailureReason: failedRoot.status?.message ?? failedRoot.name } : {},
|
|
1739
|
+
outcome,
|
|
1740
|
+
...taskFailure,
|
|
1741
|
+
splitTag: defaultSplit,
|
|
1742
|
+
scenarioId
|
|
1743
|
+
});
|
|
1744
|
+
}
|
|
1745
|
+
return runs;
|
|
1865
1746
|
}
|
|
1866
1747
|
function terminalOutcomeFromRoots(spans) {
|
|
1867
|
-
|
|
1868
|
-
|
|
1869
|
-
|
|
1870
|
-
|
|
1871
|
-
|
|
1748
|
+
const roots = spans.filter((span) => !span.parentSpanId && isTerminalRootCandidate(span));
|
|
1749
|
+
if (roots.length !== 1) return "unknown";
|
|
1750
|
+
if (roots[0].status?.code === "OK") return "succeeded";
|
|
1751
|
+
if (roots[0].status?.code === "ERROR") return "failed";
|
|
1752
|
+
return "unknown";
|
|
1872
1753
|
}
|
|
1873
1754
|
function isTerminalRootCandidate(span) {
|
|
1874
|
-
|
|
1875
|
-
|
|
1876
|
-
|
|
1877
|
-
}
|
|
1878
|
-
return true;
|
|
1755
|
+
const role = errorRoleForSpan(span);
|
|
1756
|
+
if (role === "LLM" || role === "TOOL" || role === "GUARDRAIL" || role === "EVALUATOR") return false;
|
|
1757
|
+
return true;
|
|
1879
1758
|
}
|
|
1880
1759
|
function readSpanKind(span) {
|
|
1881
|
-
|
|
1760
|
+
return readAttrString([span], [...SPAN_KIND_ATTR_KEYS, "span.kind"])?.toUpperCase();
|
|
1882
1761
|
}
|
|
1883
1762
|
function errorRoleForSpan(span) {
|
|
1884
|
-
|
|
1885
|
-
|
|
1886
|
-
|
|
1887
|
-
|
|
1888
|
-
|
|
1763
|
+
return classifyOtlpSpanRole({
|
|
1764
|
+
kind: readSpanKind(span),
|
|
1765
|
+
name: span.name,
|
|
1766
|
+
attributes: span.attributes
|
|
1767
|
+
});
|
|
1889
1768
|
}
|
|
1890
1769
|
function spanIdentity(span) {
|
|
1891
|
-
|
|
1770
|
+
return `${span.traceId}:${span.spanId}`;
|
|
1892
1771
|
}
|
|
1893
1772
|
function parentIdentity(span) {
|
|
1894
|
-
|
|
1773
|
+
return `${span.traceId}:${span.parentSpanId}`;
|
|
1895
1774
|
}
|
|
1896
1775
|
function isExplicitModelCall(span) {
|
|
1897
|
-
|
|
1898
|
-
|
|
1899
|
-
|
|
1900
|
-
|
|
1901
|
-
|
|
1776
|
+
return isOtlpModelCall({
|
|
1777
|
+
kind: readSpanKind(span),
|
|
1778
|
+
name: span.name,
|
|
1779
|
+
attributes: span.attributes
|
|
1780
|
+
});
|
|
1902
1781
|
}
|
|
1903
1782
|
function isExplicitAggregate(span) {
|
|
1904
|
-
|
|
1905
|
-
|
|
1783
|
+
const kind = readSpanKind(span);
|
|
1784
|
+
return kind !== void 0 && kind !== "LLM";
|
|
1906
1785
|
}
|
|
1907
1786
|
function groupSpans(spans) {
|
|
1908
|
-
|
|
1909
|
-
|
|
1910
|
-
|
|
1911
|
-
|
|
1912
|
-
|
|
1913
|
-
|
|
1914
|
-
|
|
1915
|
-
|
|
1787
|
+
const m = /* @__PURE__ */ new Map();
|
|
1788
|
+
for (const span of spans) {
|
|
1789
|
+
const key = span["tangle.runId"] ?? span.traceId;
|
|
1790
|
+
const list = m.get(key) ?? [];
|
|
1791
|
+
list.push(span);
|
|
1792
|
+
m.set(key, list);
|
|
1793
|
+
}
|
|
1794
|
+
return m;
|
|
1916
1795
|
}
|
|
1917
1796
|
function findRoot(group) {
|
|
1918
|
-
|
|
1919
|
-
|
|
1920
|
-
|
|
1921
|
-
return orderSpans(pool)[0];
|
|
1797
|
+
const structuralRoots = group.filter((span) => !span.parentSpanId);
|
|
1798
|
+
const terminalRoots = structuralRoots.filter(isTerminalRootCandidate);
|
|
1799
|
+
return orderSpans(terminalRoots.length > 0 ? terminalRoots : structuralRoots.length > 0 ? structuralRoots : group)[0];
|
|
1922
1800
|
}
|
|
1923
1801
|
function readAttrString(spans, keys) {
|
|
1924
|
-
|
|
1925
|
-
|
|
1926
|
-
|
|
1927
|
-
|
|
1928
|
-
}
|
|
1929
|
-
}
|
|
1930
|
-
return void 0;
|
|
1802
|
+
for (const span of spans) for (const key of keys) {
|
|
1803
|
+
const v = span.attributes[key];
|
|
1804
|
+
if (typeof v === "string" && v.length > 0) return v;
|
|
1805
|
+
}
|
|
1931
1806
|
}
|
|
1932
1807
|
function readConsistentScenarioId(runId, spans) {
|
|
1933
|
-
|
|
1934
|
-
|
|
1935
|
-
|
|
1936
|
-
|
|
1937
|
-
|
|
1938
|
-
|
|
1939
|
-
|
|
1940
|
-
|
|
1941
|
-
|
|
1942
|
-
`fromOtelSpans: conflicting scenario ids for run '${runId}': ${[...values].sort().join(", ")}`
|
|
1943
|
-
);
|
|
1944
|
-
}
|
|
1945
|
-
return values.values().next().value;
|
|
1808
|
+
const values = /* @__PURE__ */ new Set();
|
|
1809
|
+
for (const span of spans) {
|
|
1810
|
+
const topLevel = span["tangle.scenarioId"];
|
|
1811
|
+
if (typeof topLevel === "string" && topLevel.length > 0) values.add(topLevel);
|
|
1812
|
+
const attribute = span.attributes["tangle.scenarioId"];
|
|
1813
|
+
if (typeof attribute === "string" && attribute.length > 0) values.add(attribute);
|
|
1814
|
+
}
|
|
1815
|
+
if (values.size > 1) throw new ValidationError(`fromOtelSpans: conflicting scenario ids for run '${runId}': ${[...values].sort().join(", ")}`);
|
|
1816
|
+
return values.values().next().value;
|
|
1946
1817
|
}
|
|
1947
1818
|
function resolveTaskScore(runId, spans, scoreForRun) {
|
|
1948
|
-
|
|
1949
|
-
|
|
1950
|
-
|
|
1951
|
-
|
|
1952
|
-
|
|
1953
|
-
|
|
1954
|
-
|
|
1955
|
-
|
|
1956
|
-
|
|
1957
|
-
|
|
1958
|
-
|
|
1959
|
-
|
|
1960
|
-
|
|
1961
|
-
|
|
1962
|
-
|
|
1963
|
-
|
|
1964
|
-
|
|
1965
|
-
|
|
1966
|
-
|
|
1967
|
-
|
|
1968
|
-
|
|
1969
|
-
|
|
1970
|
-
|
|
1971
|
-
|
|
1972
|
-
|
|
1973
|
-
|
|
1974
|
-
|
|
1975
|
-
|
|
1976
|
-
|
|
1977
|
-
const details = sources.map((source) => `${source.label}=${source.value}`).join(", ");
|
|
1978
|
-
throw new ValidationError(
|
|
1979
|
-
`fromOtelSpans: conflicting task-quality scores for run '${runId}': ${details}`
|
|
1980
|
-
);
|
|
1981
|
-
}
|
|
1982
|
-
return score;
|
|
1819
|
+
const orderedSpans = orderSpans(spans);
|
|
1820
|
+
const sources = [];
|
|
1821
|
+
if (scoreForRun) {
|
|
1822
|
+
const supplied = scoreForRun(runId, orderedSpans);
|
|
1823
|
+
if (supplied !== void 0) {
|
|
1824
|
+
if (typeof supplied !== "number" || !Number.isFinite(supplied)) throw new ValidationError(`fromOtelSpans: scoreForRun returned a non-finite number for run '${runId}'`);
|
|
1825
|
+
sources.push({
|
|
1826
|
+
label: "scoreForRun",
|
|
1827
|
+
value: supplied
|
|
1828
|
+
});
|
|
1829
|
+
}
|
|
1830
|
+
}
|
|
1831
|
+
for (const span of orderedSpans) {
|
|
1832
|
+
const role = errorRoleForSpan(span);
|
|
1833
|
+
if (span.parentSpanId && role !== "EVALUATOR") continue;
|
|
1834
|
+
if (role === "EVALUATOR" && span.status?.code === "ERROR") continue;
|
|
1835
|
+
for (const key of TASK_SCORE_ATTR_KEYS) {
|
|
1836
|
+
if (!Object.hasOwn(span.attributes, key)) continue;
|
|
1837
|
+
sources.push({
|
|
1838
|
+
label: `span '${span.spanId}' attribute '${key}'`,
|
|
1839
|
+
value: parseTaskScoreAttribute(runId, span.spanId, key, span.attributes[key])
|
|
1840
|
+
});
|
|
1841
|
+
}
|
|
1842
|
+
}
|
|
1843
|
+
if (sources.length === 0) return void 0;
|
|
1844
|
+
sources.sort((left, right) => left.label.localeCompare(right.label));
|
|
1845
|
+
const score = sources[0].value;
|
|
1846
|
+
if (sources.some((source) => source.value !== score)) throw new ValidationError(`fromOtelSpans: conflicting task-quality scores for run '${runId}': ${sources.map((source) => `${source.label}=${source.value}`).join(", ")}`);
|
|
1847
|
+
return score;
|
|
1983
1848
|
}
|
|
1984
1849
|
function parseTaskScoreAttribute(runId, spanId, key, value) {
|
|
1985
|
-
|
|
1986
|
-
|
|
1987
|
-
|
|
1988
|
-
|
|
1989
|
-
|
|
1990
|
-
|
|
1991
|
-
|
|
1992
|
-
const parsed = Number(value);
|
|
1993
|
-
if (Number.isFinite(parsed)) return parsed;
|
|
1994
|
-
} else if (typeof value === "number" && Number.isFinite(value)) {
|
|
1995
|
-
return value;
|
|
1996
|
-
}
|
|
1997
|
-
throw new ValidationError(
|
|
1998
|
-
`fromOtelSpans: ${source} is not a finite task-quality score for run '${runId}'`
|
|
1999
|
-
);
|
|
1850
|
+
const source = `span '${spanId}' attribute '${key}'`;
|
|
1851
|
+
if (typeof value === "string") {
|
|
1852
|
+
if (value.trim().length === 0) throw new ValidationError(`fromOtelSpans: ${source} is blank for run '${runId}'; task quality must be finite`);
|
|
1853
|
+
const parsed = Number(value);
|
|
1854
|
+
if (Number.isFinite(parsed)) return parsed;
|
|
1855
|
+
} else if (typeof value === "number" && Number.isFinite(value)) return value;
|
|
1856
|
+
throw new ValidationError(`fromOtelSpans: ${source} is not a finite task-quality score for run '${runId}'`);
|
|
2000
1857
|
}
|
|
2001
1858
|
function orderSpans(spans) {
|
|
2002
|
-
|
|
2003
|
-
(left, right) => compareUnixNano(left.startTimeUnixNano, right.startTimeUnixNano) || left.spanId.localeCompare(right.spanId)
|
|
2004
|
-
);
|
|
1859
|
+
return [...spans].sort((left, right) => compareUnixNano(left.startTimeUnixNano, right.startTimeUnixNano) || left.spanId.localeCompare(right.spanId));
|
|
2005
1860
|
}
|
|
2006
1861
|
function parseUnixNano(value) {
|
|
2007
|
-
|
|
2008
|
-
|
|
2009
|
-
|
|
2010
|
-
|
|
2011
|
-
|
|
1862
|
+
try {
|
|
1863
|
+
return BigInt(value);
|
|
1864
|
+
} catch {
|
|
1865
|
+
throw new ValidationError(`fromOtelSpans: invalid Unix nanosecond timestamp '${value}'`);
|
|
1866
|
+
}
|
|
2012
1867
|
}
|
|
2013
1868
|
function compareUnixNano(left, right) {
|
|
2014
|
-
|
|
2015
|
-
|
|
2016
|
-
|
|
1869
|
+
const leftValue = parseUnixNano(left);
|
|
1870
|
+
const rightValue = parseUnixNano(right);
|
|
1871
|
+
return leftValue < rightValue ? -1 : leftValue > rightValue ? 1 : 0;
|
|
2017
1872
|
}
|
|
2018
1873
|
function unixNanoDurationMs(start, end) {
|
|
2019
|
-
|
|
2020
|
-
|
|
2021
|
-
|
|
2022
|
-
|
|
2023
|
-
|
|
2024
|
-
|
|
2025
|
-
|
|
2026
|
-
}
|
|
2027
|
-
return value;
|
|
1874
|
+
const delta = parseUnixNano(end) - parseUnixNano(start);
|
|
1875
|
+
if (delta <= 0n) return 0;
|
|
1876
|
+
const wholeMs = delta / 1000000n;
|
|
1877
|
+
const fractionalMs = delta % 1000000n;
|
|
1878
|
+
const value = Number(wholeMs) + Number(fractionalMs) / 1e6;
|
|
1879
|
+
if (!Number.isSafeInteger(Number(wholeMs))) throw new ValidationError("fromOtelSpans: span duration exceeds the safe millisecond range");
|
|
1880
|
+
return value;
|
|
2028
1881
|
}
|
|
2029
1882
|
function collectNumericAttrs(spans) {
|
|
2030
|
-
|
|
2031
|
-
|
|
2032
|
-
|
|
2033
|
-
|
|
2034
|
-
|
|
2035
|
-
|
|
2036
|
-
|
|
2037
|
-
}
|
|
2038
|
-
export {
|
|
2039
|
-
FileSystemOutcomeStore,
|
|
2040
|
-
InMemoryOutcomeStore,
|
|
2041
|
-
REFERENCE_EQUIVALENCE_INPUT_LIMITS,
|
|
2042
|
-
REFERENCE_EQUIVALENCE_JUDGE_VERSION,
|
|
2043
|
-
SelfImproveRunError,
|
|
2044
|
-
analyzeRuns,
|
|
2045
|
-
buildDefaultAnalystRegistry,
|
|
2046
|
-
buildEvidenceVector,
|
|
2047
|
-
campaignSplitDigest,
|
|
2048
|
-
compareOptimizationMethods,
|
|
2049
|
-
composeGate,
|
|
2050
|
-
createChatClient,
|
|
2051
|
-
createReferenceEquivalenceJudge,
|
|
2052
|
-
defaultProductionGate,
|
|
2053
|
-
defineAgentEval,
|
|
2054
|
-
diffGenerations,
|
|
2055
|
-
diffRunBaselineToWinner,
|
|
2056
|
-
diffRuns,
|
|
2057
|
-
evalReportingSuite,
|
|
2058
|
-
evaluatePairedMeasurements,
|
|
2059
|
-
externalTextOptimizationMethod,
|
|
2060
|
-
fromClaudeCodeSession,
|
|
2061
|
-
fromCodexSession,
|
|
2062
|
-
fromFeedbackTable,
|
|
2063
|
-
fromKimiCodeSession,
|
|
2064
|
-
fromOpenCodeSession,
|
|
2065
|
-
fromOtelSpans,
|
|
2066
|
-
fromPiSession,
|
|
2067
|
-
fromPigraphSession,
|
|
2068
|
-
fromRunRecordDir,
|
|
2069
|
-
fsCampaignStorage,
|
|
2070
|
-
gepaOptimizationMethod,
|
|
2071
|
-
heldOutGate,
|
|
2072
|
-
inMemoryCampaignStorage,
|
|
2073
|
-
llmJudge,
|
|
2074
|
-
measuredComparisonFromCandidateExperiment,
|
|
2075
|
-
observeCodeAgentSession,
|
|
2076
|
-
paretoPolicy,
|
|
2077
|
-
paretoSignificanceGate,
|
|
2078
|
-
parseAgentTrace,
|
|
2079
|
-
parseCodeAgentJsonl,
|
|
2080
|
-
partitionRunsByAuthoringModel,
|
|
2081
|
-
runCampaign,
|
|
2082
|
-
runCandidateExperiment,
|
|
2083
|
-
runEval,
|
|
2084
|
-
runImprovementLoop,
|
|
2085
|
-
runReferenceEquivalenceJudge,
|
|
2086
|
-
sealCandidateBenchmarkSuite,
|
|
2087
|
-
sealCandidateBenchmarkTask,
|
|
2088
|
-
sealCandidateExperiment,
|
|
2089
|
-
selfImprove,
|
|
2090
|
-
skillOptOptimizationMethod,
|
|
2091
|
-
summarizeExecution,
|
|
2092
|
-
verifyCandidateBenchmarkSuite,
|
|
2093
|
-
verifyCandidateBenchmarkSuiteInputs,
|
|
2094
|
-
verifyCandidateBenchmarkTask,
|
|
2095
|
-
verifyCandidateExperiment,
|
|
2096
|
-
verifyCandidateExperimentComparison
|
|
2097
|
-
};
|
|
1883
|
+
const raw = {};
|
|
1884
|
+
for (const span of spans) for (const [k, v] of Object.entries(span.attributes)) if (typeof v === "number" && Number.isFinite(v)) raw[k] = v;
|
|
1885
|
+
return raw;
|
|
1886
|
+
}
|
|
1887
|
+
//#endregion
|
|
1888
|
+
export { FileSystemOutcomeStore, InMemoryOutcomeStore, REFERENCE_EQUIVALENCE_INPUT_LIMITS, REFERENCE_EQUIVALENCE_JUDGE_VERSION, SelfImproveRunError, analyzeRuns, buildDefaultAnalystRegistry, buildEvidenceVector, campaignSplitDigest, compareOptimizationMethods, composeGate, createChatClient, createReferenceEquivalenceJudge, defaultProductionGate, defineAgentEval, diffGenerations, diffRunBaselineToWinner, diffRuns, evalReportingSuite, evaluatePairedMeasurements, externalTextOptimizationMethod, fromClaudeCodeSession, fromCodexSession, fromFeedbackTable, fromKimiCodeSession, fromOpenCodeSession, fromOtelSpans, fromPiSession, fromPigraphSession, fromRunRecordDir, fsCampaignStorage, gepaOptimizationMethod, heldOutGate, inMemoryCampaignStorage, llmJudge, measuredComparisonFromCandidateExperiment, observeCodeAgentSession, paretoPolicy, paretoSignificanceGate, parseAgentTrace, parseCodeAgentJsonl, partitionRunsByAuthoringModel, runCampaign, runCandidateExperiment, runEval, runImprovementLoop, runReferenceEquivalenceJudge, sealCandidateBenchmarkSuite, sealCandidateBenchmarkTask, sealCandidateExperiment, selfImprove, skillOptOptimizationMethod, summarizeExecution, verifyCandidateBenchmarkSuite, verifyCandidateBenchmarkSuiteInputs, verifyCandidateBenchmarkTask, verifyCandidateExperiment, verifyCandidateExperimentComparison };
|
|
1889
|
+
|
|
2098
1890
|
//# sourceMappingURL=index.js.map
|