@tangle-network/agent-eval 0.129.0 → 0.130.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +21 -0
- package/README.md +2 -0
- package/dist/active-curriculum-C4mk67HP.js +214 -0
- package/dist/active-curriculum-C4mk67HP.js.map +1 -0
- package/dist/adversarial-smnADNFS.d.ts +21 -0
- package/dist/adversarial-smnADNFS.d.ts.map +1 -0
- package/dist/analyst/index.d.ts +81 -2872
- package/dist/analyst/index.d.ts.map +1 -0
- package/dist/analyst/index.js +319 -360
- package/dist/analyst/index.js.map +1 -1
- package/dist/analyst-BkTS3C58.d.ts +89 -0
- package/dist/analyst-BkTS3C58.d.ts.map +1 -0
- package/dist/analyst-LsnNpSkm.js +152 -0
- package/dist/analyst-LsnNpSkm.js.map +1 -0
- package/dist/analyze-runs-C1CavBMk.js +1067 -0
- package/dist/analyze-runs-C1CavBMk.js.map +1 -0
- package/dist/analyze-runs-FsgYCinh.d.ts +72 -0
- package/dist/analyze-runs-FsgYCinh.d.ts.map +1 -0
- package/dist/attribute-vocabulary-DLJ6303h.d.ts +54 -0
- package/dist/attribute-vocabulary-DLJ6303h.d.ts.map +1 -0
- package/dist/authenticity/index.d.ts +81 -79
- package/dist/authenticity/index.d.ts.map +1 -0
- package/dist/authenticity/index.js +209 -193
- package/dist/authenticity/index.js.map +1 -1
- package/dist/baseline-HsBvw_dk.js +550 -0
- package/dist/baseline-HsBvw_dk.js.map +1 -0
- package/dist/baseline-hG3K85h4.d.ts +125 -0
- package/dist/baseline-hG3K85h4.d.ts.map +1 -0
- package/dist/belief-state/index.d.ts +448 -1188
- package/dist/belief-state/index.d.ts.map +1 -0
- package/dist/belief-state/index.js +1617 -1709
- package/dist/belief-state/index.js.map +1 -1
- package/dist/benchmarks/index.d.ts +2 -891
- package/dist/benchmarks/index.js +2 -60
- package/dist/benchmarks-BJgDGkAD.js +754 -0
- package/dist/benchmarks-BJgDGkAD.js.map +1 -0
- package/dist/builder-eval/index.d.ts +150 -662
- package/dist/builder-eval/index.d.ts.map +1 -0
- package/dist/builder-eval/index.js +356 -345
- package/dist/builder-eval/index.js.map +1 -1
- package/dist/calibration-CNWWA6K8.js +94 -0
- package/dist/calibration-CNWWA6K8.js.map +1 -0
- package/dist/campaign/index.d.ts +5 -6381
- package/dist/campaign/index.js +3 -213
- package/dist/campaign-aKJt6emI.js +3886 -0
- package/dist/campaign-aKJt6emI.js.map +1 -0
- package/dist/cli.d.ts +1 -1
- package/dist/cli.js +164 -175
- package/dist/cli.js.map +1 -1
- package/dist/client-C97NMzqi.d.ts +581 -0
- package/dist/client-C97NMzqi.d.ts.map +1 -0
- package/dist/client-CYzbdJOZ.js +637 -0
- package/dist/client-CYzbdJOZ.js.map +1 -0
- package/dist/code-agent-session-BjkMTQ7H.js +1390 -0
- package/dist/code-agent-session-BjkMTQ7H.js.map +1 -0
- package/dist/code-agent-session-DqqgOJaz.d.ts +143 -0
- package/dist/code-agent-session-DqqgOJaz.d.ts.map +1 -0
- package/dist/concurrency-DIxRZF_J.js +85 -0
- package/dist/concurrency-DIxRZF_J.js.map +1 -0
- package/dist/contract/index.d.ts +645 -5565
- package/dist/contract/index.d.ts.map +1 -0
- package/dist/contract/index.js +1730 -1938
- package/dist/contract/index.js.map +1 -1
- package/dist/control.d.ts +3 -1030
- package/dist/control.js +2 -33
- package/dist/cost-ledger-DIgQUFZZ.js +801 -0
- package/dist/cost-ledger-DIgQUFZZ.js.map +1 -0
- package/dist/cost-ledger-Dye6jCgg.d.ts +236 -0
- package/dist/cost-ledger-Dye6jCgg.d.ts.map +1 -0
- package/dist/dataset-BvtnC8Dc.d.ts +115 -0
- package/dist/dataset-BvtnC8Dc.d.ts.map +1 -0
- package/dist/default-registry-C-vFCSEc.js +2579 -0
- package/dist/default-registry-C-vFCSEc.js.map +1 -0
- package/dist/default-registry-CNPo-Vsb.d.ts +540 -0
- package/dist/default-registry-CNPo-Vsb.d.ts.map +1 -0
- package/dist/emitter-CPBAhxum.js +266 -0
- package/dist/emitter-CPBAhxum.js.map +1 -0
- package/dist/emitter-DGQGoLyj.d.ts +113 -0
- package/dist/emitter-DGQGoLyj.d.ts.map +1 -0
- package/dist/errors-8YnH8WlF.js +64 -0
- package/dist/errors-8YnH8WlF.js.map +1 -0
- package/dist/errors-CEk209JS.d.ts +76 -0
- package/dist/errors-CEk209JS.d.ts.map +1 -0
- package/dist/eval-campaign-DEm6c8ru.js +349 -0
- package/dist/eval-campaign-DEm6c8ru.js.map +1 -0
- package/dist/execution-tracks-CpgFPpS5.js +93 -0
- package/dist/execution-tracks-CpgFPpS5.js.map +1 -0
- package/dist/exporters-q9iL-2Jf.js +148 -0
- package/dist/exporters-q9iL-2Jf.js.map +1 -0
- package/dist/extract-usage-BrQ8mCLX.js +155 -0
- package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
- package/dist/failure-cluster-CqcvCcdR.d.ts +59 -0
- package/dist/failure-cluster-CqcvCcdR.d.ts.map +1 -0
- package/dist/feedback-trajectory-CVaeREXV.d.ts +340 -0
- package/dist/feedback-trajectory-CVaeREXV.d.ts.map +1 -0
- package/dist/fuzz.d.ts +320 -646
- package/dist/fuzz.d.ts.map +1 -0
- package/dist/fuzz.js +670 -618
- package/dist/fuzz.js.map +1 -1
- package/dist/hf-dataset-DBJXXoY1.js +763 -0
- package/dist/hf-dataset-DBJXXoY1.js.map +1 -0
- package/dist/hosted/index.d.ts +11 -831
- package/dist/hosted/index.d.ts.map +1 -0
- package/dist/hosted/index.js +2 -37
- package/dist/index-2JJSA6-r2.d.ts +926 -0
- package/dist/index-2JJSA6-r2.d.ts.map +1 -0
- package/dist/index-6N0aYmpW.d.ts +217 -0
- package/dist/index-6N0aYmpW.d.ts.map +1 -0
- package/dist/index-BAvgST_9.d.ts +131 -0
- package/dist/index-BAvgST_9.d.ts.map +1 -0
- package/dist/index-BvnJuTGD.d.ts +68 -0
- package/dist/index-BvnJuTGD.d.ts.map +1 -0
- package/dist/index-C61Wi7yg.d.ts +547 -0
- package/dist/index-C61Wi7yg.d.ts.map +1 -0
- package/dist/index-CD_WZ_Xr.d.ts +2250 -0
- package/dist/index-CD_WZ_Xr.d.ts.map +1 -0
- package/dist/index-DSC51roc.d.ts +102 -0
- package/dist/index-DSC51roc.d.ts.map +1 -0
- package/dist/index-Em67JBjs.d.ts +335 -0
- package/dist/index-Em67JBjs.d.ts.map +1 -0
- package/dist/index.d.ts +3755 -15555
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +11182 -11216
- package/dist/index.js.map +1 -1
- package/dist/integrity-BzRbCHzi.js +424 -0
- package/dist/integrity-BzRbCHzi.js.map +1 -0
- package/dist/integrity-rmVhXWA7.d.ts +61 -0
- package/dist/integrity-rmVhXWA7.d.ts.map +1 -0
- package/dist/judge-calibration-DFtEMlde.d.ts +146 -0
- package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
- package/dist/ledger-core/index.d.ts +2 -0
- package/dist/ledger-core/index.js +2 -0
- package/dist/ledger-core-DtZz1RG0.js +388 -0
- package/dist/ledger-core-DtZz1RG0.js.map +1 -0
- package/dist/llm-client--GR4JbZE.js +687 -0
- package/dist/llm-client--GR4JbZE.js.map +1 -0
- package/dist/llm-client-B_nIBlYo.d.ts +290 -0
- package/dist/llm-client-B_nIBlYo.d.ts.map +1 -0
- package/dist/matrix/index.d.ts +3 -155
- package/dist/matrix/index.js +2 -8
- package/dist/matrix-BzQnu2S6.js +270 -0
- package/dist/matrix-BzQnu2S6.js.map +1 -0
- package/dist/meta-eval/index.d.ts +4 -1027
- package/dist/meta-eval/index.js +393 -390
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/metrics-C9YY1OcL.js +239 -0
- package/dist/metrics-C9YY1OcL.js.map +1 -0
- package/dist/mint-yN2M2eh0.js +201 -0
- package/dist/mint-yN2M2eh0.js.map +1 -0
- package/dist/multi-layer-verifier-BHY1gWAc.d.ts +138 -0
- package/dist/multi-layer-verifier-BHY1gWAc.d.ts.map +1 -0
- package/dist/multishot/index.d.ts +273 -480
- package/dist/multishot/index.d.ts.map +1 -0
- package/dist/multishot/index.js +595 -548
- package/dist/multishot/index.js.map +1 -1
- package/dist/off-policy-DvgzvtIx.js +220 -0
- package/dist/off-policy-DvgzvtIx.js.map +1 -0
- package/dist/off-policy-mskQw8Mb.d.ts +153 -0
- package/dist/off-policy-mskQw8Mb.d.ts.map +1 -0
- package/dist/openapi.json +1 -1
- package/dist/opencode-sqlite-BGrHeDu3.js +318 -0
- package/dist/opencode-sqlite-BGrHeDu3.js.map +1 -0
- package/dist/outcome-store-BYHIuO0e.d.ts +65 -0
- package/dist/outcome-store-BYHIuO0e.d.ts.map +1 -0
- package/dist/outcome-store-ChBKlTd_.js +75 -0
- package/dist/outcome-store-ChBKlTd_.js.map +1 -0
- package/dist/paired-arms-D9D0wXj2.js +260 -0
- package/dist/paired-arms-D9D0wXj2.js.map +1 -0
- package/dist/pipelines/index.d.ts +95 -532
- package/dist/pipelines/index.d.ts.map +1 -0
- package/dist/pipelines/index.js +497 -478
- package/dist/pipelines/index.js.map +1 -1
- package/dist/pre-registration-DakwTRXk.js +96 -0
- package/dist/pre-registration-DakwTRXk.js.map +1 -0
- package/dist/propose-review-control-LhIWGzHl.js +1458 -0
- package/dist/propose-review-control-LhIWGzHl.js.map +1 -0
- package/dist/query-CJ_DX8vl.d.ts +28 -0
- package/dist/query-CJ_DX8vl.d.ts.map +1 -0
- package/dist/query-Di7eEQ79.js +83 -0
- package/dist/query-Di7eEQ79.js.map +1 -0
- package/dist/raw-provider-sink-BQd7mzyT.js +201 -0
- package/dist/raw-provider-sink-BQd7mzyT.js.map +1 -0
- package/dist/raw-provider-sink-BU29Sh8h.d.ts +134 -0
- package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +1 -0
- package/dist/redact-7Aq1ukl-.js +107 -0
- package/dist/redact-7Aq1ukl-.js.map +1 -0
- package/dist/release-report-Cz9NKH39.js +603 -0
- package/dist/release-report-Cz9NKH39.js.map +1 -0
- package/dist/release-report-mvB2G4_J.d.ts +244 -0
- package/dist/release-report-mvB2G4_J.d.ts.map +1 -0
- package/dist/replay-GnyotH0J.js +1741 -0
- package/dist/replay-GnyotH0J.js.map +1 -0
- package/dist/replay-OoidtG1E.d.ts +749 -0
- package/dist/replay-OoidtG1E.d.ts.map +1 -0
- package/dist/reporting.d.ts +6 -1312
- package/dist/reporting.js +6 -51
- package/dist/researcher-CwTdwXG1.d.ts +314 -0
- package/dist/researcher-CwTdwXG1.d.ts.map +1 -0
- package/dist/reward-hacking-eAnOsynk.d.ts +224 -0
- package/dist/reward-hacking-eAnOsynk.d.ts.map +1 -0
- package/dist/reward-hacking-qipEpKvY.js +596 -0
- package/dist/reward-hacking-qipEpKvY.js.map +1 -0
- package/dist/reward-nw2xZGZG.js +137 -0
- package/dist/reward-nw2xZGZG.js.map +1 -0
- package/dist/rl.d.ts +760 -4010
- package/dist/rl.d.ts.map +1 -0
- package/dist/rl.js +2325 -1958
- package/dist/rl.js.map +1 -1
- package/dist/rolldown-runtime-8H4AJuhK.js +14 -0
- package/dist/rollout/index.d.ts +3 -2087
- package/dist/rollout/index.js +8 -168
- package/dist/rollout-BOYjemfR.js +624 -0
- package/dist/rollout-BOYjemfR.js.map +1 -0
- package/dist/rubric-predictive-validity-B3xmbmS1.js +141 -0
- package/dist/rubric-predictive-validity-B3xmbmS1.js.map +1 -0
- package/dist/rubric-predictive-validity-Ku_clp_1.d.ts +74 -0
- package/dist/rubric-predictive-validity-Ku_clp_1.d.ts.map +1 -0
- package/dist/run-evidence-ClRX_8A9.d.ts +225 -0
- package/dist/run-evidence-ClRX_8A9.d.ts.map +1 -0
- package/dist/run-record-BuoE80Dq.js +467 -0
- package/dist/{chunk-56TAVBOK.js.map → run-record-BuoE80Dq.js.map} +1 -1
- package/dist/run-record-CnZu_gjl.d.ts +357 -0
- package/dist/run-record-CnZu_gjl.d.ts.map +1 -0
- package/dist/run-score-iEEAWiBY.js +41 -0
- package/dist/run-score-iEEAWiBY.js.map +1 -0
- package/dist/runtime-trajectory-1gyaTOoC.js +93 -0
- package/dist/runtime-trajectory-1gyaTOoC.js.map +1 -0
- package/dist/runtime-trajectory-BvSZcCHD.d.ts +50 -0
- package/dist/runtime-trajectory-BvSZcCHD.d.ts.map +1 -0
- package/dist/schema-BtVldJ3T.d.ts +206 -0
- package/dist/schema-BtVldJ3T.d.ts.map +1 -0
- package/dist/schema-C6DW4ZHR.js +821 -0
- package/dist/schema-C6DW4ZHR.js.map +1 -0
- package/dist/schema-CRhEY1SO.js +69 -0
- package/dist/schema-CRhEY1SO.js.map +1 -0
- package/dist/schema-Cef2cFmb.d.ts +408 -0
- package/dist/schema-Cef2cFmb.d.ts.map +1 -0
- package/dist/semantic-concept-judge-B6cWNJ2K.js +725 -0
- package/dist/semantic-concept-judge-B6cWNJ2K.js.map +1 -0
- package/dist/sequential-Br0mAPHA.js +148 -0
- package/dist/sequential-Br0mAPHA.js.map +1 -0
- package/dist/sequential-CYwq6Ff_.d.ts +141 -0
- package/dist/sequential-CYwq6Ff_.d.ts.map +1 -0
- package/dist/series-convergence-CjO2QdRW.js +43 -0
- package/dist/series-convergence-CjO2QdRW.js.map +1 -0
- package/dist/series-convergence-ofsqPWhs.d.ts +35 -0
- package/dist/series-convergence-ofsqPWhs.d.ts.map +1 -0
- package/dist/server-m5D9cvnG.js +1040 -0
- package/dist/server-m5D9cvnG.js.map +1 -0
- package/dist/skill-usage-C_pXm7lP.d.ts +534 -0
- package/dist/skill-usage-C_pXm7lP.d.ts.map +1 -0
- package/dist/skillopt-optimization-method-CUmHkGbI.js +7718 -0
- package/dist/skillopt-optimization-method-CUmHkGbI.js.map +1 -0
- package/dist/skillopt-optimization-method-CWKVTnks.d.ts +1740 -0
- package/dist/skillopt-optimization-method-CWKVTnks.d.ts.map +1 -0
- package/dist/statistics-Cmj6nynr.d.ts +514 -0
- package/dist/statistics-Cmj6nynr.d.ts.map +1 -0
- package/dist/statistics-CnnxdpOg.js +1437 -0
- package/dist/statistics-CnnxdpOg.js.map +1 -0
- package/dist/store-CT9YIIve.d.ts +117 -0
- package/dist/store-CT9YIIve.d.ts.map +1 -0
- package/dist/store-CxJry_cs.d.ts +229 -0
- package/dist/store-CxJry_cs.d.ts.map +1 -0
- package/dist/storyboard/index.d.ts +84 -203
- package/dist/storyboard/index.d.ts.map +1 -0
- package/dist/storyboard/index.js +609 -542
- package/dist/storyboard/index.js.map +1 -1
- package/dist/summary-report-BNs5nmXI.js +862 -0
- package/dist/summary-report-BNs5nmXI.js.map +1 -0
- package/dist/summary-report-Cj9gdw4i.d.ts +407 -0
- package/dist/summary-report-Cj9gdw4i.d.ts.map +1 -0
- package/dist/supervisor-run/index.d.ts +2 -959
- package/dist/supervisor-run/index.js +2 -65
- package/dist/supervisor-run-_lnTLM3z.js +1679 -0
- package/dist/supervisor-run-_lnTLM3z.js.map +1 -0
- package/dist/task-failure-attributes-CQZlB3et.js +311 -0
- package/dist/task-failure-attributes-CQZlB3et.js.map +1 -0
- package/dist/test-graded-scenario-BsqWLmPt.js +318 -0
- package/dist/test-graded-scenario-BsqWLmPt.js.map +1 -0
- package/dist/test-graded-scenario-D1TaI2va.d.ts +141 -0
- package/dist/test-graded-scenario-D1TaI2va.d.ts.map +1 -0
- package/dist/tools-BmuN627J.js +1085 -0
- package/dist/tools-BmuN627J.js.map +1 -0
- package/dist/trace-attributes.d.ts +2 -52
- package/dist/trace-attributes.js +131 -61
- package/dist/trace-attributes.js.map +1 -1
- package/dist/traces.d.ts +12 -2365
- package/dist/traces.js +12 -252
- package/dist/trajectory-D_7rLrvE.js +56 -0
- package/dist/trajectory-D_7rLrvE.js.map +1 -0
- package/dist/types-DGsxbAEd.d.ts +387 -0
- package/dist/types-DGsxbAEd.d.ts.map +1 -0
- package/dist/types-k9tZGKUg.d.ts +640 -0
- package/dist/types-k9tZGKUg.d.ts.map +1 -0
- package/dist/verdict-Dps8_okt.d.ts +37 -0
- package/dist/verdict-Dps8_okt.d.ts.map +1 -0
- package/dist/wire/index.d.ts +702 -1173
- package/dist/wire/index.d.ts.map +1 -0
- package/dist/wire/index.js +2 -81
- package/docs/campaign-proposers.md +1 -0
- package/package.json +17 -9
- package/dist/benchmarks/index.js.map +0 -1
- package/dist/campaign/index.js.map +0 -1
- package/dist/chunk-2QU3YOPR.js +0 -7374
- package/dist/chunk-2QU3YOPR.js.map +0 -1
- package/dist/chunk-3OCR4R5I.js +0 -728
- package/dist/chunk-3OCR4R5I.js.map +0 -1
- package/dist/chunk-3RF76KTD.js +0 -84
- package/dist/chunk-3RF76KTD.js.map +0 -1
- package/dist/chunk-56TAVBOK.js +0 -698
- package/dist/chunk-5DTSBUL2.js +0 -159
- package/dist/chunk-5DTSBUL2.js.map +0 -1
- package/dist/chunk-7FO3TNPI.js +0 -232
- package/dist/chunk-7FO3TNPI.js.map +0 -1
- package/dist/chunk-7ZZMD7UK.js +0 -386
- package/dist/chunk-7ZZMD7UK.js.map +0 -1
- package/dist/chunk-BOD4O7OF.js +0 -40
- package/dist/chunk-BOD4O7OF.js.map +0 -1
- package/dist/chunk-BSO5JDQH.js +0 -2335
- package/dist/chunk-BSO5JDQH.js.map +0 -1
- package/dist/chunk-C6LXANRU.js +0 -1550
- package/dist/chunk-C6LXANRU.js.map +0 -1
- package/dist/chunk-DODXQREJ.js +0 -752
- package/dist/chunk-DODXQREJ.js.map +0 -1
- package/dist/chunk-DRYIUNWY.js +0 -622
- package/dist/chunk-DRYIUNWY.js.map +0 -1
- package/dist/chunk-E7QXT7SX.js +0 -183
- package/dist/chunk-E7QXT7SX.js.map +0 -1
- package/dist/chunk-EG66UGL4.js +0 -341
- package/dist/chunk-EG66UGL4.js.map +0 -1
- package/dist/chunk-FXTVJPYD.js +0 -576
- package/dist/chunk-FXTVJPYD.js.map +0 -1
- package/dist/chunk-G7MGMCZD.js +0 -153
- package/dist/chunk-G7MGMCZD.js.map +0 -1
- package/dist/chunk-GGE4NNQT.js +0 -65
- package/dist/chunk-GGE4NNQT.js.map +0 -1
- package/dist/chunk-H23X7XKK.js +0 -181
- package/dist/chunk-H23X7XKK.js.map +0 -1
- package/dist/chunk-HHWE3POT.js +0 -94
- package/dist/chunk-HHWE3POT.js.map +0 -1
- package/dist/chunk-HPWUNB47.js +0 -289
- package/dist/chunk-HPWUNB47.js.map +0 -1
- package/dist/chunk-IYCLP2N2.js +0 -766
- package/dist/chunk-IYCLP2N2.js.map +0 -1
- package/dist/chunk-JHCHEVET.js +0 -274
- package/dist/chunk-JHCHEVET.js.map +0 -1
- package/dist/chunk-JQSF5DQT.js +0 -701
- package/dist/chunk-JQSF5DQT.js.map +0 -1
- package/dist/chunk-K4DBDHLK.js +0 -158
- package/dist/chunk-K4DBDHLK.js.map +0 -1
- package/dist/chunk-K6N6XJJX.js +0 -306
- package/dist/chunk-K6N6XJJX.js.map +0 -1
- package/dist/chunk-M4YBQKIJ.js +0 -1040
- package/dist/chunk-M4YBQKIJ.js.map +0 -1
- package/dist/chunk-MA6HLL3S.js +0 -65
- package/dist/chunk-MA6HLL3S.js.map +0 -1
- package/dist/chunk-MAZ26DC7.js +0 -99
- package/dist/chunk-MAZ26DC7.js.map +0 -1
- package/dist/chunk-NPCTHQIO.js +0 -91
- package/dist/chunk-NPCTHQIO.js.map +0 -1
- package/dist/chunk-NY44NC4A.js +0 -1056
- package/dist/chunk-NY44NC4A.js.map +0 -1
- package/dist/chunk-OIUOT4QD.js +0 -44
- package/dist/chunk-OIUOT4QD.js.map +0 -1
- package/dist/chunk-ONWEPEDO.js +0 -57
- package/dist/chunk-ONWEPEDO.js.map +0 -1
- package/dist/chunk-OWN5NPMC.js +0 -152
- package/dist/chunk-OWN5NPMC.js.map +0 -1
- package/dist/chunk-P6FYH6K4.js +0 -1161
- package/dist/chunk-P6FYH6K4.js.map +0 -1
- package/dist/chunk-PC4UYEBM.js +0 -166
- package/dist/chunk-PC4UYEBM.js.map +0 -1
- package/dist/chunk-PC5DOSM7.js +0 -579
- package/dist/chunk-PC5DOSM7.js.map +0 -1
- package/dist/chunk-PXE2VKMX.js +0 -140
- package/dist/chunk-PXE2VKMX.js.map +0 -1
- package/dist/chunk-PZ5AY32C.js +0 -10
- package/dist/chunk-PZ5AY32C.js.map +0 -1
- package/dist/chunk-QB6BDBP2.js +0 -4464
- package/dist/chunk-QB6BDBP2.js.map +0 -1
- package/dist/chunk-RXHCETDZ.js +0 -536
- package/dist/chunk-RXHCETDZ.js.map +0 -1
- package/dist/chunk-RZTMDUO7.js +0 -49
- package/dist/chunk-RZTMDUO7.js.map +0 -1
- package/dist/chunk-SFLLL76A.js +0 -669
- package/dist/chunk-SFLLL76A.js.map +0 -1
- package/dist/chunk-SZLVEKMJ.js +0 -1446
- package/dist/chunk-SZLVEKMJ.js.map +0 -1
- package/dist/chunk-T4SQEITX.js +0 -95
- package/dist/chunk-T4SQEITX.js.map +0 -1
- package/dist/chunk-T6RLYGAD.js +0 -158
- package/dist/chunk-T6RLYGAD.js.map +0 -1
- package/dist/chunk-TJVT4QFF.js +0 -911
- package/dist/chunk-TJVT4QFF.js.map +0 -1
- package/dist/chunk-TQ7LNKZ3.js +0 -136
- package/dist/chunk-TQ7LNKZ3.js.map +0 -1
- package/dist/chunk-U4L7JRPZ.js +0 -1706
- package/dist/chunk-U4L7JRPZ.js.map +0 -1
- package/dist/chunk-U4PHLT2N.js +0 -419
- package/dist/chunk-U4PHLT2N.js.map +0 -1
- package/dist/chunk-VCZ5FQYW.js +0 -928
- package/dist/chunk-VCZ5FQYW.js.map +0 -1
- package/dist/chunk-VI2UW6B6.js +0 -162
- package/dist/chunk-VI2UW6B6.js.map +0 -1
- package/dist/chunk-VQMK5FMP.js +0 -247
- package/dist/chunk-VQMK5FMP.js.map +0 -1
- package/dist/chunk-WGXIEX7P.js +0 -116
- package/dist/chunk-WGXIEX7P.js.map +0 -1
- package/dist/chunk-WVATSFCP.js +0 -1553
- package/dist/chunk-WVATSFCP.js.map +0 -1
- package/dist/chunk-X4YIBDER.js +0 -1662
- package/dist/chunk-X4YIBDER.js.map +0 -1
- package/dist/chunk-YQN4ICPP.js +0 -355
- package/dist/chunk-YQN4ICPP.js.map +0 -1
- package/dist/chunk-ZET2UAYW.js +0 -89
- package/dist/chunk-ZET2UAYW.js.map +0 -1
- package/dist/chunk-ZHTZ4EYI.js +0 -1212
- package/dist/chunk-ZHTZ4EYI.js.map +0 -1
- package/dist/control.js.map +0 -1
- package/dist/hosted/index.js.map +0 -1
- package/dist/matrix/index.js.map +0 -1
- package/dist/reporting.js.map +0 -1
- package/dist/rollout/index.js.map +0 -1
- package/dist/run-campaign-OJJ7CZF4.js +0 -18
- package/dist/run-campaign-OJJ7CZF4.js.map +0 -1
- package/dist/supervisor-run/index.js.map +0 -1
- package/dist/traces.js.map +0 -1
- package/dist/wire/index.js.map +0 -1
|
@@ -0,0 +1,3886 @@
|
|
|
1
|
+
import { s as ValidationError, t as AgentEvalError } from "./errors-8YnH8WlF.js";
|
|
2
|
+
import { t as canonicalize } from "./pre-registration-DakwTRXk.js";
|
|
3
|
+
import { m as buildAgentProfileCell, o as runTaskScore, r as modelHasSnapshot, s as validateRunRecord } from "./run-record-BuoE80Dq.js";
|
|
4
|
+
import { i as CostLedger } from "./cost-ledger-DIgQUFZZ.js";
|
|
5
|
+
import { f as maximumChargeForLlmRequest, l as costReceiptFromLlm, u as costReceiptFromLlmError } from "./llm-client--GR4JbZE.js";
|
|
6
|
+
import { $ as SEARCH_LEDGER_FILE_CONTEXT, B as surfaceContentHash, Bt as JudgeParseError, E as pairHoldout, Et as contentHash, F as assertCodeSurfaceIdentity, J as planCampaignRun, Lt as assertRealBackend, Pt as recoverTruncatedJson, Y as runCampaign, et as SearchLedgerConflictError, h as labelTrustRank, nt as SearchLedgerIntegrityError, tt as SearchLedgerError, zt as summarizeBackendIntegrity } from "./skillopt-optimization-method-CUmHkGbI.js";
|
|
7
|
+
import { c as eProcess, g as mulberry32, v as pairedBootstrap } from "./statistics-CnnxdpOg.js";
|
|
8
|
+
import { t as comparePairedArms } from "./paired-arms-D9D0wXj2.js";
|
|
9
|
+
import { t as analyzeTraces } from "./analyst-LsnNpSkm.js";
|
|
10
|
+
import { c as campaignCellToRunRecord } from "./reward-hacking-qipEpKvY.js";
|
|
11
|
+
import { n as canonicalString, t as FileLedgerJournal } from "./ledger-core-DtZz1RG0.js";
|
|
12
|
+
import { z } from "zod";
|
|
13
|
+
import { closeSync, constants, existsSync, fstatSync, lstatSync, mkdirSync, mkdtempSync, openSync, readFileSync, readSync, readdirSync, readlinkSync, realpathSync, rmSync, statSync, writeFileSync } from "node:fs";
|
|
14
|
+
import { basename, dirname, isAbsolute, join, relative, resolve, sep } from "node:path";
|
|
15
|
+
import { createHash, randomUUID } from "node:crypto";
|
|
16
|
+
import { devNull, tmpdir } from "node:os";
|
|
17
|
+
import { execFileSync } from "node:child_process";
|
|
18
|
+
import { harnessSupportsModel } from "@tangle-network/agent-interface";
|
|
19
|
+
//#region src/completion-verifier.ts
|
|
20
|
+
/**
|
|
21
|
+
* Completion verifier — the task-completion oracle.
|
|
22
|
+
*
|
|
23
|
+
* Answers the only eval question that is not a proxy: did the agent actually
|
|
24
|
+
* COMPLETE the task — produce every required deliverable, persisted and
|
|
25
|
+
* correct — rather than describe what should be done. A fluent transcript
|
|
26
|
+
* that never produces the artifact scores zero here.
|
|
27
|
+
*
|
|
28
|
+
* Per requirement, a two-stage check:
|
|
29
|
+
* 1. Structural — a produced item (vault artifact / approved proposal /
|
|
30
|
+
* tool call) of the right kind is matched against the requirement and
|
|
31
|
+
* carries non-empty content. Deterministic; no LLM.
|
|
32
|
+
* 2. Correctness — only if structurally present AND the matched item
|
|
33
|
+
* carries content, one targeted check decides whether that item
|
|
34
|
+
* actually fulfils the requirement. A hallucinated artifact fails here;
|
|
35
|
+
* an absent one already failed stage 1.
|
|
36
|
+
*
|
|
37
|
+
* `completionRate` is satisfied / MEASURABLE requirements (unmeasured rows —
|
|
38
|
+
* checker failures — are excluded from the denominator, never scored as
|
|
39
|
+
* zeros). Quality dimensions are meaningless on an incomplete task — callers
|
|
40
|
+
* gate on `fullyComplete` / `completionRate` before scoring quality.
|
|
41
|
+
*/
|
|
42
|
+
/**
|
|
43
|
+
* Construct a `CompletionVerdict` from the per-requirement checks, deriving
|
|
44
|
+
* `completionRate` / `fullyComplete` and the spine fields (`valid` =
|
|
45
|
+
* `fullyComplete`, `score` = `completionRate`) in one place. Throws on zero
|
|
46
|
+
* requirements — a verdict over nothing is a misconfiguration, mirroring
|
|
47
|
+
* `verifyCompletion`'s gold-spec guard.
|
|
48
|
+
*/
|
|
49
|
+
function completionVerdict(input) {
|
|
50
|
+
if (input.requirements.length === 0) throw new Error(`completionVerdict: task '${input.taskId}' has no requirement checks — nothing to derive a verdict from`);
|
|
51
|
+
const measurable = input.requirements.filter((r) => !r.unmeasured);
|
|
52
|
+
const unmeasuredCount = input.requirements.length - measurable.length;
|
|
53
|
+
if (measurable.length === 0) throw new Error(`completionVerdict: task '${input.taskId}' has no measurable requirements — all ${input.requirements.length} correctness checks failed (${input.requirements[0]?.unmeasuredReason ?? "unknown reason"})`);
|
|
54
|
+
const satisfiedCount = measurable.filter((r) => r.satisfied).length;
|
|
55
|
+
const completionRate = satisfiedCount / measurable.length;
|
|
56
|
+
const fullyComplete = unmeasuredCount === 0 && satisfiedCount === measurable.length;
|
|
57
|
+
return {
|
|
58
|
+
taskId: input.taskId,
|
|
59
|
+
requirements: input.requirements,
|
|
60
|
+
completionRate,
|
|
61
|
+
fullyComplete,
|
|
62
|
+
unmeasuredCount,
|
|
63
|
+
valid: fullyComplete,
|
|
64
|
+
score: completionRate
|
|
65
|
+
};
|
|
66
|
+
}
|
|
67
|
+
const STOPWORDS = /* @__PURE__ */ new Set([
|
|
68
|
+
"the",
|
|
69
|
+
"a",
|
|
70
|
+
"an",
|
|
71
|
+
"of",
|
|
72
|
+
"for",
|
|
73
|
+
"and",
|
|
74
|
+
"or",
|
|
75
|
+
"to",
|
|
76
|
+
"in",
|
|
77
|
+
"on",
|
|
78
|
+
"with",
|
|
79
|
+
"by"
|
|
80
|
+
]);
|
|
81
|
+
const REQUIREMENT_FORM_STOPWORDS = /* @__PURE__ */ new Set([
|
|
82
|
+
"generated",
|
|
83
|
+
"generate",
|
|
84
|
+
"view",
|
|
85
|
+
"render",
|
|
86
|
+
"rendered",
|
|
87
|
+
"persisted",
|
|
88
|
+
"persist",
|
|
89
|
+
"artifact",
|
|
90
|
+
"file",
|
|
91
|
+
"document",
|
|
92
|
+
"note",
|
|
93
|
+
"proposal",
|
|
94
|
+
"deliverable",
|
|
95
|
+
"output",
|
|
96
|
+
"created",
|
|
97
|
+
"create",
|
|
98
|
+
"produce",
|
|
99
|
+
"produced",
|
|
100
|
+
"flag"
|
|
101
|
+
]);
|
|
102
|
+
const MATCH_THRESHOLD = .5;
|
|
103
|
+
const MIN_CONTENT_CHARS = 50;
|
|
104
|
+
function tokens(s, extraStop) {
|
|
105
|
+
return new Set(s.toLowerCase().split(/[^a-z0-9]+/).filter((t) => t.length > 1 && !STOPWORDS.has(t) && !extraStop?.has(t)));
|
|
106
|
+
}
|
|
107
|
+
/**
|
|
108
|
+
* Recall of the requirement's tokens within a candidate's identifying text.
|
|
109
|
+
* Recall, not Jaccard — a candidate's path/id legitimately carries extra
|
|
110
|
+
* tokens the requirement does not name. The requirement side drops
|
|
111
|
+
* deliverable-FORM vocabulary so recall keys on the distinctive domain tokens.
|
|
112
|
+
*/
|
|
113
|
+
function tokenRecall(requirementText, candidateText) {
|
|
114
|
+
const req = tokens(requirementText, REQUIREMENT_FORM_STOPWORDS);
|
|
115
|
+
if (req.size === 0) return 0;
|
|
116
|
+
const cand = tokens(candidateText);
|
|
117
|
+
let hit = 0;
|
|
118
|
+
for (const t of req) if (cand.has(t)) hit++;
|
|
119
|
+
return hit / req.size;
|
|
120
|
+
}
|
|
121
|
+
function artifactCandidates(req, reqIndex, artifacts) {
|
|
122
|
+
const reqText = `${req.title} ${req.category ?? ""}`;
|
|
123
|
+
const out = [];
|
|
124
|
+
artifacts.forEach((a, i) => {
|
|
125
|
+
if ((a.content ?? "").trim().length < MIN_CONTENT_CHARS) return;
|
|
126
|
+
let score = tokenRecall(reqText, `${a.path ?? ""} ${a.kind} ${(a.content ?? "").slice(0, 4e3)}`);
|
|
127
|
+
if (req.category && a.kind && req.category.toLowerCase() === a.kind.toLowerCase()) score = Math.max(score, 1);
|
|
128
|
+
if (score < MATCH_THRESHOLD) return;
|
|
129
|
+
out.push({
|
|
130
|
+
reqIndex,
|
|
131
|
+
itemKey: `artifact:${i}`,
|
|
132
|
+
score,
|
|
133
|
+
evidence: `artifact '${a.path ?? a.kind}' matched (token recall ${score.toFixed(2)})`,
|
|
134
|
+
content: a.content ?? null
|
|
135
|
+
});
|
|
136
|
+
});
|
|
137
|
+
return out;
|
|
138
|
+
}
|
|
139
|
+
function proposalCandidates(req, reqIndex, proposals) {
|
|
140
|
+
const reqText = `${req.title} ${req.category ?? ""}`;
|
|
141
|
+
const out = [];
|
|
142
|
+
for (const p of proposals) {
|
|
143
|
+
if (p.status !== "approved") continue;
|
|
144
|
+
const body = (p.content ?? "").trim();
|
|
145
|
+
if (body.length < MIN_CONTENT_CHARS) continue;
|
|
146
|
+
const score = tokenRecall(reqText, `${p.title} ${body}`);
|
|
147
|
+
if (score < MATCH_THRESHOLD) continue;
|
|
148
|
+
out.push({
|
|
149
|
+
reqIndex,
|
|
150
|
+
itemKey: `proposal:${p.id}`,
|
|
151
|
+
score,
|
|
152
|
+
evidence: `approved proposal '${p.title}' matched (token recall ${score.toFixed(2)})`,
|
|
153
|
+
content: body
|
|
154
|
+
});
|
|
155
|
+
}
|
|
156
|
+
return out;
|
|
157
|
+
}
|
|
158
|
+
function toolCallCandidates(req, reqIndex, toolCalls) {
|
|
159
|
+
const out = [];
|
|
160
|
+
toolCalls.forEach((name, i) => {
|
|
161
|
+
const score = tokenRecall(req.title, name);
|
|
162
|
+
if (score < MATCH_THRESHOLD) return;
|
|
163
|
+
out.push({
|
|
164
|
+
reqIndex,
|
|
165
|
+
itemKey: `tool:${i}`,
|
|
166
|
+
score,
|
|
167
|
+
evidence: `tool call '${name}' matched (token recall ${score.toFixed(2)})`,
|
|
168
|
+
content: null
|
|
169
|
+
});
|
|
170
|
+
});
|
|
171
|
+
return out;
|
|
172
|
+
}
|
|
173
|
+
/**
|
|
174
|
+
* Verify whether a run completed the task. `checkCorrectness` is injected —
|
|
175
|
+
* `createLlmCorrectnessChecker` for production, a deterministic stub in tests.
|
|
176
|
+
*
|
|
177
|
+
* Throws on a gold spec with no requirements: an eval task that requires
|
|
178
|
+
* nothing is a misconfiguration, not a vacuously-complete task.
|
|
179
|
+
*/
|
|
180
|
+
async function verifyCompletion(gold, state, checkCorrectness) {
|
|
181
|
+
if (gold.requirements.length === 0) throw new Error(`verifyCompletion: task '${gold.taskId}' has no requirements — malformed gold spec`);
|
|
182
|
+
const candidates = [];
|
|
183
|
+
gold.requirements.forEach((req, i) => {
|
|
184
|
+
const by = req.satisfiedBy ?? "any";
|
|
185
|
+
if (by === "artifact" || by === "any") candidates.push(...artifactCandidates(req, i, state.artifacts));
|
|
186
|
+
if (by === "proposal" || by === "any") candidates.push(...proposalCandidates(req, i, state.proposals));
|
|
187
|
+
if (by === "tool-call" || by === "any") candidates.push(...toolCallCandidates(req, i, state.toolCalls));
|
|
188
|
+
});
|
|
189
|
+
candidates.sort((a, b) => b.score - a.score);
|
|
190
|
+
const assigned = /* @__PURE__ */ new Map();
|
|
191
|
+
const itemTaken = /* @__PURE__ */ new Set();
|
|
192
|
+
for (const c of candidates) {
|
|
193
|
+
if (assigned.has(c.reqIndex) || itemTaken.has(c.itemKey)) continue;
|
|
194
|
+
assigned.set(c.reqIndex, c);
|
|
195
|
+
itemTaken.add(c.itemKey);
|
|
196
|
+
}
|
|
197
|
+
const requirements = [];
|
|
198
|
+
for (let i = 0; i < gold.requirements.length; i++) {
|
|
199
|
+
const req = gold.requirements[i];
|
|
200
|
+
const match = assigned.get(i);
|
|
201
|
+
const evidence = [];
|
|
202
|
+
let correct = null;
|
|
203
|
+
let unmeasuredReason;
|
|
204
|
+
if (match) {
|
|
205
|
+
evidence.push(match.evidence);
|
|
206
|
+
if (match.content !== null) try {
|
|
207
|
+
const r = await checkCorrectness(req, match.content);
|
|
208
|
+
correct = r.correct;
|
|
209
|
+
evidence.push(`correctness: ${r.correct ? "pass" : "fail"} — ${r.reason}`);
|
|
210
|
+
} catch (err) {
|
|
211
|
+
unmeasuredReason = err instanceof JudgeParseError ? `checker response unparseable after retry: ${err.raw.slice(0, 200)}` : `checker call failed: ${err instanceof Error ? err.message : String(err)}`;
|
|
212
|
+
evidence.push(`correctness: UNMEASURED — ${unmeasuredReason}`);
|
|
213
|
+
}
|
|
214
|
+
else evidence.push("correctness: not assessed — matched item carries no content");
|
|
215
|
+
} else {
|
|
216
|
+
const by = req.satisfiedBy ?? "any";
|
|
217
|
+
const kind = by === "any" ? "artifact/proposal/tool-call" : by;
|
|
218
|
+
evidence.push(`no produced ${kind} matched this requirement`);
|
|
219
|
+
}
|
|
220
|
+
const structurallyPresent = match !== void 0;
|
|
221
|
+
const unmeasured = unmeasuredReason !== void 0;
|
|
222
|
+
const satisfied = structurallyPresent && !unmeasured && correct !== false;
|
|
223
|
+
requirements.push({
|
|
224
|
+
reqId: req.reqId,
|
|
225
|
+
title: req.title,
|
|
226
|
+
structurallyPresent,
|
|
227
|
+
correct,
|
|
228
|
+
satisfied,
|
|
229
|
+
...unmeasured ? {
|
|
230
|
+
unmeasured: true,
|
|
231
|
+
unmeasuredReason
|
|
232
|
+
} : {},
|
|
233
|
+
evidence
|
|
234
|
+
});
|
|
235
|
+
}
|
|
236
|
+
return completionVerdict({
|
|
237
|
+
taskId: gold.taskId,
|
|
238
|
+
requirements
|
|
239
|
+
});
|
|
240
|
+
}
|
|
241
|
+
/**
|
|
242
|
+
* Parse the correctness checker's model response. Tolerates a response
|
|
243
|
+
* truncated mid-JSON (max_tokens cap) by auto-closing the prefix — the
|
|
244
|
+
* verdict boolean usually lands in the first few tokens, so a recovered
|
|
245
|
+
* prefix with a boolean `correct` is a real measurement, not a guess.
|
|
246
|
+
* Fails loud (JudgeParseError) when no boolean verdict is recoverable.
|
|
247
|
+
*/
|
|
248
|
+
function parseCorrectnessResponse(raw) {
|
|
249
|
+
const readVerdict = (candidate) => {
|
|
250
|
+
if (candidate === null || typeof candidate !== "object") return null;
|
|
251
|
+
const { correct, reason } = candidate;
|
|
252
|
+
if (typeof correct !== "boolean") return null;
|
|
253
|
+
return {
|
|
254
|
+
correct,
|
|
255
|
+
reason: typeof reason === "string" ? reason : ""
|
|
256
|
+
};
|
|
257
|
+
};
|
|
258
|
+
const match = raw.match(/\{[\s\S]*\}/);
|
|
259
|
+
if (match) try {
|
|
260
|
+
const strict = readVerdict(JSON.parse(match[0]));
|
|
261
|
+
if (strict) return strict;
|
|
262
|
+
} catch {}
|
|
263
|
+
const start = raw.indexOf("{");
|
|
264
|
+
if (start !== -1) {
|
|
265
|
+
const recovered = readVerdict(recoverTruncatedJson(raw.slice(start)));
|
|
266
|
+
if (recovered) return recovered;
|
|
267
|
+
}
|
|
268
|
+
throw new JudgeParseError("correctness-checker", raw);
|
|
269
|
+
}
|
|
270
|
+
/**
|
|
271
|
+
* Production `CorrectnessChecker` — one LLM call per matched artifact,
|
|
272
|
+
* deterministic (temperature 0), structured JSON out. Judges fulfilment
|
|
273
|
+
* only: a plan, a gesture, or a description of what should be done does not
|
|
274
|
+
* fulfil a requirement — the artifact must BE the deliverable.
|
|
275
|
+
*/
|
|
276
|
+
function createLlmCorrectnessChecker(chat, opts = {}) {
|
|
277
|
+
const model = opts.model ?? "claude-sonnet-4-6";
|
|
278
|
+
const maxContentChars = opts.maxContentChars ?? 8e3;
|
|
279
|
+
const maxAttempts = opts.maxAttempts ?? 2;
|
|
280
|
+
const costLedger = opts.costLedger ?? new CostLedger();
|
|
281
|
+
const sink = opts.rawSink;
|
|
282
|
+
const record = async (event) => {
|
|
283
|
+
try {
|
|
284
|
+
await sink?.record(event);
|
|
285
|
+
} catch {}
|
|
286
|
+
};
|
|
287
|
+
return async (requirement, content) => {
|
|
288
|
+
const request = {
|
|
289
|
+
model,
|
|
290
|
+
messages: [{
|
|
291
|
+
role: "system",
|
|
292
|
+
content: "You verify whether a produced work artifact actually fulfils a stated requirement. Judge fulfilment only — is the deliverable substantively present and on-point — not polish. A plan to do it later, a vague gesture, or a description of what should be done does NOT fulfil a requirement; the artifact must BE the deliverable. Respond with a single JSON object: {\"correct\": boolean, \"reason\": string (<= 30 words)}."
|
|
293
|
+
}, {
|
|
294
|
+
role: "user",
|
|
295
|
+
content: `Requirement: ${requirement.title}\n${requirement.category ? `Category: ${requirement.category}\n` : ""}\nProduced artifact:\n${content.slice(0, maxContentChars)}`
|
|
296
|
+
}],
|
|
297
|
+
temperature: 0,
|
|
298
|
+
maxTokens: 200
|
|
299
|
+
};
|
|
300
|
+
let lastErr;
|
|
301
|
+
for (let attempt = 0; attempt < maxAttempts; attempt++) {
|
|
302
|
+
const started = Date.now();
|
|
303
|
+
await record({
|
|
304
|
+
eventId: randomUUID(),
|
|
305
|
+
provider: chat.transport,
|
|
306
|
+
model,
|
|
307
|
+
endpoint: "/chat",
|
|
308
|
+
baseUrl: "",
|
|
309
|
+
attemptIndex: attempt,
|
|
310
|
+
direction: "request",
|
|
311
|
+
timestamp: started,
|
|
312
|
+
requestBody: request,
|
|
313
|
+
redactedFields: []
|
|
314
|
+
});
|
|
315
|
+
try {
|
|
316
|
+
const paid = await costLedger.runPaidCall({
|
|
317
|
+
channel: "verifier",
|
|
318
|
+
phase: opts.costPhase ?? "completion.correctness",
|
|
319
|
+
actor: "correctness-checker",
|
|
320
|
+
model,
|
|
321
|
+
maximumCharge: chat.maximumAttempts === void 0 ? void 0 : maximumChargeForLlmRequest(request, { maximumAttempts: chat.maximumAttempts }),
|
|
322
|
+
tags: {
|
|
323
|
+
...opts.costTags,
|
|
324
|
+
requirementId: requirement.reqId,
|
|
325
|
+
attempt: String(attempt)
|
|
326
|
+
},
|
|
327
|
+
signal: opts.signal,
|
|
328
|
+
execute: (signal, callId) => chat.chat(request, {
|
|
329
|
+
signal,
|
|
330
|
+
idempotencyKey: callId
|
|
331
|
+
}),
|
|
332
|
+
receipt: costReceiptFromLlm,
|
|
333
|
+
receiptFromError: costReceiptFromLlmError
|
|
334
|
+
});
|
|
335
|
+
if (!paid.succeeded) throw paid.error;
|
|
336
|
+
const resp = paid.value;
|
|
337
|
+
const raw = resp.content;
|
|
338
|
+
await record({
|
|
339
|
+
eventId: randomUUID(),
|
|
340
|
+
provider: chat.transport,
|
|
341
|
+
model,
|
|
342
|
+
endpoint: "/chat",
|
|
343
|
+
baseUrl: "",
|
|
344
|
+
attemptIndex: attempt,
|
|
345
|
+
direction: "response",
|
|
346
|
+
timestamp: Date.now(),
|
|
347
|
+
durationMs: Date.now() - started,
|
|
348
|
+
responseBody: resp,
|
|
349
|
+
redactedFields: []
|
|
350
|
+
});
|
|
351
|
+
return parseCorrectnessResponse(raw);
|
|
352
|
+
} catch (err) {
|
|
353
|
+
lastErr = err;
|
|
354
|
+
await record({
|
|
355
|
+
eventId: randomUUID(),
|
|
356
|
+
provider: chat.transport,
|
|
357
|
+
model,
|
|
358
|
+
endpoint: "/chat",
|
|
359
|
+
baseUrl: "",
|
|
360
|
+
attemptIndex: attempt,
|
|
361
|
+
direction: "error",
|
|
362
|
+
timestamp: Date.now(),
|
|
363
|
+
durationMs: Date.now() - started,
|
|
364
|
+
errorMessage: err instanceof Error ? err.message : String(err),
|
|
365
|
+
redactedFields: []
|
|
366
|
+
});
|
|
367
|
+
}
|
|
368
|
+
}
|
|
369
|
+
throw lastErr instanceof Error ? lastErr : new Error(String(lastErr));
|
|
370
|
+
};
|
|
371
|
+
}
|
|
372
|
+
/** Stopwords for requirement-title tokenization — drops the imperative verbs
|
|
373
|
+
* ('review', 'update', …) common to deliverable titles so recall keys on the
|
|
374
|
+
* substantive nouns, not the boilerplate ask. */
|
|
375
|
+
const TITLE_STOPWORDS = /* @__PURE__ */ new Set([
|
|
376
|
+
"the",
|
|
377
|
+
"a",
|
|
378
|
+
"an",
|
|
379
|
+
"and",
|
|
380
|
+
"or",
|
|
381
|
+
"for",
|
|
382
|
+
"to",
|
|
383
|
+
"of",
|
|
384
|
+
"in",
|
|
385
|
+
"on",
|
|
386
|
+
"with",
|
|
387
|
+
"review",
|
|
388
|
+
"update",
|
|
389
|
+
"new",
|
|
390
|
+
"proposed"
|
|
391
|
+
]);
|
|
392
|
+
/**
|
|
393
|
+
* Deterministic `CorrectnessChecker` — the no-LLM counterpart to
|
|
394
|
+
* `createLlmCorrectnessChecker`. A produced item fulfils a requirement when its
|
|
395
|
+
* content is substantive (≥ `minContentLength` chars) AND recalls ≥ `minRecall`
|
|
396
|
+
* of the requirement title's significant tokens. No network.
|
|
397
|
+
*
|
|
398
|
+
* Polarity-blind: token recall credits a negation that contains the
|
|
399
|
+
* requirement's tokens ("I will NOT produce the comparison" recalls every token
|
|
400
|
+
* of "produce the comparison"). The structural match stage is ALSO lexical, so
|
|
401
|
+
* pairing the two collapses to a single gameable gate. Use this only as an
|
|
402
|
+
* opt-in structural pre-filter or for tasks whose requirements have no polarity
|
|
403
|
+
* to invert; for produced-state grading the correctness checker MUST be semantic
|
|
404
|
+
* (`createLlmCorrectnessChecker`). See the anti-game fixtures in the test suite.
|
|
405
|
+
*/
|
|
406
|
+
function createTokenRecallChecker(opts = {}) {
|
|
407
|
+
const minRecall = opts.minRecall ?? .5;
|
|
408
|
+
const minLen = opts.minContentLength ?? 120;
|
|
409
|
+
return async (requirement, content) => {
|
|
410
|
+
const body = content.trim();
|
|
411
|
+
if (body.length < minLen) return {
|
|
412
|
+
correct: false,
|
|
413
|
+
reason: `content too thin (${body.length} chars) to be the deliverable`
|
|
414
|
+
};
|
|
415
|
+
const titleTokens = requirement.title.toLowerCase().split(/[^a-z0-9]+/).filter((t) => t.length > 2 && !TITLE_STOPWORDS.has(t));
|
|
416
|
+
if (titleTokens.length === 0) return {
|
|
417
|
+
correct: true,
|
|
418
|
+
reason: "requirement title has no significant tokens — structural match accepted"
|
|
419
|
+
};
|
|
420
|
+
const lower = body.toLowerCase();
|
|
421
|
+
const hits = titleTokens.filter((t) => lower.includes(t)).length;
|
|
422
|
+
return hits / titleTokens.length >= minRecall ? {
|
|
423
|
+
correct: true,
|
|
424
|
+
reason: `content recalls ${hits}/${titleTokens.length} requirement tokens`
|
|
425
|
+
} : {
|
|
426
|
+
correct: false,
|
|
427
|
+
reason: `content recalls only ${hits}/${titleTokens.length} requirement tokens`
|
|
428
|
+
};
|
|
429
|
+
};
|
|
430
|
+
}
|
|
431
|
+
//#endregion
|
|
432
|
+
//#region src/produced-state.ts
|
|
433
|
+
function artifactKind(mimeType) {
|
|
434
|
+
if (!mimeType) return "file";
|
|
435
|
+
if (mimeType.includes("json")) return "json";
|
|
436
|
+
if (mimeType.startsWith("text/")) return "text";
|
|
437
|
+
return "file";
|
|
438
|
+
}
|
|
439
|
+
/**
|
|
440
|
+
* Normalize a run's runtime event stream into `ProducedState`.
|
|
441
|
+
*
|
|
442
|
+
* Pure and total — unrecognized event types are skipped. `toolCalls` is
|
|
443
|
+
* deduplicated by name in first-seen order (completion cares about a tool's
|
|
444
|
+
* presence, not its call count). An artifact with neither a name nor a uri
|
|
445
|
+
* still yields an entry keyed by its `artifactId` so it is never silently
|
|
446
|
+
* dropped; an artifact with no `content` yields empty content, which the
|
|
447
|
+
* completion oracle's structural check then rejects on its own.
|
|
448
|
+
*/
|
|
449
|
+
function extractProducedState(events) {
|
|
450
|
+
const artifacts = [];
|
|
451
|
+
const proposals = [];
|
|
452
|
+
const toolCalls = [];
|
|
453
|
+
const seenTools = /* @__PURE__ */ new Set();
|
|
454
|
+
for (const ev of events) if (ev.type === "tool_call") {
|
|
455
|
+
const name = ev.toolName;
|
|
456
|
+
if (name && !seenTools.has(name)) {
|
|
457
|
+
seenTools.add(name);
|
|
458
|
+
toolCalls.push(name);
|
|
459
|
+
}
|
|
460
|
+
} else if (ev.type === "artifact") {
|
|
461
|
+
const a = ev;
|
|
462
|
+
artifacts.push({
|
|
463
|
+
kind: artifactKind(a.mimeType),
|
|
464
|
+
path: a.name ?? a.uri ?? a.artifactId,
|
|
465
|
+
content: a.content ?? ""
|
|
466
|
+
});
|
|
467
|
+
} else if (ev.type === "proposal_created") {
|
|
468
|
+
const p = ev;
|
|
469
|
+
proposals.push({
|
|
470
|
+
id: p.proposalId,
|
|
471
|
+
title: p.title,
|
|
472
|
+
status: p.status ?? "pending",
|
|
473
|
+
...p.content !== void 0 ? { content: p.content } : {}
|
|
474
|
+
});
|
|
475
|
+
}
|
|
476
|
+
return {
|
|
477
|
+
artifacts,
|
|
478
|
+
proposals,
|
|
479
|
+
toolCalls
|
|
480
|
+
};
|
|
481
|
+
}
|
|
482
|
+
//#endregion
|
|
483
|
+
//#region src/agent-profile.ts
|
|
484
|
+
/**
|
|
485
|
+
* The agentic coding harnesses an eval sweeps by default — the ones we care about
|
|
486
|
+
* ranking. This is the SINGLE source of that list; consumers import it instead of
|
|
487
|
+
* re-declaring their own (a re-declared list is how the fleet drifts). Pass an
|
|
488
|
+
* explicit `harnesses` (e.g. `harnessTypeSchema.options` for literally every known
|
|
489
|
+
* harness) to widen beyond these.
|
|
490
|
+
*/
|
|
491
|
+
const CODING_HARNESSES = [
|
|
492
|
+
"opencode",
|
|
493
|
+
"claude-code",
|
|
494
|
+
"codex",
|
|
495
|
+
"kimi-code"
|
|
496
|
+
];
|
|
497
|
+
/** Model sentinel for a vendor-locked harness that supports none of the swept models:
|
|
498
|
+
* it carries no provider prefix, so `harnessSupportsModel` accepts it and the harness
|
|
499
|
+
* resolves it to its own native default model at runtime (e.g. kimi-code → its Kimi
|
|
500
|
+
* model). Lets `expandProfileAxes` snap-instead-of-drop without a per-harness flagship
|
|
501
|
+
* table that would rot as router catalogs change. */
|
|
502
|
+
const HARNESS_NATIVE_MODEL = "default";
|
|
503
|
+
/**
|
|
504
|
+
* Expand a base profile across the harness × model matrix into the `AgentProfile[]`
|
|
505
|
+
* that `runProfileMatrix` / `selfImprove` score — the ONE place "which harnesses ×
|
|
506
|
+
* which models do we evaluate" lives, so no product hand-rolls its own harness list
|
|
507
|
+
* or column→profile mapping (the pattern that let those copies drift and silently
|
|
508
|
+
* break the harness pivot).
|
|
509
|
+
*
|
|
510
|
+
* Each cell clones `base`, sets `model.default`, and stamps `metadata.harness` +
|
|
511
|
+
* `metadata.harnessModel` (both hash-bearing, so every cell gets a distinct
|
|
512
|
+
* `agentProfileId` row and results join back by harness/model via {@link harnessAxisOf}
|
|
513
|
+
* with no hand-recomputed key). A vendor-locked harness snaps to its family's swept
|
|
514
|
+
* models — or its native default ({@link HARNESS_NATIVE_MODEL}) when it supports none —
|
|
515
|
+
* so every requested harness runs; `keepIncompatible` forces every pair verbatim.
|
|
516
|
+
*
|
|
517
|
+
* Omit `harnesses`/`models` to sweep the full default set — the "turn it on for
|
|
518
|
+
* everything we care about" switch, identical in shape whether one harness or all.
|
|
519
|
+
*/
|
|
520
|
+
function expandProfileAxes(spec) {
|
|
521
|
+
const harnesses = spec.harnesses ?? CODING_HARNESSES;
|
|
522
|
+
if (harnesses.length === 0) throw new ValidationError("expandProfileAxes: no harnesses to sweep");
|
|
523
|
+
const baseModel = spec.base.model?.default;
|
|
524
|
+
const models = spec.models ?? (baseModel ? [baseModel] : []);
|
|
525
|
+
if (models.length === 0) throw new ValidationError("expandProfileAxes: no models to sweep — base profile has no model.default and none were supplied");
|
|
526
|
+
const out = [];
|
|
527
|
+
const seen = /* @__PURE__ */ new Set();
|
|
528
|
+
for (const harness of harnesses) {
|
|
529
|
+
const supported = spec.keepIncompatible ? models : models.filter((model) => harnessSupportsModel(harness, model));
|
|
530
|
+
const effective = supported.length > 0 ? supported : [HARNESS_NATIVE_MODEL];
|
|
531
|
+
for (const model of effective) {
|
|
532
|
+
const profile = {
|
|
533
|
+
...spec.base,
|
|
534
|
+
name: `${spec.base.name ?? "agent"}/${harness}/${model}`,
|
|
535
|
+
model: {
|
|
536
|
+
...spec.base.model,
|
|
537
|
+
default: model
|
|
538
|
+
},
|
|
539
|
+
metadata: {
|
|
540
|
+
...spec.base.metadata ?? {},
|
|
541
|
+
harness,
|
|
542
|
+
harnessModel: model
|
|
543
|
+
}
|
|
544
|
+
};
|
|
545
|
+
const id = agentProfileId(profile);
|
|
546
|
+
if (seen.has(id)) continue;
|
|
547
|
+
seen.add(id);
|
|
548
|
+
out.push(profile);
|
|
549
|
+
}
|
|
550
|
+
}
|
|
551
|
+
if (out.length === 0) throw new ValidationError(`expandProfileAxes: produced no profiles (harnesses=[${harnesses.join(", ")}], models=[${models.join(", ")}]).`);
|
|
552
|
+
return out;
|
|
553
|
+
}
|
|
554
|
+
/**
|
|
555
|
+
* Read the (harness, model) a matrix cell ran under, off a profile or a result row's
|
|
556
|
+
* profile — the join-back for a `byHarness` pivot. Returns undefined when the profile
|
|
557
|
+
* wasn't produced by {@link expandProfileAxes}. Callers group `result.byProfile` by
|
|
558
|
+
* this instead of recomputing an id (recomputing the wrong key is what broke the pivot
|
|
559
|
+
* in the hand-rolled copies).
|
|
560
|
+
*/
|
|
561
|
+
function harnessAxisOf(profile) {
|
|
562
|
+
const m = profile.metadata;
|
|
563
|
+
const harness = m?.harness;
|
|
564
|
+
const model = m?.harnessModel;
|
|
565
|
+
if (typeof harness === "string" && typeof model === "string") return {
|
|
566
|
+
harness,
|
|
567
|
+
model
|
|
568
|
+
};
|
|
569
|
+
}
|
|
570
|
+
/**
|
|
571
|
+
* Collision-resistant, path-safe, human-readable profile id for eval artifacts.
|
|
572
|
+
* Scorecard joins still use `agentProfileHash`; this id is for run ids, matrix
|
|
573
|
+
* keys, and directory names where two profiles must not collapse onto one row.
|
|
574
|
+
* The suffix is the first 64 bits of the behaviour hash, enough for ordinary
|
|
575
|
+
* eval matrices while keeping filenames readable.
|
|
576
|
+
*/
|
|
577
|
+
function agentProfileId(profile) {
|
|
578
|
+
return `${pathSafeProfileLabel(agentProfileDisplayLabel(profile)) ?? "profile"}-${agentProfileHash(profile).slice(0, 16)}`;
|
|
579
|
+
}
|
|
580
|
+
/**
|
|
581
|
+
* Model snapshot used for `RunRecord.model`. Eval surfaces require a concrete
|
|
582
|
+
* model id because run records reject bare/missing model aliases.
|
|
583
|
+
*/
|
|
584
|
+
function agentProfileModelId(profile) {
|
|
585
|
+
const model = profile.model?.default?.trim();
|
|
586
|
+
if (!model) throw new ValidationError(`AgentProfile "${agentProfileDisplayLabel(profile) ?? "unnamed profile"}" has no model.default — cannot record eval run`);
|
|
587
|
+
return model;
|
|
588
|
+
}
|
|
589
|
+
function agentProfileDisplayLabel(profile) {
|
|
590
|
+
return profile.name?.trim() || profile.version?.trim() || void 0;
|
|
591
|
+
}
|
|
592
|
+
function pathSafeProfileLabel(label) {
|
|
593
|
+
return label?.trim().replace(/[^A-Za-z0-9._-]+/g, "-").replace(/-+/g, "-").replace(/^-|-$/g, "") || void 0;
|
|
594
|
+
}
|
|
595
|
+
function compact(input) {
|
|
596
|
+
const out = {};
|
|
597
|
+
for (const [key, value] of Object.entries(input)) if (value !== void 0) out[key] = value;
|
|
598
|
+
return out;
|
|
599
|
+
}
|
|
600
|
+
/**
|
|
601
|
+
* Deterministic behaviour identity for the canonical
|
|
602
|
+
* `@tangle-network/agent-interface` AgentProfile.
|
|
603
|
+
*
|
|
604
|
+
* `name` and `description` are labels and do not affect the hash. Profile
|
|
605
|
+
* `version`, prompt, model hints, tools, resources, hooks, modes, permissions,
|
|
606
|
+
* and extensions do affect the hash. Resource array order is hash-bearing
|
|
607
|
+
* because mount order can change agent behaviour. Undefined fields are treated
|
|
608
|
+
* as absent; explicit `null` fields remain hash-bearing.
|
|
609
|
+
*/
|
|
610
|
+
function agentProfileHash(profile) {
|
|
611
|
+
const model = agentProfileModelId(profile);
|
|
612
|
+
const behaviour = {
|
|
613
|
+
...profile,
|
|
614
|
+
name: void 0,
|
|
615
|
+
description: void 0,
|
|
616
|
+
tags: profile.tags ? [...profile.tags].sort() : void 0,
|
|
617
|
+
model: compact({
|
|
618
|
+
...profile.model,
|
|
619
|
+
default: model
|
|
620
|
+
})
|
|
621
|
+
};
|
|
622
|
+
return createHash("sha256").update(JSON.stringify(canonicalize(behaviour))).digest("hex");
|
|
623
|
+
}
|
|
624
|
+
//#endregion
|
|
625
|
+
//#region src/campaign/analyst-surface.ts
|
|
626
|
+
function surfaceToText(surface) {
|
|
627
|
+
if (typeof surface === "string") return surface;
|
|
628
|
+
throw new Error(`buildAnalystSurfaceDispatch: the analyst surface must be a string actorDescription, got a ${surface.kind}-tier surface. The analyst prompt is prompt-tier.`);
|
|
629
|
+
}
|
|
630
|
+
/**
|
|
631
|
+
* Build the `dispatchWithSurface(surface, scenario, ctx)` the improvement loop
|
|
632
|
+
* calls: run the analyst with `surface` as its actorDescription over the
|
|
633
|
+
* scenario's trace corpus and return its findings.
|
|
634
|
+
*/
|
|
635
|
+
function buildAnalystSurfaceDispatch(opts) {
|
|
636
|
+
const analyze = opts.analyze ?? analyzeTraces;
|
|
637
|
+
return async (surface, scenario, _ctx) => {
|
|
638
|
+
const actorDescription = surfaceToText(surface);
|
|
639
|
+
const res = await analyze({ question: scenario.question }, {
|
|
640
|
+
...opts.analystOptions,
|
|
641
|
+
actorDescription,
|
|
642
|
+
source: scenario.source
|
|
643
|
+
});
|
|
644
|
+
return {
|
|
645
|
+
answer: res.answer,
|
|
646
|
+
findings: res.findings,
|
|
647
|
+
actorPromptVersion: res.actorPromptVersion
|
|
648
|
+
};
|
|
649
|
+
};
|
|
650
|
+
}
|
|
651
|
+
/**
|
|
652
|
+
* Deterministic, ground-truth judge for analyst findings. Composite =
|
|
653
|
+
* recall of the scenario's `expectedFailureModes` (optionally blended with a
|
|
654
|
+
* precision term that penalizes findings tripping `forbiddenCues`). No LLM —
|
|
655
|
+
* the score is a function of the labels, so the analyst prompt is optimized
|
|
656
|
+
* toward surfacing real failures, not toward a judge it can flatter.
|
|
657
|
+
*/
|
|
658
|
+
function failureModeRecallJudge(opts = {}) {
|
|
659
|
+
const recallWeight = opts.recallWeight ?? .5;
|
|
660
|
+
return {
|
|
661
|
+
name: "failure-mode-recall",
|
|
662
|
+
dimensions: [{
|
|
663
|
+
key: "recall",
|
|
664
|
+
description: "fraction of ground-truth failure modes the analyst surfaced"
|
|
665
|
+
}, {
|
|
666
|
+
key: "precision",
|
|
667
|
+
description: "1 − share of findings that named a failure/tool/error absent from this corpus"
|
|
668
|
+
}],
|
|
669
|
+
appliesTo: (s) => s.kind === "analyst-surface",
|
|
670
|
+
score({ artifact, scenario }) {
|
|
671
|
+
const modes = scenario.expectedFailureModes;
|
|
672
|
+
if (modes.length === 0) throw new Error(`failureModeRecallJudge: scenario '${scenario.id}' has no expectedFailureModes — refusing to score (a vacuous 1.0 would corrupt the comparison)`);
|
|
673
|
+
const hay = artifact.findings.join("\n").toLowerCase();
|
|
674
|
+
const matched = modes.filter((m) => m.cues.some((c) => hay.includes(c.toLowerCase())));
|
|
675
|
+
const recall = matched.length / modes.length;
|
|
676
|
+
const forbidden = (scenario.forbiddenCues ?? []).map((c) => c.toLowerCase());
|
|
677
|
+
let precision = 1;
|
|
678
|
+
let hallucinated = 0;
|
|
679
|
+
if (forbidden.length > 0) {
|
|
680
|
+
const denom = Math.max(1, artifact.findings.length);
|
|
681
|
+
hallucinated = artifact.findings.filter((f) => forbidden.some((c) => f.toLowerCase().includes(c))).length;
|
|
682
|
+
precision = 1 - hallucinated / denom;
|
|
683
|
+
}
|
|
684
|
+
const composite = forbidden.length > 0 ? recallWeight * recall + (1 - recallWeight) * precision : recall;
|
|
685
|
+
const missed = modes.filter((m) => !matched.includes(m)).map((m) => m.id);
|
|
686
|
+
const notes = `matched ${matched.length}/${modes.length} failure modes` + (missed.length ? `; missed [${missed.join(", ")}]` : "") + (hallucinated ? `; ${hallucinated} out-of-corpus finding(s)` : "");
|
|
687
|
+
return {
|
|
688
|
+
dimensions: {
|
|
689
|
+
recall,
|
|
690
|
+
precision
|
|
691
|
+
},
|
|
692
|
+
composite,
|
|
693
|
+
notes
|
|
694
|
+
};
|
|
695
|
+
}
|
|
696
|
+
};
|
|
697
|
+
}
|
|
698
|
+
//#endregion
|
|
699
|
+
//#region src/campaign/cross-surface-context.ts
|
|
700
|
+
function validateCrossSurfaceInput(input) {
|
|
701
|
+
assertNonEmptyUnique(input.taskOrder, "taskOrder");
|
|
702
|
+
assertNonEmptyUnique(input.componentOrder, "componentOrder");
|
|
703
|
+
if (input.componentOrder.length < 2) throw new ValidationError("analyzeCrossSurfaceInteractions: componentOrder must contain at least two surfaces");
|
|
704
|
+
assertNonEmptyUnique(input.candidateOrder, "candidateOrder");
|
|
705
|
+
assertNonEmptyUnique(input.costMetricOrder, "costMetricOrder");
|
|
706
|
+
if (input.costMetricOrder.includes("score")) throw new ValidationError(`analyzeCrossSurfaceInteractions: costMetricOrder cannot contain reserved metric 'score'`);
|
|
707
|
+
validateBootstrap(input);
|
|
708
|
+
validateSelection(input);
|
|
709
|
+
const componentById = indexUnique(input.components, (component) => component.componentId, "component");
|
|
710
|
+
assertExactSet(input.componentOrder, componentById.keys(), "componentOrder", "components");
|
|
711
|
+
const surfaces = /* @__PURE__ */ new Set();
|
|
712
|
+
for (const component of componentById.values()) {
|
|
713
|
+
assertNonEmpty(component.componentId, "componentId");
|
|
714
|
+
assertNonEmpty(component.surfaceId, `surfaceId for component '${component.componentId}'`);
|
|
715
|
+
if (surfaces.has(component.surfaceId)) throw new ValidationError(`analyzeCrossSurfaceInteractions: surfaceId '${component.surfaceId}' has more than one component; the interaction stage requires one independently selected finalist per surface`);
|
|
716
|
+
surfaces.add(component.surfaceId);
|
|
717
|
+
if (typeof component.bestSingleEligible !== "boolean") throw new ValidationError(`analyzeCrossSurfaceInteractions: component '${component.componentId}' must declare bestSingleEligible`);
|
|
718
|
+
}
|
|
719
|
+
const componentIndex = new Map(input.componentOrder.map((id, index) => [id, index]));
|
|
720
|
+
const candidateById = indexUnique(input.candidates, (candidate) => candidate.candidateId, "candidate");
|
|
721
|
+
assertExactSet(input.candidateOrder, candidateById.keys(), "candidateOrder", "candidates");
|
|
722
|
+
if (!candidateById.has(input.baselineCandidateId)) throw new ValidationError(`analyzeCrossSurfaceInteractions: unknown baselineCandidateId '${input.baselineCandidateId}'`);
|
|
723
|
+
const candidateByComponents = /* @__PURE__ */ new Map();
|
|
724
|
+
const singleByComponent = /* @__PURE__ */ new Map();
|
|
725
|
+
for (const candidate of candidateById.values()) {
|
|
726
|
+
validateCandidate(candidate, componentById, componentIndex, input.baselineCandidateId);
|
|
727
|
+
const key = crossSurfaceComponentSetKey(candidate.componentIds);
|
|
728
|
+
const duplicate = candidateByComponents.get(key);
|
|
729
|
+
if (duplicate) throw new ValidationError(`analyzeCrossSurfaceInteractions: candidates '${duplicate.candidateId}' and '${candidate.candidateId}' materialize the same component set`);
|
|
730
|
+
candidateByComponents.set(key, candidate);
|
|
731
|
+
if (candidate.componentIds.length === 1) singleByComponent.set(candidate.componentIds[0], candidate);
|
|
732
|
+
}
|
|
733
|
+
const baseline = candidateById.get(input.baselineCandidateId);
|
|
734
|
+
if (baseline.componentIds.length !== 0) throw new ValidationError(`analyzeCrossSurfaceInteractions: baseline '${baseline.candidateId}' must have zero components`);
|
|
735
|
+
for (const componentId of input.componentOrder) if (!singleByComponent.has(componentId)) throw new ValidationError(`analyzeCrossSurfaceInteractions: no single-surface candidate for component '${componentId}'`);
|
|
736
|
+
for (let left = 0; left < input.componentOrder.length; left++) for (let right = left + 1; right < input.componentOrder.length; right++) {
|
|
737
|
+
const ids = [input.componentOrder[left], input.componentOrder[right]];
|
|
738
|
+
if (!candidateByComponents.has(crossSurfaceComponentSetKey(ids))) throw new ValidationError(`analyzeCrossSurfaceInteractions: missing pair candidate for components [${ids.join(", ")}]`);
|
|
739
|
+
}
|
|
740
|
+
const rowsByCandidate = /* @__PURE__ */ new Map();
|
|
741
|
+
const taskIds = new Set(input.taskOrder);
|
|
742
|
+
for (const row of input.rows) {
|
|
743
|
+
const candidate = candidateById.get(row.candidateId);
|
|
744
|
+
if (!candidate) throw new ValidationError(`analyzeCrossSurfaceInteractions: row for task '${row.taskId}' names unknown candidate '${row.candidateId}'`);
|
|
745
|
+
if (!taskIds.has(row.taskId)) throw new ValidationError(`analyzeCrossSurfaceInteractions: row for candidate '${row.candidateId}' names task '${row.taskId}' outside the declared taskOrder`);
|
|
746
|
+
validateRow(input, row, candidate);
|
|
747
|
+
const byTask = rowsByCandidate.get(row.candidateId) ?? /* @__PURE__ */ new Map();
|
|
748
|
+
if (byTask.has(row.taskId)) throw new ValidationError(`analyzeCrossSurfaceInteractions: duplicate row for candidate '${row.candidateId}' and task '${row.taskId}'`);
|
|
749
|
+
byTask.set(row.taskId, row);
|
|
750
|
+
rowsByCandidate.set(row.candidateId, byTask);
|
|
751
|
+
}
|
|
752
|
+
for (const candidateId of input.candidateOrder) {
|
|
753
|
+
const byTask = rowsByCandidate.get(candidateId);
|
|
754
|
+
const missing = input.taskOrder.filter((taskId) => !byTask?.has(taskId));
|
|
755
|
+
if (missing.length > 0) throw new ValidationError(`analyzeCrossSurfaceInteractions: candidate '${candidateId}' is missing declared task row(s) [${missing.join(", ")}]; encode failed attempts explicitly instead of changing the task axis`);
|
|
756
|
+
}
|
|
757
|
+
const expectedRows = input.candidateOrder.length * input.taskOrder.length;
|
|
758
|
+
if (input.rows.length !== expectedRows) throw new ValidationError(`analyzeCrossSurfaceInteractions: expected exactly ${expectedRows} candidate × task rows, got ${input.rows.length}`);
|
|
759
|
+
return {
|
|
760
|
+
input,
|
|
761
|
+
components: input.componentOrder.map((id) => componentById.get(id)),
|
|
762
|
+
candidates: input.candidateOrder.map((id) => candidateById.get(id)),
|
|
763
|
+
componentById,
|
|
764
|
+
candidateById,
|
|
765
|
+
candidateByComponents,
|
|
766
|
+
singleByComponent,
|
|
767
|
+
rowsByCandidate,
|
|
768
|
+
componentIndex,
|
|
769
|
+
candidateIndex: new Map(input.candidateOrder.map((id, index) => [id, index]))
|
|
770
|
+
};
|
|
771
|
+
}
|
|
772
|
+
function crossSurfaceRowsFor(context, candidateId) {
|
|
773
|
+
return context.input.taskOrder.map((taskId) => crossSurfaceRowFor(context, candidateId, taskId));
|
|
774
|
+
}
|
|
775
|
+
function crossSurfaceRowFor(context, candidateId, taskId) {
|
|
776
|
+
return context.rowsByCandidate.get(candidateId).get(taskId);
|
|
777
|
+
}
|
|
778
|
+
function canonicalCrossSurfaceComponents(context, componentIds) {
|
|
779
|
+
return [...componentIds].sort((left, right) => context.componentIndex.get(left) - context.componentIndex.get(right));
|
|
780
|
+
}
|
|
781
|
+
function crossSurfaceComponentSetKey(componentIds) {
|
|
782
|
+
return JSON.stringify(componentIds);
|
|
783
|
+
}
|
|
784
|
+
function validateBootstrap(input) {
|
|
785
|
+
const { seed, resamples, confidence } = input.bootstrap;
|
|
786
|
+
if (!Number.isInteger(seed)) throw new ValidationError(`analyzeCrossSurfaceInteractions: bootstrap.seed must be an integer`);
|
|
787
|
+
if (!Number.isInteger(resamples) || resamples <= 0) throw new ValidationError(`analyzeCrossSurfaceInteractions: bootstrap.resamples must be a positive integer`);
|
|
788
|
+
if (!Number.isFinite(confidence) || confidence <= 0 || confidence >= 1) throw new ValidationError(`analyzeCrossSurfaceInteractions: bootstrap.confidence must be in (0,1)`);
|
|
789
|
+
}
|
|
790
|
+
function validateSelection(input) {
|
|
791
|
+
const policy = input.selection;
|
|
792
|
+
for (const [name, value] of [["minimumFiringTasks", policy.minimumFiringTasks], ["minimumEffectTasks", policy.minimumEffectTasks]]) if (!Number.isInteger(value) || value < 0 || value > input.taskOrder.length) throw new ValidationError(`analyzeCrossSurfaceInteractions: selection.${name} must be an integer in [0,${input.taskOrder.length}]`);
|
|
793
|
+
if (!Number.isInteger(policy.minimumBundleComponents) || policy.minimumBundleComponents < 2 || policy.minimumBundleComponents > input.componentOrder.length) throw new ValidationError(`analyzeCrossSurfaceInteractions: selection.minimumBundleComponents must be an integer in [2,${input.componentOrder.length}]`);
|
|
794
|
+
for (const [metric, limit] of Object.entries(policy.maximumMedianCostRatioToBaseline)) {
|
|
795
|
+
if (!input.costMetricOrder.includes(metric)) throw new ValidationError(`analyzeCrossSurfaceInteractions: cost limit names undeclared metric '${metric}'`);
|
|
796
|
+
if (!Number.isFinite(limit) || limit <= 0) throw new ValidationError(`analyzeCrossSurfaceInteractions: cost ratio limit for '${metric}' must be positive and finite`);
|
|
797
|
+
}
|
|
798
|
+
}
|
|
799
|
+
function validateCandidate(candidate, componentById, componentIndex, baselineCandidateId) {
|
|
800
|
+
assertNonEmpty(candidate.candidateId, "candidateId");
|
|
801
|
+
assertNonEmpty(candidate.contentHash, `contentHash for candidate '${candidate.candidateId}'`);
|
|
802
|
+
if (!Number.isInteger(candidate.artifactBytes) || candidate.artifactBytes < 0) throw new ValidationError(`analyzeCrossSurfaceInteractions: artifactBytes for candidate '${candidate.candidateId}' must be a non-negative integer`);
|
|
803
|
+
assertUnique$1(candidate.componentIds, `componentIds for candidate '${candidate.candidateId}'`);
|
|
804
|
+
for (const componentId of candidate.componentIds) if (!componentById.has(componentId)) throw new ValidationError(`analyzeCrossSurfaceInteractions: candidate '${candidate.candidateId}' names unknown component '${componentId}'`);
|
|
805
|
+
const canonical = [...candidate.componentIds].sort((left, right) => componentIndex.get(left) - componentIndex.get(right));
|
|
806
|
+
if (!arraysEqual(candidate.componentIds, canonical)) throw new ValidationError(`analyzeCrossSurfaceInteractions: candidate '${candidate.candidateId}' components must follow componentOrder`);
|
|
807
|
+
if (candidate.candidateId !== baselineCandidateId && candidate.componentIds.length === 0) throw new ValidationError(`analyzeCrossSurfaceInteractions: only baseline '${baselineCandidateId}' may have zero components`);
|
|
808
|
+
}
|
|
809
|
+
function validateRow(input, row, candidate) {
|
|
810
|
+
if (!arraysEqual(row.componentIds, candidate.componentIds)) throw new ValidationError(`analyzeCrossSurfaceInteractions: row '${row.candidateId}/${row.taskId}' componentIds do not match its candidate`);
|
|
811
|
+
if (![
|
|
812
|
+
"complete",
|
|
813
|
+
"missing",
|
|
814
|
+
"invalid"
|
|
815
|
+
].includes(row.completeness)) throw new ValidationError(`analyzeCrossSurfaceInteractions: row '${row.candidateId}/${row.taskId}' has unknown completeness '${String(row.completeness)}'`);
|
|
816
|
+
if (row.completeness === "complete") {
|
|
817
|
+
if (typeof row.pass !== "boolean" || !Number.isFinite(row.score)) throw new ValidationError(`analyzeCrossSurfaceInteractions: complete row '${row.candidateId}/${row.taskId}' requires boolean pass and finite score`);
|
|
818
|
+
if (row.rejectReason !== null) throw new ValidationError(`analyzeCrossSurfaceInteractions: complete row '${row.candidateId}/${row.taskId}' cannot carry rejectReason`);
|
|
819
|
+
} else {
|
|
820
|
+
if (row.pass !== null || row.score !== null) throw new ValidationError(`analyzeCrossSurfaceInteractions: ${row.completeness} row '${row.candidateId}/${row.taskId}' must use null pass and score`);
|
|
821
|
+
if (typeof row.rejectReason !== "string" || row.rejectReason.trim() === "") throw new ValidationError(`analyzeCrossSurfaceInteractions: ${row.completeness} row '${row.candidateId}/${row.taskId}' requires rejectReason`);
|
|
822
|
+
}
|
|
823
|
+
assertExactSet(input.costMetricOrder, Object.keys(row.cost), `cost keys for row '${row.candidateId}/${row.taskId}'`, "costMetricOrder");
|
|
824
|
+
for (const metric of input.costMetricOrder) {
|
|
825
|
+
const value = row.cost[metric];
|
|
826
|
+
if (value === null || value === void 0 || !Number.isFinite(value) || value < 0) throw new ValidationError(`analyzeCrossSurfaceInteractions: unknown or invalid cost '${metric}' on row '${row.candidateId}/${row.taskId}'; every attempt must report a non-negative finite value`);
|
|
827
|
+
}
|
|
828
|
+
const evidenceByComponent = indexUnique(row.componentEvidence, (evidence) => evidence.componentId, `componentEvidence on row '${row.candidateId}/${row.taskId}'`);
|
|
829
|
+
assertExactSet(candidate.componentIds, evidenceByComponent.keys(), `componentEvidence on row '${row.candidateId}/${row.taskId}'`, "candidate components");
|
|
830
|
+
for (const evidence of evidenceByComponent.values()) {
|
|
831
|
+
assertTriState(evidence.fired, "fired", row);
|
|
832
|
+
assertTriState(evidence.effectObserved, "effectObserved", row);
|
|
833
|
+
if (evidence.effectObserved === true && evidence.fired !== true) throw new ValidationError(`analyzeCrossSurfaceInteractions: effectObserved=true requires fired=true for component '${evidence.componentId}' on row '${row.candidateId}/${row.taskId}'`);
|
|
834
|
+
}
|
|
835
|
+
}
|
|
836
|
+
function assertTriState(value, field, row) {
|
|
837
|
+
if (value !== true && value !== false && value !== null) throw new ValidationError(`analyzeCrossSurfaceInteractions: ${field} on row '${row.candidateId}/${row.taskId}' must be boolean or null`);
|
|
838
|
+
}
|
|
839
|
+
function indexUnique(items, id, label) {
|
|
840
|
+
const result = /* @__PURE__ */ new Map();
|
|
841
|
+
for (const item of items) {
|
|
842
|
+
const key = id(item);
|
|
843
|
+
assertNonEmpty(key, `${label} id`);
|
|
844
|
+
if (result.has(key)) throw new ValidationError(`analyzeCrossSurfaceInteractions: duplicate ${label} id '${key}'`);
|
|
845
|
+
result.set(key, item);
|
|
846
|
+
}
|
|
847
|
+
return result;
|
|
848
|
+
}
|
|
849
|
+
function assertNonEmptyUnique(values, label) {
|
|
850
|
+
if (values.length === 0) throw new ValidationError(`analyzeCrossSurfaceInteractions: ${label} is empty`);
|
|
851
|
+
assertUnique$1(values, label);
|
|
852
|
+
for (const value of values) assertNonEmpty(value, label);
|
|
853
|
+
}
|
|
854
|
+
function assertUnique$1(values, label) {
|
|
855
|
+
if (new Set(values).size !== values.length) throw new ValidationError(`analyzeCrossSurfaceInteractions: ${label} contains duplicates`);
|
|
856
|
+
}
|
|
857
|
+
function assertNonEmpty(value, label) {
|
|
858
|
+
if (typeof value !== "string" || value.trim() === "") throw new ValidationError(`analyzeCrossSurfaceInteractions: ${label} must be a non-empty string`);
|
|
859
|
+
}
|
|
860
|
+
function assertExactSet(expected, actualIterable, actualLabel, expectedLabel) {
|
|
861
|
+
const actual = [...actualIterable];
|
|
862
|
+
const expectedSet = new Set(expected);
|
|
863
|
+
const actualSet = new Set(actual);
|
|
864
|
+
const missing = expected.filter((value) => !actualSet.has(value));
|
|
865
|
+
const extra = actual.filter((value) => !expectedSet.has(value));
|
|
866
|
+
if (missing.length > 0 || extra.length > 0 || actual.length !== expected.length) throw new ValidationError(`analyzeCrossSurfaceInteractions: ${actualLabel} does not match ${expectedLabel}; missing=[${missing.join(", ")}], extra=[${extra.join(", ")}]`);
|
|
867
|
+
}
|
|
868
|
+
function arraysEqual(left, right) {
|
|
869
|
+
return left.length === right.length && left.every((value, index) => value === right[index]);
|
|
870
|
+
}
|
|
871
|
+
//#endregion
|
|
872
|
+
//#region src/campaign/cross-surface-interaction.ts
|
|
873
|
+
/**
|
|
874
|
+
* Task-paired comparison and deterministic selection for independently
|
|
875
|
+
* proposed agent surfaces. Unlike factorial cell-mean attribution, every
|
|
876
|
+
* benefit, regression, and interaction remains attached to its task row.
|
|
877
|
+
*/
|
|
878
|
+
/**
|
|
879
|
+
* Build the complete cross-surface evidence matrix and derive all three frozen
|
|
880
|
+
* candidates. The task/candidate/component orders are part of the input so
|
|
881
|
+
* neither insertion order nor an after-the-fact tie-break can change a result.
|
|
882
|
+
*/
|
|
883
|
+
function analyzeCrossSurfaceInteractions(input) {
|
|
884
|
+
const context = validateCrossSurfaceInput(input);
|
|
885
|
+
const baseline = context.candidateById.get(input.baselineCandidateId);
|
|
886
|
+
const preliminary = context.candidates.map((candidate) => summarizeCandidate(context, candidate, baseline));
|
|
887
|
+
const baselineSummary = preliminary.find((summary) => summary.candidate.candidateId === input.baselineCandidateId);
|
|
888
|
+
const summaries = preliminary.map((summary) => summary.candidate.candidateId === input.baselineCandidateId ? summary : {
|
|
889
|
+
...summary,
|
|
890
|
+
eligibility: candidateEligibility(context, summary, baselineSummary)
|
|
891
|
+
});
|
|
892
|
+
const summaryById = new Map(summaries.map((summary) => [summary.candidate.candidateId, summary]));
|
|
893
|
+
const eligibleSingles = rankedEligibleSingles(context, summaryById);
|
|
894
|
+
const interactionReadySingles = rankedInteractionReadySingles(context, summaryById);
|
|
895
|
+
const pairwise = buildPairwise(context, summaryById, interactionReadySingles, baselineSummary);
|
|
896
|
+
const selections = selectCandidates(context, summaryById, eligibleSingles, interactionReadySingles, pairwise);
|
|
897
|
+
const rows = context.candidates.flatMap((candidate) => input.taskOrder.map((taskId) => context.rowsByCandidate.get(candidate.candidateId).get(taskId)));
|
|
898
|
+
return {
|
|
899
|
+
taskIds: [...input.taskOrder],
|
|
900
|
+
componentIds: [...input.componentOrder],
|
|
901
|
+
candidateIds: [...input.candidateOrder],
|
|
902
|
+
costMetrics: [...input.costMetricOrder],
|
|
903
|
+
rows,
|
|
904
|
+
missingAttempts: rows.filter((row) => row.completeness === "missing"),
|
|
905
|
+
invalidAttempts: rows.filter((row) => row.completeness === "invalid"),
|
|
906
|
+
candidates: summaries,
|
|
907
|
+
pairwise,
|
|
908
|
+
selections
|
|
909
|
+
};
|
|
910
|
+
}
|
|
911
|
+
function summarizeCandidate(context, candidate, baseline) {
|
|
912
|
+
const rows = crossSurfaceRowsFor(context, candidate.candidateId);
|
|
913
|
+
const outcome = summarizeOutcome(context, rows, crossSurfaceRowsFor(context, baseline.candidateId), candidate.candidateId === baseline.candidateId);
|
|
914
|
+
const completeScores = rows.filter((row) => row.completeness === "complete").map((row) => row.score);
|
|
915
|
+
const costs = Object.fromEntries(context.input.costMetricOrder.map((metric) => [metric, distribution(rows.map((row) => row.cost[metric]))]));
|
|
916
|
+
return {
|
|
917
|
+
candidate,
|
|
918
|
+
outcome,
|
|
919
|
+
score: completeScores.length === 0 ? null : distribution(completeScores),
|
|
920
|
+
costs,
|
|
921
|
+
firing: summarizeCandidateEvidence(rows, candidate.componentIds, "fired"),
|
|
922
|
+
effect: summarizeCandidateEvidence(rows, candidate.componentIds, "effectObserved"),
|
|
923
|
+
comparisonToBaseline: candidate.candidateId === baseline.candidateId ? null : pairedComparison(context, baseline.candidateId, candidate.candidateId),
|
|
924
|
+
eligibility: null
|
|
925
|
+
};
|
|
926
|
+
}
|
|
927
|
+
function summarizeOutcome(context, rows, baselineRows, isBaseline) {
|
|
928
|
+
const resolvedTaskIds = [];
|
|
929
|
+
const failedTaskIds = [];
|
|
930
|
+
const missingTaskIds = [];
|
|
931
|
+
const invalidTaskIds = [];
|
|
932
|
+
const benefitTaskIds = [];
|
|
933
|
+
const regressionTaskIds = [];
|
|
934
|
+
const comparisonMissingTaskIds = [];
|
|
935
|
+
for (let index = 0; index < context.input.taskOrder.length; index++) {
|
|
936
|
+
const row = rows[index];
|
|
937
|
+
const baseline = baselineRows[index];
|
|
938
|
+
if (row.completeness === "missing") missingTaskIds.push(row.taskId);
|
|
939
|
+
else if (row.completeness === "invalid") invalidTaskIds.push(row.taskId);
|
|
940
|
+
else if (row.pass) resolvedTaskIds.push(row.taskId);
|
|
941
|
+
else failedTaskIds.push(row.taskId);
|
|
942
|
+
if (isBaseline) continue;
|
|
943
|
+
if (row.completeness !== "complete" || baseline.completeness !== "complete") comparisonMissingTaskIds.push(row.taskId);
|
|
944
|
+
else if (row.pass && !baseline.pass) benefitTaskIds.push(row.taskId);
|
|
945
|
+
else if (!row.pass && baseline.pass) regressionTaskIds.push(row.taskId);
|
|
946
|
+
}
|
|
947
|
+
return {
|
|
948
|
+
resolvedTaskIds,
|
|
949
|
+
failedTaskIds,
|
|
950
|
+
missingTaskIds,
|
|
951
|
+
invalidTaskIds,
|
|
952
|
+
benefitTaskIds,
|
|
953
|
+
regressionTaskIds,
|
|
954
|
+
comparisonMissingTaskIds,
|
|
955
|
+
netBenefit: benefitTaskIds.length - regressionTaskIds.length
|
|
956
|
+
};
|
|
957
|
+
}
|
|
958
|
+
function summarizeCandidateEvidence(rows, componentIds, field) {
|
|
959
|
+
if (componentIds.length === 0) return {
|
|
960
|
+
byComponent: [],
|
|
961
|
+
allObservedTaskIds: [],
|
|
962
|
+
someObservedTaskIds: [],
|
|
963
|
+
noneObservedTaskIds: [],
|
|
964
|
+
unobservedTaskIds: []
|
|
965
|
+
};
|
|
966
|
+
const byComponent = componentIds.map((componentId) => {
|
|
967
|
+
const observedTaskIds = [];
|
|
968
|
+
const notObservedTaskIds = [];
|
|
969
|
+
const unobservedTaskIds = [];
|
|
970
|
+
for (const row of rows) {
|
|
971
|
+
const value = evidenceValue(row, componentId, field);
|
|
972
|
+
if (value === true) observedTaskIds.push(row.taskId);
|
|
973
|
+
else if (value === false) notObservedTaskIds.push(row.taskId);
|
|
974
|
+
else unobservedTaskIds.push(row.taskId);
|
|
975
|
+
}
|
|
976
|
+
return {
|
|
977
|
+
componentId,
|
|
978
|
+
observedTaskIds,
|
|
979
|
+
notObservedTaskIds,
|
|
980
|
+
unobservedTaskIds
|
|
981
|
+
};
|
|
982
|
+
});
|
|
983
|
+
const allObservedTaskIds = [];
|
|
984
|
+
const someObservedTaskIds = [];
|
|
985
|
+
const noneObservedTaskIds = [];
|
|
986
|
+
const unobservedTaskIds = [];
|
|
987
|
+
for (const row of rows) {
|
|
988
|
+
const values = componentIds.map((componentId) => evidenceValue(row, componentId, field));
|
|
989
|
+
if (values.some((value) => value === null)) unobservedTaskIds.push(row.taskId);
|
|
990
|
+
else if (values.every(Boolean)) allObservedTaskIds.push(row.taskId);
|
|
991
|
+
else if (values.some(Boolean)) someObservedTaskIds.push(row.taskId);
|
|
992
|
+
else noneObservedTaskIds.push(row.taskId);
|
|
993
|
+
}
|
|
994
|
+
return {
|
|
995
|
+
byComponent,
|
|
996
|
+
allObservedTaskIds,
|
|
997
|
+
someObservedTaskIds,
|
|
998
|
+
noneObservedTaskIds,
|
|
999
|
+
unobservedTaskIds
|
|
1000
|
+
};
|
|
1001
|
+
}
|
|
1002
|
+
function candidateEligibility(context, summary, baseline) {
|
|
1003
|
+
const reasons = [];
|
|
1004
|
+
if (summary.outcome.missingTaskIds.length > 0) reasons.push("missing_attempt");
|
|
1005
|
+
if (summary.outcome.invalidTaskIds.length > 0) reasons.push("invalid_attempt");
|
|
1006
|
+
if (summary.outcome.comparisonMissingTaskIds.length > 0) reasons.push("baseline_outcome_missing");
|
|
1007
|
+
if (summary.outcome.benefitTaskIds.length <= summary.outcome.regressionTaskIds.length) reasons.push("benefit_not_greater_than_regression");
|
|
1008
|
+
appendEvidenceEligibilityReasons(reasons, summary.firing, context.input.selection.minimumFiringTasks, context.input.selection.requireObservedFiring, "firing_below_minimum", "firing_unobserved");
|
|
1009
|
+
appendEvidenceEligibilityReasons(reasons, summary.effect, context.input.selection.minimumEffectTasks, context.input.selection.requireObservedEffect, "effect_below_minimum", "effect_unobserved");
|
|
1010
|
+
if (!withinCostLimits(context, summary, baseline)) reasons.push("cost_limit_exceeded");
|
|
1011
|
+
return {
|
|
1012
|
+
eligible: reasons.length === 0,
|
|
1013
|
+
reasons: unique(reasons)
|
|
1014
|
+
};
|
|
1015
|
+
}
|
|
1016
|
+
function appendEvidenceEligibilityReasons(reasons, evidence, minimum, requireObserved, belowReason, unobservedReason) {
|
|
1017
|
+
if (evidence.byComponent.some((component) => component.observedTaskIds.length < minimum)) reasons.push(belowReason);
|
|
1018
|
+
if (requireObserved && evidence.byComponent.some((component) => component.unobservedTaskIds.length > 0)) reasons.push(unobservedReason);
|
|
1019
|
+
}
|
|
1020
|
+
function rankedEligibleSingles(context, summaryById) {
|
|
1021
|
+
return [...context.singleByComponent.values()].map((candidate) => summaryById.get(candidate.candidateId)).filter((summary) => summary.eligibility?.eligible).sort((left, right) => compareSingleSummaries(context, left, right));
|
|
1022
|
+
}
|
|
1023
|
+
/**
|
|
1024
|
+
* A neutral constituent may seed a composition when it has complete, bounded,
|
|
1025
|
+
* observed evidence and causes no baseline regression. Individual benefit is
|
|
1026
|
+
* deliberately left to the best-single arm rather than used as a pair gate.
|
|
1027
|
+
*/
|
|
1028
|
+
function rankedInteractionReadySingles(context, summaryById) {
|
|
1029
|
+
return [...context.singleByComponent.values()].map((candidate) => summaryById.get(candidate.candidateId)).filter((summary) => summary.outcome.regressionTaskIds.length === 0 && summary.eligibility?.reasons.every((reason) => reason === "benefit_not_greater_than_regression")).sort((left, right) => compareSingleSummaries(context, left, right));
|
|
1030
|
+
}
|
|
1031
|
+
function compareSingleSummaries(context, left, right) {
|
|
1032
|
+
const byNetBenefit = right.outcome.netBenefit - left.outcome.netBenefit;
|
|
1033
|
+
if (byNetBenefit !== 0) return byNetBenefit;
|
|
1034
|
+
const byBenefit = right.outcome.benefitTaskIds.length - left.outcome.benefitTaskIds.length;
|
|
1035
|
+
if (byBenefit !== 0) return byBenefit;
|
|
1036
|
+
const byRegression = left.outcome.regressionTaskIds.length - right.outcome.regressionTaskIds.length;
|
|
1037
|
+
if (byRegression !== 0) return byRegression;
|
|
1038
|
+
for (const metric of context.input.costMetricOrder) {
|
|
1039
|
+
const byCost = left.costs[metric].median - right.costs[metric].median;
|
|
1040
|
+
if (byCost !== 0) return byCost;
|
|
1041
|
+
}
|
|
1042
|
+
const byBytes = left.candidate.artifactBytes - right.candidate.artifactBytes;
|
|
1043
|
+
if (byBytes !== 0) return byBytes;
|
|
1044
|
+
return context.candidateIndex.get(left.candidate.candidateId) - context.candidateIndex.get(right.candidate.candidateId);
|
|
1045
|
+
}
|
|
1046
|
+
function buildPairwise(context, summaryById, interactionReadySingles, baseline) {
|
|
1047
|
+
const interactionReadyIds = new Set(interactionReadySingles.map((summary) => summary.candidate.candidateId));
|
|
1048
|
+
const entries = [];
|
|
1049
|
+
for (let leftIndex = 0; leftIndex < context.components.length; leftIndex++) for (let rightIndex = leftIndex + 1; rightIndex < context.components.length; rightIndex++) {
|
|
1050
|
+
const leftComponent = context.components[leftIndex];
|
|
1051
|
+
const rightComponent = context.components[rightIndex];
|
|
1052
|
+
const leftSingle = context.singleByComponent.get(leftComponent.componentId);
|
|
1053
|
+
const rightSingle = context.singleByComponent.get(rightComponent.componentId);
|
|
1054
|
+
const pair = context.candidateByComponents.get(crossSurfaceComponentSetKey([leftComponent.componentId, rightComponent.componentId]));
|
|
1055
|
+
const pairSummary = summaryById.get(pair.candidateId);
|
|
1056
|
+
const leftSummary = summaryById.get(leftSingle.candidateId);
|
|
1057
|
+
const rightSummary = summaryById.get(rightSingle.candidateId);
|
|
1058
|
+
const synergyTaskIds = [];
|
|
1059
|
+
const interferenceTaskIds = [];
|
|
1060
|
+
for (const taskId of context.input.taskOrder) {
|
|
1061
|
+
const pairRow = crossSurfaceRowFor(context, pair.candidateId, taskId);
|
|
1062
|
+
const leftRow = crossSurfaceRowFor(context, leftSingle.candidateId, taskId);
|
|
1063
|
+
const rightRow = crossSurfaceRowFor(context, rightSingle.candidateId, taskId);
|
|
1064
|
+
if (![
|
|
1065
|
+
pairRow,
|
|
1066
|
+
leftRow,
|
|
1067
|
+
rightRow
|
|
1068
|
+
].every((row) => row.completeness === "complete")) continue;
|
|
1069
|
+
if (pairRow.pass && !leftRow.pass && !rightRow.pass) synergyTaskIds.push(taskId);
|
|
1070
|
+
if (!pairRow.pass && (leftRow.pass || rightRow.pass)) interferenceTaskIds.push(taskId);
|
|
1071
|
+
}
|
|
1072
|
+
const incrementalVsConstituents = [compareCandidates(context, summaryById, leftSingle.candidateId, pair.candidateId), compareCandidates(context, summaryById, rightSingle.candidateId, pair.candidateId)];
|
|
1073
|
+
const betterSingle = compareSingleSummaries(context, leftSummary, rightSummary) <= 0 ? leftSummary : rightSummary;
|
|
1074
|
+
const firing = summarizePairEvidence(crossSurfaceRowsFor(context, pair.candidateId), leftComponent.componentId, rightComponent.componentId, "fired");
|
|
1075
|
+
const effect = summarizePairEvidence(crossSurfaceRowsFor(context, pair.candidateId), leftComponent.componentId, rightComponent.componentId, "effectObserved");
|
|
1076
|
+
const compatibility = pairCompatibility(context, pairSummary, baseline, betterSingle, incrementalVsConstituents, interactionReadyIds.has(leftSingle.candidateId) && interactionReadyIds.has(rightSingle.candidateId), interferenceTaskIds, firing, effect);
|
|
1077
|
+
entries.push({
|
|
1078
|
+
componentIds: [leftComponent.componentId, rightComponent.componentId],
|
|
1079
|
+
singleCandidateIds: [leftSingle.candidateId, rightSingle.candidateId],
|
|
1080
|
+
compositionCandidateId: pair.candidateId,
|
|
1081
|
+
benefitTaskIds: [...pairSummary.outcome.benefitTaskIds],
|
|
1082
|
+
regressionTaskIds: [...pairSummary.outcome.regressionTaskIds],
|
|
1083
|
+
synergyTaskIds,
|
|
1084
|
+
interferenceTaskIds,
|
|
1085
|
+
incrementalVsConstituents,
|
|
1086
|
+
relativeCostToBaseline: relativeCosts(context, pairSummary, baseline),
|
|
1087
|
+
firing,
|
|
1088
|
+
effect,
|
|
1089
|
+
interaction: interactionEffect(context, baseline.candidate, [leftSingle, rightSingle], pair),
|
|
1090
|
+
compatibility
|
|
1091
|
+
});
|
|
1092
|
+
}
|
|
1093
|
+
return entries;
|
|
1094
|
+
}
|
|
1095
|
+
function pairCompatibility(context, pair, baseline, betterSingle, comparisons, constituentsReady, interferenceTaskIds, firing, effect) {
|
|
1096
|
+
const reasons = [];
|
|
1097
|
+
if (!constituentsReady) reasons.push("constituent_not_ready");
|
|
1098
|
+
if (pair.outcome.missingTaskIds.length > 0 || pair.outcome.invalidTaskIds.length > 0 || pair.outcome.comparisonMissingTaskIds.length > 0) reasons.push("pair_incomplete");
|
|
1099
|
+
if (pair.outcome.regressionTaskIds.length > 0) reasons.push("baseline_regression");
|
|
1100
|
+
if (interferenceTaskIds.length > 0) reasons.push("interference");
|
|
1101
|
+
if (comparisons.find((comparison) => comparison.comparatorCandidateId === betterSingle.candidate.candidateId).winsTaskIds.length === 0) reasons.push("no_incremental_resolution");
|
|
1102
|
+
appendPairEvidenceReasons(reasons, firing, context.input.selection.minimumFiringTasks, context.input.selection.requireObservedFiring, "firing_below_minimum", "firing_unobserved");
|
|
1103
|
+
appendPairEvidenceReasons(reasons, effect, context.input.selection.minimumEffectTasks, context.input.selection.requireObservedEffect, "effect_below_minimum", "effect_unobserved");
|
|
1104
|
+
if (!withinCostLimits(context, pair, baseline)) reasons.push("cost_limit_exceeded");
|
|
1105
|
+
return {
|
|
1106
|
+
compatible: reasons.length === 0,
|
|
1107
|
+
reasons: unique(reasons),
|
|
1108
|
+
betterSingleCandidateId: betterSingle.candidate.candidateId
|
|
1109
|
+
};
|
|
1110
|
+
}
|
|
1111
|
+
function appendPairEvidenceReasons(reasons, evidence, minimum, requireObserved, belowReason, unobservedReason) {
|
|
1112
|
+
const leftObserved = evidence.bothTaskIds.length + evidence.leftOnlyTaskIds.length;
|
|
1113
|
+
const rightObserved = evidence.bothTaskIds.length + evidence.rightOnlyTaskIds.length;
|
|
1114
|
+
if (leftObserved < minimum || rightObserved < minimum) reasons.push(belowReason);
|
|
1115
|
+
if (requireObserved && evidence.unobservedTaskIds.length > 0) reasons.push(unobservedReason);
|
|
1116
|
+
}
|
|
1117
|
+
function interactionEffect(context, baseline, singles, composition) {
|
|
1118
|
+
const perTask = context.input.taskOrder.map((taskId) => {
|
|
1119
|
+
const baselineRow = crossSurfaceRowFor(context, baseline.candidateId, taskId);
|
|
1120
|
+
const singleRows = singles.map((single) => crossSurfaceRowFor(context, single.candidateId, taskId));
|
|
1121
|
+
const compositionRow = crossSurfaceRowFor(context, composition.candidateId, taskId);
|
|
1122
|
+
if (![
|
|
1123
|
+
baselineRow,
|
|
1124
|
+
...singleRows,
|
|
1125
|
+
compositionRow
|
|
1126
|
+
].every((row) => row.completeness === "complete")) return {
|
|
1127
|
+
taskId,
|
|
1128
|
+
passInteraction: null,
|
|
1129
|
+
scoreInteraction: null
|
|
1130
|
+
};
|
|
1131
|
+
const additivePass = singleRows.reduce((sum, row) => sum + Number(row.pass), 0) - (singles.length - 1) * Number(baselineRow.pass);
|
|
1132
|
+
const additiveScore = singleRows.reduce((sum, row) => sum + row.score, 0) - (singles.length - 1) * baselineRow.score;
|
|
1133
|
+
return {
|
|
1134
|
+
taskId,
|
|
1135
|
+
passInteraction: Number(compositionRow.pass) - additivePass,
|
|
1136
|
+
scoreInteraction: compositionRow.score - additiveScore
|
|
1137
|
+
};
|
|
1138
|
+
});
|
|
1139
|
+
const complete = perTask.filter((row) => row.passInteraction !== null && row.scoreInteraction !== null);
|
|
1140
|
+
const passInteractions = complete.map((row) => row.passInteraction);
|
|
1141
|
+
const scoreInteractions = complete.map((row) => row.scoreInteraction);
|
|
1142
|
+
return {
|
|
1143
|
+
perTask,
|
|
1144
|
+
n: complete.length,
|
|
1145
|
+
nMissing: perTask.length - complete.length,
|
|
1146
|
+
meanPassInteraction: meanOrNull$1(passInteractions),
|
|
1147
|
+
meanScoreInteraction: meanOrNull$1(scoreInteractions),
|
|
1148
|
+
passBootstrap: bootstrapInteraction(passInteractions, context),
|
|
1149
|
+
scoreBootstrap: bootstrapInteraction(scoreInteractions, context)
|
|
1150
|
+
};
|
|
1151
|
+
}
|
|
1152
|
+
function bootstrapInteraction(interactions, context) {
|
|
1153
|
+
if (interactions.length === 0) return null;
|
|
1154
|
+
return pairedBootstrap(new Array(interactions.length).fill(0), interactions, {
|
|
1155
|
+
...context.input.bootstrap,
|
|
1156
|
+
statistic: "mean"
|
|
1157
|
+
});
|
|
1158
|
+
}
|
|
1159
|
+
function selectCandidates(context, summaryById, eligibleSingles, interactionReadySingles, pairwise) {
|
|
1160
|
+
const bestSingleRanking = eligibleSingles.filter((summary) => {
|
|
1161
|
+
const componentId = summary.candidate.componentIds[0];
|
|
1162
|
+
return context.componentById.get(componentId).bestSingleEligible;
|
|
1163
|
+
});
|
|
1164
|
+
const ranking = bestSingleRanking.map((summary, index) => ({
|
|
1165
|
+
rank: index + 1,
|
|
1166
|
+
candidateId: summary.candidate.candidateId,
|
|
1167
|
+
componentId: summary.candidate.componentIds[0]
|
|
1168
|
+
}));
|
|
1169
|
+
const best = bestSingleRanking[0];
|
|
1170
|
+
return {
|
|
1171
|
+
bestSingle: best ? {
|
|
1172
|
+
candidateId: best.candidate.candidateId,
|
|
1173
|
+
componentId: best.candidate.componentIds[0],
|
|
1174
|
+
ranking
|
|
1175
|
+
} : null,
|
|
1176
|
+
naiveStack: selectNaiveStack(context, eligibleSingles),
|
|
1177
|
+
interactionAware: selectInteractionAware(context, summaryById, interactionReadySingles, pairwise)
|
|
1178
|
+
};
|
|
1179
|
+
}
|
|
1180
|
+
function selectNaiveStack(context, eligibleSingles) {
|
|
1181
|
+
if (eligibleSingles.length === 0) return null;
|
|
1182
|
+
const componentIds = eligibleSingles.map((summary) => summary.candidate.componentIds[0]).sort((left, right) => context.componentIndex.get(left) - context.componentIndex.get(right));
|
|
1183
|
+
const candidate = context.candidateByComponents.get(crossSurfaceComponentSetKey(componentIds));
|
|
1184
|
+
if (!candidate) throw new ValidationError(`analyzeCrossSurfaceInteractions: naive stack [${componentIds.join(", ")}] was not evaluated`);
|
|
1185
|
+
return {
|
|
1186
|
+
candidateId: candidate.candidateId,
|
|
1187
|
+
componentIds
|
|
1188
|
+
};
|
|
1189
|
+
}
|
|
1190
|
+
function selectInteractionAware(context, summaryById, interactionReadySingles, pairwise) {
|
|
1191
|
+
const baseline = summaryById.get(context.input.baselineCandidateId);
|
|
1192
|
+
const pairByComponents = new Map(pairwise.map((entry) => [crossSurfaceComponentSetKey(entry.componentIds), entry]));
|
|
1193
|
+
const paths = pairwise.filter((entry) => entry.compatibility.compatible).map((entry) => growInteractionPath(context, summaryById, interactionReadySingles, pairByComponents, baseline, summaryById.get(entry.compositionCandidateId)));
|
|
1194
|
+
if (paths.length === 0) return null;
|
|
1195
|
+
const qualified = paths.filter((path) => path.qualified).sort((left, right) => compareInteractionPaths(context, summaryById, left, right));
|
|
1196
|
+
const winning = qualified[0] ?? [...paths].sort((left, right) => compareInteractionPaths(context, summaryById, left, right))[0];
|
|
1197
|
+
return {
|
|
1198
|
+
seedCandidateId: winning.seedCandidateId,
|
|
1199
|
+
terminalCandidateId: winning.terminalCandidateId,
|
|
1200
|
+
terminalComponentIds: [...winning.terminalComponentIds],
|
|
1201
|
+
selectedCandidateId: qualified.length > 0 ? winning.terminalCandidateId : null,
|
|
1202
|
+
qualified: qualified.length > 0,
|
|
1203
|
+
evaluatedPaths: paths,
|
|
1204
|
+
steps: winning.steps
|
|
1205
|
+
};
|
|
1206
|
+
}
|
|
1207
|
+
function growInteractionPath(context, summaryById, interactionReadySingles, pairByComponents, baseline, seed) {
|
|
1208
|
+
let current = seed;
|
|
1209
|
+
let retained = [...seed.candidate.componentIds];
|
|
1210
|
+
let remaining = interactionReadySingles.filter((summary) => !retained.includes(summary.candidate.componentIds[0]));
|
|
1211
|
+
const steps = [];
|
|
1212
|
+
while (remaining.length > 0) {
|
|
1213
|
+
const decisions = remaining.map((addition) => evaluateAddition(context, summaryById, pairByComponents, baseline, current, retained, addition));
|
|
1214
|
+
const selected = decisions.filter((decision) => decision.eligible).sort((left, right) => compareAdditionDecisions(context, summaryById, left, right))[0];
|
|
1215
|
+
if (selected) selected.selected = true;
|
|
1216
|
+
steps.push({
|
|
1217
|
+
fromCandidateId: current.candidate.candidateId,
|
|
1218
|
+
retainedComponentIds: [...retained],
|
|
1219
|
+
considered: decisions.sort((left, right) => context.candidateIndex.get(left.additionCandidateId) - context.candidateIndex.get(right.additionCandidateId)),
|
|
1220
|
+
selectedCandidateId: selected?.bundleCandidateId ?? null
|
|
1221
|
+
});
|
|
1222
|
+
if (!selected?.bundleCandidateId) break;
|
|
1223
|
+
current = summaryById.get(selected.bundleCandidateId);
|
|
1224
|
+
retained = [...current.candidate.componentIds];
|
|
1225
|
+
remaining = remaining.filter((summary) => summary.candidate.candidateId !== selected.additionCandidateId);
|
|
1226
|
+
}
|
|
1227
|
+
const qualified = retained.length >= context.input.selection.minimumBundleComponents;
|
|
1228
|
+
return {
|
|
1229
|
+
seedCandidateId: seed.candidate.candidateId,
|
|
1230
|
+
terminalCandidateId: current.candidate.candidateId,
|
|
1231
|
+
terminalComponentIds: retained,
|
|
1232
|
+
qualified,
|
|
1233
|
+
steps
|
|
1234
|
+
};
|
|
1235
|
+
}
|
|
1236
|
+
function compareInteractionPaths(context, summaryById, left, right) {
|
|
1237
|
+
const leftSummary = summaryById.get(left.terminalCandidateId);
|
|
1238
|
+
const rightSummary = summaryById.get(right.terminalCandidateId);
|
|
1239
|
+
const byNetBenefit = rightSummary.outcome.netBenefit - leftSummary.outcome.netBenefit;
|
|
1240
|
+
if (byNetBenefit !== 0) return byNetBenefit;
|
|
1241
|
+
const byBenefit = rightSummary.outcome.benefitTaskIds.length - leftSummary.outcome.benefitTaskIds.length;
|
|
1242
|
+
if (byBenefit !== 0) return byBenefit;
|
|
1243
|
+
const byRegression = leftSummary.outcome.regressionTaskIds.length - rightSummary.outcome.regressionTaskIds.length;
|
|
1244
|
+
if (byRegression !== 0) return byRegression;
|
|
1245
|
+
for (const metric of context.input.costMetricOrder) {
|
|
1246
|
+
const byCost = leftSummary.costs[metric].median - rightSummary.costs[metric].median;
|
|
1247
|
+
if (byCost !== 0) return byCost;
|
|
1248
|
+
}
|
|
1249
|
+
const byBytes = leftSummary.candidate.artifactBytes - rightSummary.candidate.artifactBytes;
|
|
1250
|
+
if (byBytes !== 0) return byBytes;
|
|
1251
|
+
const byCandidateOrder = context.candidateIndex.get(left.terminalCandidateId) - context.candidateIndex.get(right.terminalCandidateId);
|
|
1252
|
+
if (byCandidateOrder !== 0) return byCandidateOrder;
|
|
1253
|
+
return context.candidateIndex.get(left.seedCandidateId) - context.candidateIndex.get(right.seedCandidateId);
|
|
1254
|
+
}
|
|
1255
|
+
function evaluateAddition(context, summaryById, pairByComponents, baseline, current, retained, addition) {
|
|
1256
|
+
const additionComponentId = addition.candidate.componentIds[0];
|
|
1257
|
+
const reasons = [];
|
|
1258
|
+
for (const retainedComponentId of retained) if (!pairByComponents.get(crossSurfaceComponentSetKey(canonicalCrossSurfaceComponents(context, [retainedComponentId, additionComponentId])))?.compatibility.compatible) reasons.push("pair_incompatible");
|
|
1259
|
+
const bundleComponents = canonicalCrossSurfaceComponents(context, [...retained, additionComponentId]);
|
|
1260
|
+
const bundle = context.candidateByComponents.get(crossSurfaceComponentSetKey(bundleComponents));
|
|
1261
|
+
if (!bundle) {
|
|
1262
|
+
reasons.push("full_bundle_not_evaluated");
|
|
1263
|
+
return emptyAdditionDecision(addition, additionComponentId, reasons);
|
|
1264
|
+
}
|
|
1265
|
+
const bundleSummary = summaryById.get(bundle.candidateId);
|
|
1266
|
+
if (bundleSummary.outcome.missingTaskIds.length > 0 || bundleSummary.outcome.invalidTaskIds.length > 0 || bundleSummary.outcome.comparisonMissingTaskIds.length > 0) reasons.push("bundle_incomplete");
|
|
1267
|
+
if (bundleSummary.outcome.regressionTaskIds.length > 0) reasons.push("baseline_regression");
|
|
1268
|
+
const comparison = compareCandidates(context, summaryById, current.candidate.candidateId, bundle.candidateId);
|
|
1269
|
+
if (comparison.winsTaskIds.length === 0) reasons.push("no_incremental_resolution");
|
|
1270
|
+
if (comparison.regressionTaskIds.length > 0) reasons.push("incremental_regression");
|
|
1271
|
+
appendBundleEvidenceReasons(reasons, bundleSummary.firing, context.input.selection.minimumFiringTasks, context.input.selection.requireObservedFiring, "firing_below_minimum", "firing_unobserved");
|
|
1272
|
+
appendBundleEvidenceReasons(reasons, bundleSummary.effect, context.input.selection.minimumEffectTasks, context.input.selection.requireObservedEffect, "effect_below_minimum", "effect_unobserved");
|
|
1273
|
+
if (!withinCostLimits(context, bundleSummary, baseline)) reasons.push("cost_limit_exceeded");
|
|
1274
|
+
return {
|
|
1275
|
+
additionCandidateId: addition.candidate.candidateId,
|
|
1276
|
+
additionComponentId,
|
|
1277
|
+
bundleCandidateId: bundle.candidateId,
|
|
1278
|
+
incrementalResolutionTaskIds: comparison.winsTaskIds,
|
|
1279
|
+
incrementalRegressionTaskIds: comparison.regressionTaskIds,
|
|
1280
|
+
incrementalMedianCost: Object.fromEntries(context.input.costMetricOrder.map((metric) => [metric, bundleSummary.costs[metric].median - current.costs[metric].median])),
|
|
1281
|
+
eligible: reasons.length === 0,
|
|
1282
|
+
selected: false,
|
|
1283
|
+
reasons: unique(reasons)
|
|
1284
|
+
};
|
|
1285
|
+
}
|
|
1286
|
+
function emptyAdditionDecision(addition, additionComponentId, reasons) {
|
|
1287
|
+
return {
|
|
1288
|
+
additionCandidateId: addition.candidate.candidateId,
|
|
1289
|
+
additionComponentId,
|
|
1290
|
+
bundleCandidateId: null,
|
|
1291
|
+
incrementalResolutionTaskIds: [],
|
|
1292
|
+
incrementalRegressionTaskIds: [],
|
|
1293
|
+
incrementalMedianCost: null,
|
|
1294
|
+
eligible: false,
|
|
1295
|
+
selected: false,
|
|
1296
|
+
reasons: unique(reasons)
|
|
1297
|
+
};
|
|
1298
|
+
}
|
|
1299
|
+
function appendBundleEvidenceReasons(reasons, evidence, minimum, requireObserved, belowReason, unobservedReason) {
|
|
1300
|
+
if (evidence.byComponent.some((component) => component.observedTaskIds.length < minimum)) reasons.push(belowReason);
|
|
1301
|
+
if (requireObserved && evidence.byComponent.some((component) => component.unobservedTaskIds.length > 0)) reasons.push(unobservedReason);
|
|
1302
|
+
}
|
|
1303
|
+
function compareAdditionDecisions(context, summaryById, left, right) {
|
|
1304
|
+
const byWins = right.incrementalResolutionTaskIds.length - left.incrementalResolutionTaskIds.length;
|
|
1305
|
+
if (byWins !== 0) return byWins;
|
|
1306
|
+
for (const metric of context.input.costMetricOrder) {
|
|
1307
|
+
const byCost = left.incrementalMedianCost[metric] - right.incrementalMedianCost[metric];
|
|
1308
|
+
if (byCost !== 0) return byCost;
|
|
1309
|
+
}
|
|
1310
|
+
const byBytes = summaryById.get(left.additionCandidateId).candidate.artifactBytes - summaryById.get(right.additionCandidateId).candidate.artifactBytes;
|
|
1311
|
+
if (byBytes !== 0) return byBytes;
|
|
1312
|
+
return context.candidateIndex.get(left.additionCandidateId) - context.candidateIndex.get(right.additionCandidateId);
|
|
1313
|
+
}
|
|
1314
|
+
function compareCandidates(context, summaryById, comparatorCandidateId, treatmentCandidateId) {
|
|
1315
|
+
const winsTaskIds = [];
|
|
1316
|
+
const regressionTaskIds = [];
|
|
1317
|
+
const missingTaskIds = [];
|
|
1318
|
+
for (const taskId of context.input.taskOrder) {
|
|
1319
|
+
const comparator = crossSurfaceRowFor(context, comparatorCandidateId, taskId);
|
|
1320
|
+
const treatment = crossSurfaceRowFor(context, treatmentCandidateId, taskId);
|
|
1321
|
+
if (comparator.completeness !== "complete" || treatment.completeness !== "complete") missingTaskIds.push(taskId);
|
|
1322
|
+
else if (treatment.pass && !comparator.pass) winsTaskIds.push(taskId);
|
|
1323
|
+
else if (!treatment.pass && comparator.pass) regressionTaskIds.push(taskId);
|
|
1324
|
+
}
|
|
1325
|
+
return {
|
|
1326
|
+
comparatorCandidateId,
|
|
1327
|
+
treatmentCandidateId,
|
|
1328
|
+
winsTaskIds,
|
|
1329
|
+
regressionTaskIds,
|
|
1330
|
+
missingTaskIds,
|
|
1331
|
+
paired: pairedComparison(context, comparatorCandidateId, treatmentCandidateId),
|
|
1332
|
+
relativeCost: relativeCosts(context, summaryById.get(treatmentCandidateId), summaryById.get(comparatorCandidateId))
|
|
1333
|
+
};
|
|
1334
|
+
}
|
|
1335
|
+
function pairedComparison(context, comparatorCandidateId, treatmentCandidateId) {
|
|
1336
|
+
const rows = [];
|
|
1337
|
+
for (const taskId of context.input.taskOrder) {
|
|
1338
|
+
rows.push(toPairedArmRow(context, crossSurfaceRowFor(context, comparatorCandidateId, taskId)));
|
|
1339
|
+
rows.push(toPairedArmRow(context, crossSurfaceRowFor(context, treatmentCandidateId, taskId)));
|
|
1340
|
+
}
|
|
1341
|
+
return comparePairedArms(rows, {
|
|
1342
|
+
baselineArm: comparatorCandidateId,
|
|
1343
|
+
treatmentArm: treatmentCandidateId,
|
|
1344
|
+
metricNames: ["score", ...context.input.costMetricOrder],
|
|
1345
|
+
bootstrap: {
|
|
1346
|
+
...context.input.bootstrap,
|
|
1347
|
+
statistic: "mean"
|
|
1348
|
+
}
|
|
1349
|
+
});
|
|
1350
|
+
}
|
|
1351
|
+
function toPairedArmRow(context, row) {
|
|
1352
|
+
const metrics = Object.fromEntries(context.input.costMetricOrder.map((metric) => [metric, row.cost[metric]]));
|
|
1353
|
+
if (row.completeness === "complete") metrics.score = row.score;
|
|
1354
|
+
return {
|
|
1355
|
+
pairKey: row.taskId,
|
|
1356
|
+
arm: row.candidateId,
|
|
1357
|
+
...row.completeness === "complete" ? { pass: row.pass } : {},
|
|
1358
|
+
metrics
|
|
1359
|
+
};
|
|
1360
|
+
}
|
|
1361
|
+
function summarizePairEvidence(rows, leftComponentId, rightComponentId, field) {
|
|
1362
|
+
const bothTaskIds = [];
|
|
1363
|
+
const leftOnlyTaskIds = [];
|
|
1364
|
+
const rightOnlyTaskIds = [];
|
|
1365
|
+
const neitherTaskIds = [];
|
|
1366
|
+
const unobservedTaskIds = [];
|
|
1367
|
+
for (const row of rows) {
|
|
1368
|
+
const left = evidenceValue(row, leftComponentId, field);
|
|
1369
|
+
const right = evidenceValue(row, rightComponentId, field);
|
|
1370
|
+
if (left === null || right === null) unobservedTaskIds.push(row.taskId);
|
|
1371
|
+
else if (left && right) bothTaskIds.push(row.taskId);
|
|
1372
|
+
else if (left) leftOnlyTaskIds.push(row.taskId);
|
|
1373
|
+
else if (right) rightOnlyTaskIds.push(row.taskId);
|
|
1374
|
+
else neitherTaskIds.push(row.taskId);
|
|
1375
|
+
}
|
|
1376
|
+
return {
|
|
1377
|
+
bothTaskIds,
|
|
1378
|
+
leftOnlyTaskIds,
|
|
1379
|
+
rightOnlyTaskIds,
|
|
1380
|
+
neitherTaskIds,
|
|
1381
|
+
unobservedTaskIds
|
|
1382
|
+
};
|
|
1383
|
+
}
|
|
1384
|
+
function relativeCosts(context, treatment, comparator) {
|
|
1385
|
+
return Object.fromEntries(context.input.costMetricOrder.map((metric) => {
|
|
1386
|
+
const treatmentMedian = treatment.costs[metric].median;
|
|
1387
|
+
const comparatorMedian = comparator.costs[metric].median;
|
|
1388
|
+
return [metric, {
|
|
1389
|
+
treatmentMedian,
|
|
1390
|
+
comparatorMedian,
|
|
1391
|
+
medianDelta: treatmentMedian - comparatorMedian,
|
|
1392
|
+
medianRatio: comparatorMedian === 0 ? treatmentMedian === 0 ? 1 : null : treatmentMedian / comparatorMedian
|
|
1393
|
+
}];
|
|
1394
|
+
}));
|
|
1395
|
+
}
|
|
1396
|
+
function withinCostLimits(context, treatment, baseline) {
|
|
1397
|
+
const relative = relativeCosts(context, treatment, baseline);
|
|
1398
|
+
return Object.entries(context.input.selection.maximumMedianCostRatioToBaseline).every(([metric, limit]) => {
|
|
1399
|
+
const ratio = relative[metric].medianRatio;
|
|
1400
|
+
return ratio !== null && ratio <= limit;
|
|
1401
|
+
});
|
|
1402
|
+
}
|
|
1403
|
+
function distribution(values) {
|
|
1404
|
+
const sorted = [...values].sort((left, right) => left - right);
|
|
1405
|
+
const total = sorted.reduce((sum, value) => sum + value, 0);
|
|
1406
|
+
const middle = Math.floor(sorted.length / 2);
|
|
1407
|
+
const median = sorted.length % 2 === 1 ? sorted[middle] : (sorted[middle - 1] + sorted[middle]) / 2;
|
|
1408
|
+
return {
|
|
1409
|
+
n: sorted.length,
|
|
1410
|
+
min: sorted[0],
|
|
1411
|
+
median,
|
|
1412
|
+
mean: total / sorted.length,
|
|
1413
|
+
max: sorted[sorted.length - 1],
|
|
1414
|
+
total
|
|
1415
|
+
};
|
|
1416
|
+
}
|
|
1417
|
+
function evidenceValue(row, componentId, field) {
|
|
1418
|
+
return row.componentEvidence.find((evidence) => evidence.componentId === componentId)[field];
|
|
1419
|
+
}
|
|
1420
|
+
function meanOrNull$1(values) {
|
|
1421
|
+
return values.length === 0 ? null : values.reduce((sum, value) => sum + value, 0) / values.length;
|
|
1422
|
+
}
|
|
1423
|
+
function unique(values) {
|
|
1424
|
+
return [...new Set(values)];
|
|
1425
|
+
}
|
|
1426
|
+
//#endregion
|
|
1427
|
+
//#region src/campaign/fixtures.ts
|
|
1428
|
+
/** Walk `evalsDir` and return the relative name of every fixture directory (one containing an exact-case `PROMPT.md`). */
|
|
1429
|
+
function discoverEvalFixtures(evalsDir) {
|
|
1430
|
+
const root = resolve(evalsDir);
|
|
1431
|
+
if (!existsSync(root)) throw new Error(`discoverEvalFixtures: evalsDir not found: ${root}`);
|
|
1432
|
+
const fixtures = [];
|
|
1433
|
+
const walk = (dir, base = "") => {
|
|
1434
|
+
for (const entry of readdirSync(dir).sort()) {
|
|
1435
|
+
if (entry.startsWith(".") || entry === "node_modules") continue;
|
|
1436
|
+
const fullPath = join(dir, entry);
|
|
1437
|
+
if (!statSync(fullPath).isDirectory()) continue;
|
|
1438
|
+
const name = base ? `${base}/${entry}` : entry;
|
|
1439
|
+
if (existsWithExactCase(fullPath, "PROMPT.md")) fixtures.push(name);
|
|
1440
|
+
else walk(fullPath, name);
|
|
1441
|
+
}
|
|
1442
|
+
};
|
|
1443
|
+
walk(root);
|
|
1444
|
+
return fixtures;
|
|
1445
|
+
}
|
|
1446
|
+
/**
|
|
1447
|
+
* Load ONE fixture by name: reads `PROMPT.md` (plus `EVAL.ts`/`EVAL.tsx` and `package.json` under
|
|
1448
|
+
* `vitest` validation) and content-fingerprints the full file set for cache identity.
|
|
1449
|
+
*/
|
|
1450
|
+
function loadEvalFixture(evalsDir, name, options = {}) {
|
|
1451
|
+
const validation = options.validation ?? "vitest";
|
|
1452
|
+
const root = resolve(evalsDir);
|
|
1453
|
+
const fixturePath = resolve(root, name);
|
|
1454
|
+
assertInside(root, fixturePath, name);
|
|
1455
|
+
if (!existsSync(fixturePath) || !statSync(fixturePath).isDirectory()) throw new Error(`loadEvalFixture: fixture not found: ${name}`);
|
|
1456
|
+
if (!existsWithExactCase(fixturePath, "PROMPT.md")) throw new Error(`loadEvalFixture: ${name} is missing exact-case PROMPT.md`);
|
|
1457
|
+
const promptPath = join(fixturePath, "PROMPT.md");
|
|
1458
|
+
const evalPath = resolveEvalPath(fixturePath);
|
|
1459
|
+
const packageJsonPath = existsWithExactCase(fixturePath, "package.json") ? join(fixturePath, "package.json") : void 0;
|
|
1460
|
+
if (validation !== "none") {
|
|
1461
|
+
if (!evalPath) throw new Error(`loadEvalFixture: ${name} is missing exact-case EVAL.ts or EVAL.tsx`);
|
|
1462
|
+
if (!packageJsonPath) throw new Error(`loadEvalFixture: ${name} is missing exact-case package.json`);
|
|
1463
|
+
assertModulePackage(packageJsonPath, name);
|
|
1464
|
+
}
|
|
1465
|
+
const files = collectFixtureFiles(fixturePath);
|
|
1466
|
+
return {
|
|
1467
|
+
name,
|
|
1468
|
+
path: fixturePath,
|
|
1469
|
+
promptPath,
|
|
1470
|
+
evalPath,
|
|
1471
|
+
packageJsonPath,
|
|
1472
|
+
prompt: readFileSync(promptPath, "utf8"),
|
|
1473
|
+
files,
|
|
1474
|
+
fingerprint: contentHash({
|
|
1475
|
+
files,
|
|
1476
|
+
config: options.fingerprintConfig ?? null
|
|
1477
|
+
})
|
|
1478
|
+
};
|
|
1479
|
+
}
|
|
1480
|
+
/** Load fixtures (all discovered, or just `names`) as campaign `Scenario`s tagged `eval-fixture`. */
|
|
1481
|
+
function loadEvalFixtureScenarios(evalsDir, options = {}) {
|
|
1482
|
+
return (options.names ?? discoverEvalFixtures(evalsDir)).map((name) => {
|
|
1483
|
+
const fixture = loadEvalFixture(evalsDir, name, options);
|
|
1484
|
+
return {
|
|
1485
|
+
id: fixture.name,
|
|
1486
|
+
kind: "eval-fixture",
|
|
1487
|
+
tags: ["eval-fixture"],
|
|
1488
|
+
fixtureName: fixture.name,
|
|
1489
|
+
fixturePath: fixture.path,
|
|
1490
|
+
promptPath: fixture.promptPath,
|
|
1491
|
+
evalPath: fixture.evalPath,
|
|
1492
|
+
packageJsonPath: fixture.packageJsonPath,
|
|
1493
|
+
prompt: fixture.prompt,
|
|
1494
|
+
fingerprint: fixture.fingerprint
|
|
1495
|
+
};
|
|
1496
|
+
});
|
|
1497
|
+
}
|
|
1498
|
+
/**
|
|
1499
|
+
* Dry-run planner for a fixture campaign: loads the scenarios, delegates to `planCampaignRun`,
|
|
1500
|
+
* and returns the plan plus each fixture's name/path/fingerprint.
|
|
1501
|
+
*/
|
|
1502
|
+
function planEvalFixtureRun(options) {
|
|
1503
|
+
const scenarios = loadEvalFixtureScenarios(options.evalsDir, {
|
|
1504
|
+
names: options.names,
|
|
1505
|
+
validation: options.validation,
|
|
1506
|
+
fingerprintConfig: options.fingerprintConfig
|
|
1507
|
+
});
|
|
1508
|
+
return {
|
|
1509
|
+
...planCampaignRun({
|
|
1510
|
+
scenarios,
|
|
1511
|
+
dispatchRef: options.dispatchRef ?? "eval-fixture-dispatch",
|
|
1512
|
+
judges: options.judges,
|
|
1513
|
+
seed: options.seed,
|
|
1514
|
+
reps: options.reps,
|
|
1515
|
+
resumable: options.resumable,
|
|
1516
|
+
runDir: options.runDir,
|
|
1517
|
+
storage: options.storage
|
|
1518
|
+
}),
|
|
1519
|
+
fixtures: scenarios.map((scenario) => ({
|
|
1520
|
+
fixtureName: scenario.fixtureName,
|
|
1521
|
+
fixturePath: scenario.fixturePath,
|
|
1522
|
+
fingerprint: scenario.fingerprint
|
|
1523
|
+
}))
|
|
1524
|
+
};
|
|
1525
|
+
}
|
|
1526
|
+
function existsWithExactCase(dirPath, fileName) {
|
|
1527
|
+
try {
|
|
1528
|
+
return readdirSync(dirPath).includes(fileName);
|
|
1529
|
+
} catch {
|
|
1530
|
+
return false;
|
|
1531
|
+
}
|
|
1532
|
+
}
|
|
1533
|
+
function resolveEvalPath(fixturePath) {
|
|
1534
|
+
if (existsWithExactCase(fixturePath, "EVAL.ts")) return join(fixturePath, "EVAL.ts");
|
|
1535
|
+
if (existsWithExactCase(fixturePath, "EVAL.tsx")) return join(fixturePath, "EVAL.tsx");
|
|
1536
|
+
}
|
|
1537
|
+
function assertModulePackage(packageJsonPath, name) {
|
|
1538
|
+
let parsed;
|
|
1539
|
+
try {
|
|
1540
|
+
parsed = JSON.parse(readFileSync(packageJsonPath, "utf8"));
|
|
1541
|
+
} catch (err) {
|
|
1542
|
+
throw new Error(`loadEvalFixture: ${name} package.json is invalid JSON: ${err instanceof Error ? err.message : String(err)}`);
|
|
1543
|
+
}
|
|
1544
|
+
if (typeof parsed !== "object" || parsed === null || parsed.type !== "module") throw new Error(`loadEvalFixture: ${name} package.json must set "type": "module"`);
|
|
1545
|
+
}
|
|
1546
|
+
function collectFixtureFiles(fixturePath, base = "") {
|
|
1547
|
+
const files = [];
|
|
1548
|
+
for (const entry of readdirSync(join(fixturePath, base)).sort()) {
|
|
1549
|
+
if (entry === "node_modules" || entry === ".git") continue;
|
|
1550
|
+
const relativePath = base ? `${base}/${entry}` : entry;
|
|
1551
|
+
const fullPath = join(fixturePath, relativePath);
|
|
1552
|
+
if (statSync(fullPath).isDirectory()) {
|
|
1553
|
+
files.push(...collectFixtureFiles(fixturePath, relativePath));
|
|
1554
|
+
continue;
|
|
1555
|
+
}
|
|
1556
|
+
const bytes = readFileSync(fullPath);
|
|
1557
|
+
files.push({
|
|
1558
|
+
path: relativePath,
|
|
1559
|
+
sha256: createHash("sha256").update(bytes).digest("hex"),
|
|
1560
|
+
bytes: bytes.byteLength
|
|
1561
|
+
});
|
|
1562
|
+
}
|
|
1563
|
+
return files;
|
|
1564
|
+
}
|
|
1565
|
+
function assertInside(root, target, label) {
|
|
1566
|
+
const rel = relative(root, target);
|
|
1567
|
+
if (rel === "" || !rel.startsWith("..") && !isAbsolute(rel)) return;
|
|
1568
|
+
throw new Error(`loadEvalFixture: fixture path escapes evalsDir: ${label}`);
|
|
1569
|
+
}
|
|
1570
|
+
//#endregion
|
|
1571
|
+
//#region src/campaign/gates/neutralization-gate.ts
|
|
1572
|
+
/** Mean of a numeric array; 0 for an empty array (callers guard n separately). */
|
|
1573
|
+
function mean$1(xs) {
|
|
1574
|
+
return xs.length === 0 ? 0 : xs.reduce((a, b) => a + b, 0) / xs.length;
|
|
1575
|
+
}
|
|
1576
|
+
/** Paired mean held-out lift of `arm` over baseline, on the in-scope cells. */
|
|
1577
|
+
function pairedLift(arm, baseline, scenarioIds) {
|
|
1578
|
+
const paired = pairHoldout(arm, baseline, scenarioIds, (s) => s.composite);
|
|
1579
|
+
const deltas = paired.after.map((a, i) => a - (paired.before[i] ?? 0));
|
|
1580
|
+
return {
|
|
1581
|
+
lift: mean$1(deltas),
|
|
1582
|
+
n: deltas.length
|
|
1583
|
+
};
|
|
1584
|
+
}
|
|
1585
|
+
/**
|
|
1586
|
+
* Composable placebo gate: ships only when the candidate's held-out lift is NOT
|
|
1587
|
+
* mostly reproduced by a footprint-matched neutralized variant.
|
|
1588
|
+
*/
|
|
1589
|
+
function neutralizationGate(options) {
|
|
1590
|
+
const maxDecorativeFraction = options.maxDecorativeFraction ?? .5;
|
|
1591
|
+
return {
|
|
1592
|
+
name: "neutralizationGate",
|
|
1593
|
+
async decide(ctx) {
|
|
1594
|
+
if (!ctx.baselineJudgeScores) throw new Error("neutralizationGate: ctx.baselineJudgeScores is required — the placebo control measures lift OVER baseline.");
|
|
1595
|
+
if (!ctx.neutralizedJudgeScores) throw new Error("neutralizationGate: ctx.neutralizedJudgeScores is required. It is populated by runImprovementLoop only when a `neutralize` function is supplied — composing this gate without that wiring would pass an unproven candidate.");
|
|
1596
|
+
const scenarioIds = new Set(options.scenarios.map((s) => s.id));
|
|
1597
|
+
const cand = pairedLift(ctx.judgeScores, ctx.baselineJudgeScores, scenarioIds);
|
|
1598
|
+
const neut = pairedLift(ctx.neutralizedJudgeScores, ctx.baselineJudgeScores, scenarioIds);
|
|
1599
|
+
if (cand.lift <= 0) return {
|
|
1600
|
+
decision: "hold",
|
|
1601
|
+
reasons: [`neutralization: candidate held-out lift ${cand.lift.toFixed(3)} ≤ 0 — no positive lift to attribute to content`],
|
|
1602
|
+
contributingGates: [{
|
|
1603
|
+
name: "neutralizationGate",
|
|
1604
|
+
status: "fail",
|
|
1605
|
+
detail: {
|
|
1606
|
+
candidateLift: cand.lift,
|
|
1607
|
+
neutralizedLift: neut.lift,
|
|
1608
|
+
n: cand.n
|
|
1609
|
+
}
|
|
1610
|
+
}],
|
|
1611
|
+
delta: cand.lift
|
|
1612
|
+
};
|
|
1613
|
+
const decorativeFraction = neut.lift / cand.lift;
|
|
1614
|
+
const passed = decorativeFraction < maxDecorativeFraction;
|
|
1615
|
+
const pct = (decorativeFraction * 100).toFixed(0);
|
|
1616
|
+
return {
|
|
1617
|
+
decision: passed ? "ship" : "hold",
|
|
1618
|
+
reasons: passed ? [`neutralization: content is causal — blanked variant reproduces ${pct}% of the lift (< ${(maxDecorativeFraction * 100).toFixed(0)}%); candidate Δ ${cand.lift.toFixed(3)}, neutralized Δ ${neut.lift.toFixed(3)}`] : [`neutralization: lift is DECORATIVE — blanking the content (footprint-matched) reproduces ${pct}% of the lift (≥ ${(maxDecorativeFraction * 100).toFixed(0)}%); candidate Δ ${cand.lift.toFixed(3)}, neutralized Δ ${neut.lift.toFixed(3)}`],
|
|
1619
|
+
contributingGates: [{
|
|
1620
|
+
name: "neutralizationGate",
|
|
1621
|
+
status: passed ? "pass" : "fail",
|
|
1622
|
+
detail: {
|
|
1623
|
+
candidateLift: cand.lift,
|
|
1624
|
+
neutralizedLift: neut.lift,
|
|
1625
|
+
decorativeFraction,
|
|
1626
|
+
maxDecorativeFraction,
|
|
1627
|
+
n: cand.n
|
|
1628
|
+
}
|
|
1629
|
+
}],
|
|
1630
|
+
delta: cand.lift
|
|
1631
|
+
};
|
|
1632
|
+
}
|
|
1633
|
+
};
|
|
1634
|
+
}
|
|
1635
|
+
//#endregion
|
|
1636
|
+
//#region src/campaign/gates/sequential.ts
|
|
1637
|
+
/**
|
|
1638
|
+
* Anytime-valid sequential promotion gate — an e-process (betting
|
|
1639
|
+
* test-martingale, see `eProcess` in `statistics.ts`) over paired
|
|
1640
|
+
* per-scenario deltas, so a campaign stops the MOMENT evidence decides
|
|
1641
|
+
* instead of burning a fixed-n budget. Decisions remain valid at any
|
|
1642
|
+
* data-dependent stopping time (Ville's inequality), which is exactly what
|
|
1643
|
+
* fixed-n machinery cannot offer: peeking at a bootstrap CI after every
|
|
1644
|
+
* observation and stopping on the first significant peek inflates type-I
|
|
1645
|
+
* error far beyond alpha.
|
|
1646
|
+
*
|
|
1647
|
+
* REPLACES, never layers on, a fixed-n gate. Running `heldoutSignificance`
|
|
1648
|
+
* or `paretoSignificanceGate` repeatedly on a growing sample and stopping
|
|
1649
|
+
* early is optional stopping no matter how it is dressed up; this gate is
|
|
1650
|
+
* the valid way to stop early. Use one or the other per evidence stream.
|
|
1651
|
+
*
|
|
1652
|
+
* Pre-registration binding: anytime validity holds only for the
|
|
1653
|
+
* PRE-REGISTERED statistic. When a `SignedManifest` is bound, the gate takes
|
|
1654
|
+
* alpha from `manifest.alpha`, the observation budget from
|
|
1655
|
+
* `manifest.preRegisteredN`, orients deltas by `manifest.direction`, and
|
|
1656
|
+
* shifts the null boundary by `manifest.minEffect` — re-deciding the same
|
|
1657
|
+
* stream under different parameters after seeing data would reopen optional
|
|
1658
|
+
* stopping under a fancier name. The manifest's content hash is verified at
|
|
1659
|
+
* construction (sync, same `sha256-content` scheme as `signManifest`).
|
|
1660
|
+
*
|
|
1661
|
+
* Non-iid caveat (stated honestly): the supermartingale guarantee needs each
|
|
1662
|
+
* delta's conditional mean under H0 to stay ≤ the null boundary given the
|
|
1663
|
+
* past — exchangeable scenario deltas suffice. Scenario streams ordered by
|
|
1664
|
+
* difficulty or by scenario family violate this; `decide(ctx)` therefore
|
|
1665
|
+
* shuffles the paired deltas with a SEEDED permutation by default (the
|
|
1666
|
+
* permutation is data-independent, so bet predictability is preserved).
|
|
1667
|
+
* Stratified betting (per-stratum λ) is future work, not implemented here.
|
|
1668
|
+
*/
|
|
1669
|
+
/** Sync twin of `verifyManifest` — same `sha256-content` scheme
|
|
1670
|
+
* (sha256 over the canonicalized manifest minus contentHash/algo), via
|
|
1671
|
+
* node:crypto so gate construction can stay synchronous and fail loud
|
|
1672
|
+
* before any observation is consumed. */
|
|
1673
|
+
function verifyManifestSync(m) {
|
|
1674
|
+
if (m.algo !== void 0 && m.algo !== "sha256-content") throw new Error(`sequentialPairedGate: unrecognized manifest hash algo '${m.algo}'`);
|
|
1675
|
+
const { contentHash, algo: _algo, ...rest } = m;
|
|
1676
|
+
const bytes = JSON.stringify(canonicalize(rest));
|
|
1677
|
+
return createHash("sha256").update(bytes, "utf8").digest("hex") === contentHash;
|
|
1678
|
+
}
|
|
1679
|
+
function resolveConfig(opts) {
|
|
1680
|
+
const m = opts.preRegistration;
|
|
1681
|
+
let alpha;
|
|
1682
|
+
let maxN;
|
|
1683
|
+
let direction;
|
|
1684
|
+
let minEffect;
|
|
1685
|
+
if (m) {
|
|
1686
|
+
if (!verifyManifestSync(m)) throw new Error(`sequentialPairedGate: pre-registration manifest '${m.id}' content hash mismatch (tampered)`);
|
|
1687
|
+
if (opts.alpha !== void 0 && opts.alpha !== m.alpha) throw new Error(`sequentialPairedGate: alpha ${opts.alpha} conflicts with pre-registered alpha ${m.alpha} — the registered statistic is the only one anytime validity covers`);
|
|
1688
|
+
if (opts.maxN !== void 0 && opts.maxN !== m.preRegisteredN) throw new Error(`sequentialPairedGate: maxN ${opts.maxN} conflicts with pre-registered N ${m.preRegisteredN}`);
|
|
1689
|
+
alpha = m.alpha;
|
|
1690
|
+
maxN = m.preRegisteredN;
|
|
1691
|
+
direction = m.direction;
|
|
1692
|
+
minEffect = m.minEffect;
|
|
1693
|
+
} else {
|
|
1694
|
+
alpha = opts.alpha ?? .05;
|
|
1695
|
+
if (opts.maxN === void 0) throw new Error("sequentialPairedGate: maxN is required (or bind a preRegistration manifest whose preRegisteredN is the budget) — an unbounded stream has no pre-registered budget");
|
|
1696
|
+
maxN = opts.maxN;
|
|
1697
|
+
direction = "increase";
|
|
1698
|
+
minEffect = 0;
|
|
1699
|
+
}
|
|
1700
|
+
const scale = opts.scale ?? 1;
|
|
1701
|
+
if (!Number.isFinite(scale) || scale <= 0) throw new Error(`sequentialPairedGate: scale must be > 0, got ${scale}`);
|
|
1702
|
+
if (!Number.isInteger(maxN) || maxN < 1) throw new Error(`sequentialPairedGate: maxN must be a positive integer, got ${maxN}`);
|
|
1703
|
+
const minN = opts.minN ?? 5;
|
|
1704
|
+
if (!Number.isInteger(minN) || minN < 1 || minN > maxN) throw new Error(`sequentialPairedGate: minN must be an integer in [1, maxN=${maxN}], got ${minN}`);
|
|
1705
|
+
if (!Number.isFinite(minEffect) || minEffect < 0 || minEffect >= scale) throw new Error(`sequentialPairedGate: minEffect must be in [0, scale=${scale}), got ${minEffect}`);
|
|
1706
|
+
const nullMean = .5 + minEffect / (2 * scale);
|
|
1707
|
+
return {
|
|
1708
|
+
alpha,
|
|
1709
|
+
minN,
|
|
1710
|
+
maxN,
|
|
1711
|
+
maxBet: opts.maxBet ?? .5,
|
|
1712
|
+
scale,
|
|
1713
|
+
shuffleSeed: opts.shuffleSeed ?? 1337,
|
|
1714
|
+
direction,
|
|
1715
|
+
nullMean,
|
|
1716
|
+
minEffect
|
|
1717
|
+
};
|
|
1718
|
+
}
|
|
1719
|
+
function makeStream(cfg) {
|
|
1720
|
+
const proc = eProcess({
|
|
1721
|
+
alpha: cfg.alpha,
|
|
1722
|
+
maxBet: cfg.maxBet,
|
|
1723
|
+
nullMean: cfg.nullMean
|
|
1724
|
+
});
|
|
1725
|
+
const threshold = 1 / cfg.alpha;
|
|
1726
|
+
let terminal;
|
|
1727
|
+
return {
|
|
1728
|
+
observe(delta) {
|
|
1729
|
+
if (terminal === "undecided-at-maxN") throw new Error(`sequentialPairedGate: pre-registered maxN=${cfg.maxN} exhausted — extending the stream after seeing the result reopens optional stopping; start a NEW pre-registered test`);
|
|
1730
|
+
if (!Number.isFinite(delta) || Math.abs(delta) > cfg.scale) throw new Error(`sequentialPairedGate: delta ${delta} outside ±scale=${cfg.scale} — pass the judge's native scale explicitly (detectScale helps pick 1 vs 100)`);
|
|
1731
|
+
const d = cfg.direction === "decrease" ? -delta : delta;
|
|
1732
|
+
const step = proc.update((d / cfg.scale + 1) / 2);
|
|
1733
|
+
if (terminal === void 0 && step.n >= cfg.minN && step.wealth >= threshold) terminal = "promote";
|
|
1734
|
+
if (terminal === "promote") return {
|
|
1735
|
+
decision: "promote",
|
|
1736
|
+
eValue: step.wealth,
|
|
1737
|
+
n: step.n,
|
|
1738
|
+
reason: `e-value ${step.wealth.toFixed(2)} ≥ 1/α=${threshold.toFixed(2)} at n=${step.n} (minN=${cfg.minN}): the paired improvement exceeds ${cfg.minEffect} at anytime-valid level α=${cfg.alpha}`
|
|
1739
|
+
};
|
|
1740
|
+
if (step.n >= cfg.maxN) {
|
|
1741
|
+
terminal = "undecided-at-maxN";
|
|
1742
|
+
return {
|
|
1743
|
+
decision: "undecided-at-maxN",
|
|
1744
|
+
eValue: step.wealth,
|
|
1745
|
+
n: step.n,
|
|
1746
|
+
reason: `undecided at pre-registered maxN=${cfg.maxN} (e-value ${step.wealth.toFixed(2)} < 1/α=${threshold.toFixed(2)}). This is NOT evidence of no effect — the effect may be real but smaller than this budget can detect; re-register with a larger N to test that`
|
|
1747
|
+
};
|
|
1748
|
+
}
|
|
1749
|
+
return {
|
|
1750
|
+
decision: "continue",
|
|
1751
|
+
eValue: step.wealth,
|
|
1752
|
+
n: step.n,
|
|
1753
|
+
reason: `e-value ${step.wealth.toFixed(2)} < 1/α=${threshold.toFixed(2)} at n=${step.n}/${cfg.maxN} — keep observing`
|
|
1754
|
+
};
|
|
1755
|
+
},
|
|
1756
|
+
state() {
|
|
1757
|
+
return {
|
|
1758
|
+
...proc.state(),
|
|
1759
|
+
decision: terminal ?? "continue"
|
|
1760
|
+
};
|
|
1761
|
+
}
|
|
1762
|
+
};
|
|
1763
|
+
}
|
|
1764
|
+
/** Data-independent in-place Fisher–Yates with a seeded PRNG — the permutation
|
|
1765
|
+
* depends only on the seed, never the values, so bet predictability survives. */
|
|
1766
|
+
function seededShuffle(items, seed) {
|
|
1767
|
+
const rng = mulberry32(seed);
|
|
1768
|
+
for (let i = items.length - 1; i > 0; i--) {
|
|
1769
|
+
const j = Math.floor(rng() * (i + 1));
|
|
1770
|
+
const tmp = items[i];
|
|
1771
|
+
items[i] = items[j];
|
|
1772
|
+
items[j] = tmp;
|
|
1773
|
+
}
|
|
1774
|
+
return items;
|
|
1775
|
+
}
|
|
1776
|
+
/**
|
|
1777
|
+
* Anytime-valid sequential paired gate. Conforms to the existing `Gate`
|
|
1778
|
+
* contract (`decide(ctx)` consumes candidate vs baseline judge scores via
|
|
1779
|
+
* `pairHoldout` — same pairing granularity as the fixed-n gates: full cellId,
|
|
1780
|
+
* never scenarioId) and adds a streaming `observe(delta)` entry for campaigns
|
|
1781
|
+
* that score cells incrementally and want to stop mid-stream.
|
|
1782
|
+
*
|
|
1783
|
+
* Decision mapping onto the substrate's five-valued `GateDecision`:
|
|
1784
|
+
* - 'promote' → 'ship'
|
|
1785
|
+
* - 'continue' → 'need_more_work' (stream ended before maxN with
|
|
1786
|
+
* the e-value undecided — more reps could decide)
|
|
1787
|
+
* - 'undecided-at-maxN' → 'hold', with the reason stating it is NOT
|
|
1788
|
+
* evidence of no effect (never a silent default)
|
|
1789
|
+
*/
|
|
1790
|
+
function sequentialPairedGate(options) {
|
|
1791
|
+
const cfg = resolveConfig(options);
|
|
1792
|
+
const name = options.name ?? "sequentialPairedGate";
|
|
1793
|
+
const manifest = options.preRegistration;
|
|
1794
|
+
const observeStream = makeStream(cfg);
|
|
1795
|
+
return {
|
|
1796
|
+
name,
|
|
1797
|
+
observe(delta) {
|
|
1798
|
+
return observeStream.observe(delta);
|
|
1799
|
+
},
|
|
1800
|
+
state() {
|
|
1801
|
+
return observeStream.state();
|
|
1802
|
+
},
|
|
1803
|
+
async decide(ctx) {
|
|
1804
|
+
if (!ctx.baselineJudgeScores) throw new Error(`${name}: ctx.baselineJudgeScores is required — falling back to the candidate's own scores would compare the candidate against itself (delta 0, silent no-op)`);
|
|
1805
|
+
const scenarioIds = new Set(ctx.scenarios.map((s) => s.id));
|
|
1806
|
+
const paired = pairHoldout(ctx.judgeScores, ctx.baselineJudgeScores, scenarioIds, (s) => s.composite);
|
|
1807
|
+
const deltas = paired.after.map((a, i) => a - paired.before[i]);
|
|
1808
|
+
seededShuffle(deltas, cfg.shuffleSeed);
|
|
1809
|
+
const stream = makeStream(cfg);
|
|
1810
|
+
let last;
|
|
1811
|
+
for (const d of deltas) {
|
|
1812
|
+
last = stream.observe(d);
|
|
1813
|
+
if (last.decision !== "continue") break;
|
|
1814
|
+
}
|
|
1815
|
+
const detail = {
|
|
1816
|
+
...stream.state(),
|
|
1817
|
+
minN: cfg.minN,
|
|
1818
|
+
maxN: cfg.maxN,
|
|
1819
|
+
scale: cfg.scale,
|
|
1820
|
+
shuffleSeed: cfg.shuffleSeed,
|
|
1821
|
+
direction: cfg.direction,
|
|
1822
|
+
minEffect: cfg.minEffect,
|
|
1823
|
+
pairedN: deltas.length,
|
|
1824
|
+
...manifest ? {
|
|
1825
|
+
preRegisteredId: manifest.id,
|
|
1826
|
+
metric: manifest.metric
|
|
1827
|
+
} : {}
|
|
1828
|
+
};
|
|
1829
|
+
const meanDelta = deltas.length === 0 ? void 0 : deltas.reduce((s, d) => s + d, 0) / deltas.length;
|
|
1830
|
+
if (last === void 0) return {
|
|
1831
|
+
decision: "need_more_work",
|
|
1832
|
+
reasons: [`${name}: no paired holdout observations — nothing to test`],
|
|
1833
|
+
contributingGates: [{
|
|
1834
|
+
name,
|
|
1835
|
+
status: "not_evaluated",
|
|
1836
|
+
detail
|
|
1837
|
+
}]
|
|
1838
|
+
};
|
|
1839
|
+
const decision = last.decision === "promote" ? "ship" : last.decision === "continue" ? "need_more_work" : "hold";
|
|
1840
|
+
return {
|
|
1841
|
+
decision,
|
|
1842
|
+
reasons: [`${name}: ${last.reason}`],
|
|
1843
|
+
contributingGates: [{
|
|
1844
|
+
name,
|
|
1845
|
+
status: decision === "ship" ? "pass" : decision === "need_more_work" ? "not_evaluated" : "fail",
|
|
1846
|
+
detail
|
|
1847
|
+
}],
|
|
1848
|
+
delta: meanDelta
|
|
1849
|
+
};
|
|
1850
|
+
}
|
|
1851
|
+
};
|
|
1852
|
+
}
|
|
1853
|
+
/**
|
|
1854
|
+
* `SurfaceProposer.decide` adapter — stops the optimization loop the moment
|
|
1855
|
+
* the e-process decides the loop has produced a real improvement, instead of
|
|
1856
|
+
* always running `maxGenerations`.
|
|
1857
|
+
*
|
|
1858
|
+
* Stream: for each generation g ≥ 1, the per-scenario composite deltas of
|
|
1859
|
+
* generation g's top candidate vs the generation-0 top candidate (the
|
|
1860
|
+
* incumbent the loop set out to beat), paired by scenarioId. H0: no proposed
|
|
1861
|
+
* surface improves any scenario's expected composite over the incumbent —
|
|
1862
|
+
* under it every delta has conditional mean ≤ 0 and the e-process is valid.
|
|
1863
|
+
* Once wealth ≥ 1/alpha the loop stops and hands the winner to the promotion
|
|
1864
|
+
* gate (which re-scores on HELD-OUT data — this adapter only spends the
|
|
1865
|
+
* exploration budget, it never promotes).
|
|
1866
|
+
*
|
|
1867
|
+
* Honesty caveats: (1) the incumbent's scores are measured once and shared
|
|
1868
|
+
* across all generations' deltas, so type-I control is exact only insofar as
|
|
1869
|
+
* those scores approximate the incumbent's true per-scenario means (more reps
|
|
1870
|
+
* → tighter); (2) an UNDECIDED process never stops the loop — absence of a
|
|
1871
|
+
* crossing is NOT evidence of no effect, so the loop simply runs its normal
|
|
1872
|
+
* course. Calling the adapter repeatedly with a growing history consumes each
|
|
1873
|
+
* generation exactly once (re-feeding an already-seen record would double-count
|
|
1874
|
+
* evidence).
|
|
1875
|
+
*/
|
|
1876
|
+
function sequentialDecide(options = {}) {
|
|
1877
|
+
const alpha = options.alpha ?? .05;
|
|
1878
|
+
const minN = options.minN ?? 5;
|
|
1879
|
+
const scale = options.scale ?? 1;
|
|
1880
|
+
if (!Number.isFinite(scale) || scale <= 0) throw new Error(`sequentialDecide: scale must be > 0, got ${scale}`);
|
|
1881
|
+
const proc = eProcess({
|
|
1882
|
+
alpha,
|
|
1883
|
+
maxBet: options.maxBet ?? .5,
|
|
1884
|
+
nullMean: .5
|
|
1885
|
+
});
|
|
1886
|
+
const threshold = 1 / alpha;
|
|
1887
|
+
let processedGenerations = 0;
|
|
1888
|
+
let reference;
|
|
1889
|
+
let stopped;
|
|
1890
|
+
const topCandidate = (record) => {
|
|
1891
|
+
if (record.candidates.length === 0) throw new Error(`sequentialDecide: generation ${record.generationIndex} has no candidates — cannot extract a top candidate`);
|
|
1892
|
+
return record.candidates.reduce((best, candidate) => {
|
|
1893
|
+
if (best.composite === null) return candidate;
|
|
1894
|
+
if (candidate.composite === null) return best;
|
|
1895
|
+
return candidate.composite > best.composite ? candidate : best;
|
|
1896
|
+
});
|
|
1897
|
+
};
|
|
1898
|
+
const decide = ({ history }) => {
|
|
1899
|
+
if (stopped) return stopped;
|
|
1900
|
+
if (history.length === 0) return { stop: false };
|
|
1901
|
+
if (reference === void 0) reference = new Map(topCandidate(history[0]).scenarios.map((s) => [s.scenarioId, s.composite]));
|
|
1902
|
+
for (let g = Math.max(1, processedGenerations); g < history.length; g++) {
|
|
1903
|
+
const top = topCandidate(history[g]);
|
|
1904
|
+
const byScenario = new Map(top.scenarios.map((s) => [s.scenarioId, s.composite]));
|
|
1905
|
+
for (const [scenarioId, refComposite] of reference) {
|
|
1906
|
+
const candComposite = byScenario.get(scenarioId);
|
|
1907
|
+
if (candComposite === void 0) throw new Error(`sequentialDecide: generation ${history[g].generationIndex} top candidate is missing scenario '${scenarioId}' — generations must score the same scenario set to pair`);
|
|
1908
|
+
const delta = candComposite - refComposite;
|
|
1909
|
+
if (!Number.isFinite(delta) || Math.abs(delta) > scale) throw new Error(`sequentialDecide: paired delta ${delta} outside ±scale=${scale} on scenario '${scenarioId}' — pass the composite scale explicitly`);
|
|
1910
|
+
const step = proc.update((delta / scale + 1) / 2);
|
|
1911
|
+
if (step.n >= minN && step.wealth >= threshold) {
|
|
1912
|
+
stopped = {
|
|
1913
|
+
stop: true,
|
|
1914
|
+
reason: `sequential e-process decided at generation ${history[g].generationIndex}: e-value ${step.wealth.toFixed(2)} ≥ 1/α=${threshold.toFixed(2)} after n=${step.n} paired deltas vs the generation-0 incumbent — the improvement is real at α=${alpha}; stop exploring and promote via the gate`
|
|
1915
|
+
};
|
|
1916
|
+
processedGenerations = history.length;
|
|
1917
|
+
return stopped;
|
|
1918
|
+
}
|
|
1919
|
+
}
|
|
1920
|
+
}
|
|
1921
|
+
processedGenerations = history.length;
|
|
1922
|
+
return { stop: false };
|
|
1923
|
+
};
|
|
1924
|
+
decide.state = () => proc.state();
|
|
1925
|
+
return decide;
|
|
1926
|
+
}
|
|
1927
|
+
//#endregion
|
|
1928
|
+
//#region src/campaign/grounded-reflection.ts
|
|
1929
|
+
/**
|
|
1930
|
+
* Deterministic per-field diff of call arguments between passing and failing
|
|
1931
|
+
* rollouts. A field set by failing rollouts but left unset by passing ones is
|
|
1932
|
+
* the classic poison-input signature; a field whose values differ across the
|
|
1933
|
+
* split points at the correct value. Feed `text` to the reviser verbatim.
|
|
1934
|
+
*/
|
|
1935
|
+
function rolloutArgumentDiff(rollouts, opts = {}) {
|
|
1936
|
+
const passThreshold = opts.passThreshold ?? 1;
|
|
1937
|
+
const maxValues = opts.maxValuesPerField ?? 4;
|
|
1938
|
+
const collect = (pass) => {
|
|
1939
|
+
const byField = /* @__PURE__ */ new Map();
|
|
1940
|
+
for (const r of rollouts) {
|
|
1941
|
+
if (pass !== r.score >= passThreshold) continue;
|
|
1942
|
+
for (const c of r.calls) for (const [k, v] of Object.entries(c.args)) {
|
|
1943
|
+
const set = byField.get(k) ?? /* @__PURE__ */ new Set();
|
|
1944
|
+
set.add(String(v));
|
|
1945
|
+
byField.set(k, set);
|
|
1946
|
+
}
|
|
1947
|
+
}
|
|
1948
|
+
return byField;
|
|
1949
|
+
};
|
|
1950
|
+
const passing = collect(true);
|
|
1951
|
+
const failing = collect(false);
|
|
1952
|
+
const lines = [];
|
|
1953
|
+
for (const field of /* @__PURE__ */ new Set([...passing.keys(), ...failing.keys()])) {
|
|
1954
|
+
const pv = [...passing.get(field) ?? []].slice(0, maxValues);
|
|
1955
|
+
const fv = [...failing.get(field) ?? []].slice(0, maxValues);
|
|
1956
|
+
const render = (vals) => vals.length ? JSON.stringify(vals) : "NOT SET (omitted)";
|
|
1957
|
+
lines.push(` ${field}: passing runs -> ${render(pv)} | failing runs -> ${render(fv)}`);
|
|
1958
|
+
}
|
|
1959
|
+
const lower = (m) => {
|
|
1960
|
+
const out = /* @__PURE__ */ new Set();
|
|
1961
|
+
for (const vals of m.values()) for (const v of vals) out.add(v.toLowerCase());
|
|
1962
|
+
return out;
|
|
1963
|
+
};
|
|
1964
|
+
return {
|
|
1965
|
+
text: lines.join("\n") || " (no calls observed)",
|
|
1966
|
+
passingValues: lower(passing),
|
|
1967
|
+
failingValues: lower(failing)
|
|
1968
|
+
};
|
|
1969
|
+
}
|
|
1970
|
+
/**
|
|
1971
|
+
* Scan revised artifact text for single-quoted single-word literals (the
|
|
1972
|
+
* "use exactly 'new'" pattern) that appear in no passing rollout's argument
|
|
1973
|
+
* values. Multi-word quotes pass (they are prose, not prescriptions).
|
|
1974
|
+
* Callers should reject on `harmful` (with a bounded retry) and at most log
|
|
1975
|
+
* `ungrounded` - see the module header for why the severities differ.
|
|
1976
|
+
*/
|
|
1977
|
+
function classifyUngroundedLiterals(text, diff) {
|
|
1978
|
+
const ungrounded = /* @__PURE__ */ new Set();
|
|
1979
|
+
for (const m of text.matchAll(/'([a-z][a-z_-]{1,19})'/gi)) {
|
|
1980
|
+
const w = m[1].toLowerCase();
|
|
1981
|
+
if (!diff.passingValues.has(w)) ungrounded.add(w);
|
|
1982
|
+
}
|
|
1983
|
+
const all = [...ungrounded];
|
|
1984
|
+
return {
|
|
1985
|
+
ungrounded: all,
|
|
1986
|
+
harmful: all.filter((w) => diff.failingValues.has(w))
|
|
1987
|
+
};
|
|
1988
|
+
}
|
|
1989
|
+
//#endregion
|
|
1990
|
+
//#region src/campaign/labeled-store/fs-adapter.ts
|
|
1991
|
+
/**
|
|
1992
|
+
* Filesystem `LabeledScenarioStore` adapter. The default capture sink for
|
|
1993
|
+
* traces + eval artifacts. Production deployments typically swap for a
|
|
1994
|
+
* Turso/SQLite adapter (same interface).
|
|
1995
|
+
*
|
|
1996
|
+
* Records land as one JSONL file per source under `<root>/<source>.jsonl`.
|
|
1997
|
+
* Each line is a `LabeledScenarioRecord`. Append-only — no in-place edits.
|
|
1998
|
+
*
|
|
1999
|
+
* Safety properties enforced at write-time:
|
|
2000
|
+
*
|
|
2001
|
+
* - **Provenance required**: writes without `source`, `sourceVersionHash`,
|
|
2002
|
+
* `capturedAt`, `redactionStatus` are rejected. Closes the alignment
|
|
2003
|
+
* reviewer's data-poisoning gap.
|
|
2004
|
+
* - **Per-source rate limits**: optional `rateLimitBucket` + `maxWritesPerMinute`
|
|
2005
|
+
* stops a single tenant/source from flooding the store.
|
|
2006
|
+
*
|
|
2007
|
+
* Safety properties enforced at sample-time:
|
|
2008
|
+
*
|
|
2009
|
+
* - **Required split + capturedBefore**: substrate refuses to sample without
|
|
2010
|
+
* an explicit `split` ('train' | 'test') AND a temporal cutoff. Eliminates
|
|
2011
|
+
* accidental train/test contamination.
|
|
2012
|
+
* - **Default training-source filter**: when the store is sampled with
|
|
2013
|
+
* `split: 'train'`, production-trace records are EXCLUDED unless the
|
|
2014
|
+
* caller passes `filter.source: 'production-trace'` explicitly. Closes
|
|
2015
|
+
* the contamination-by-default gap flagged by the senior eval engineer.
|
|
2016
|
+
*/
|
|
2017
|
+
/** Typed rejection from a labeled-scenario store (bad provenance, rate limit, invalid sample args) — carries a stable string `code`. */
|
|
2018
|
+
var LabeledScenarioStoreError = class extends Error {
|
|
2019
|
+
code;
|
|
2020
|
+
constructor(code, message) {
|
|
2021
|
+
super(message);
|
|
2022
|
+
this.code = code;
|
|
2023
|
+
this.name = "LabeledScenarioStoreError";
|
|
2024
|
+
}
|
|
2025
|
+
};
|
|
2026
|
+
/**
|
|
2027
|
+
* Filesystem `LabeledScenarioStore`: appends one JSONL file per source with provenance and
|
|
2028
|
+
* rate-limit guards. For tests, local dev, and small workloads — high-throughput lands in Turso.
|
|
2029
|
+
*/
|
|
2030
|
+
var FsLabeledScenarioStore = class {
|
|
2031
|
+
options;
|
|
2032
|
+
now;
|
|
2033
|
+
rateLimits = /* @__PURE__ */ new Map();
|
|
2034
|
+
constructor(options) {
|
|
2035
|
+
this.options = options;
|
|
2036
|
+
if (!existsSync(options.root)) mkdirSync(options.root, { recursive: true });
|
|
2037
|
+
this.now = options.now ?? Date.now;
|
|
2038
|
+
}
|
|
2039
|
+
async observe(write) {
|
|
2040
|
+
this.assertProvenance(write);
|
|
2041
|
+
this.assertRateLimit(write);
|
|
2042
|
+
const record = this.toRecord(write);
|
|
2043
|
+
appendLine(this.pathForSource(write.source), `${JSON.stringify(record)}\n`);
|
|
2044
|
+
}
|
|
2045
|
+
async sample(args) {
|
|
2046
|
+
if (!args.split) throw new LabeledScenarioStoreError("split_required", "sample() requires an explicit `split` (train | test) — substrate refuses ambiguous reads");
|
|
2047
|
+
if (!args.capturedBefore) throw new LabeledScenarioStoreError("capturedBefore_required", "sample() requires an explicit `capturedBefore` timestamp for temporal-split discipline");
|
|
2048
|
+
const all = [];
|
|
2049
|
+
for (const source of ALL_SOURCES) {
|
|
2050
|
+
if (args.split === "train" && source === "production-trace") {
|
|
2051
|
+
if (!sourceFilterContains(args.filter?.source, "production-trace")) continue;
|
|
2052
|
+
}
|
|
2053
|
+
const path = this.pathForSource(source);
|
|
2054
|
+
if (!existsSync(path)) continue;
|
|
2055
|
+
const lines = readFileSync(path, "utf8").split("\n").filter(Boolean);
|
|
2056
|
+
for (const line of lines) {
|
|
2057
|
+
let record;
|
|
2058
|
+
try {
|
|
2059
|
+
record = JSON.parse(line);
|
|
2060
|
+
} catch {
|
|
2061
|
+
continue;
|
|
2062
|
+
}
|
|
2063
|
+
if (!matchesFilter(record, args, source)) continue;
|
|
2064
|
+
all.push(record);
|
|
2065
|
+
}
|
|
2066
|
+
}
|
|
2067
|
+
all.sort((a, b) => {
|
|
2068
|
+
if (a.capturedAt !== b.capturedAt) return a.capturedAt.localeCompare(b.capturedAt);
|
|
2069
|
+
return a.recordHash.localeCompare(b.recordHash);
|
|
2070
|
+
});
|
|
2071
|
+
return all.slice(0, args.count);
|
|
2072
|
+
}
|
|
2073
|
+
async size() {
|
|
2074
|
+
const bySource = {};
|
|
2075
|
+
const byTrust = {
|
|
2076
|
+
unverified: 0,
|
|
2077
|
+
"verified-signal": 0,
|
|
2078
|
+
"human-rated": 0
|
|
2079
|
+
};
|
|
2080
|
+
let total = 0;
|
|
2081
|
+
for (const source of ALL_SOURCES) {
|
|
2082
|
+
const path = this.pathForSource(source);
|
|
2083
|
+
if (!existsSync(path)) {
|
|
2084
|
+
bySource[source] = 0;
|
|
2085
|
+
continue;
|
|
2086
|
+
}
|
|
2087
|
+
const lines = readFileSync(path, "utf8").split("\n").filter(Boolean);
|
|
2088
|
+
bySource[source] = lines.length;
|
|
2089
|
+
total += lines.length;
|
|
2090
|
+
for (const line of lines) {
|
|
2091
|
+
let trust = "unverified";
|
|
2092
|
+
try {
|
|
2093
|
+
trust = JSON.parse(line).labelTrust ?? "unverified";
|
|
2094
|
+
} catch {}
|
|
2095
|
+
byTrust[trust] += 1;
|
|
2096
|
+
}
|
|
2097
|
+
}
|
|
2098
|
+
return {
|
|
2099
|
+
train: total,
|
|
2100
|
+
test: total,
|
|
2101
|
+
bySource,
|
|
2102
|
+
byTrust
|
|
2103
|
+
};
|
|
2104
|
+
}
|
|
2105
|
+
assertProvenance(write) {
|
|
2106
|
+
if (!write.source) throw new LabeledScenarioStoreError("missing_source", "LabeledScenarioWrite requires `source`");
|
|
2107
|
+
if (!write.sourceVersionHash || write.sourceVersionHash.length === 0) throw new LabeledScenarioStoreError("missing_source_version", "LabeledScenarioWrite requires `sourceVersionHash` (git sha or substrate version)");
|
|
2108
|
+
if (!write.capturedAt) throw new LabeledScenarioStoreError("missing_captured_at", "LabeledScenarioWrite requires `capturedAt` ISO timestamp");
|
|
2109
|
+
if (!write.redactionStatus) throw new LabeledScenarioStoreError("missing_redaction_status", "LabeledScenarioWrite requires explicit `redactionStatus` — raw / redacted-pii / redacted-secrets / fully-redacted");
|
|
2110
|
+
if (!ALL_SOURCES.includes(write.source)) throw new LabeledScenarioStoreError("unknown_source", `LabeledScenarioWrite.source must be one of: ${ALL_SOURCES.join(", ")}`);
|
|
2111
|
+
}
|
|
2112
|
+
assertRateLimit(write) {
|
|
2113
|
+
const cap = this.options.maxWritesPerMinutePerBucket;
|
|
2114
|
+
if (!cap || !write.rateLimitBucket) return;
|
|
2115
|
+
const now = this.now();
|
|
2116
|
+
const windowMs = 6e4;
|
|
2117
|
+
let state = this.rateLimits.get(write.rateLimitBucket);
|
|
2118
|
+
if (!state || now - state.windowStartMs >= windowMs) {
|
|
2119
|
+
state = {
|
|
2120
|
+
bucket: write.rateLimitBucket,
|
|
2121
|
+
windowStartMs: now,
|
|
2122
|
+
count: 0
|
|
2123
|
+
};
|
|
2124
|
+
this.rateLimits.set(write.rateLimitBucket, state);
|
|
2125
|
+
}
|
|
2126
|
+
if (state.count >= cap) throw new LabeledScenarioStoreError("rate_limit_exceeded", `LabeledScenarioStore: bucket ${write.rateLimitBucket} exceeded ${cap} writes/min`);
|
|
2127
|
+
state.count += 1;
|
|
2128
|
+
}
|
|
2129
|
+
toRecord(write) {
|
|
2130
|
+
const recordHash = sha256$1(JSON.stringify({
|
|
2131
|
+
id: write.scenario.id,
|
|
2132
|
+
src: write.source,
|
|
2133
|
+
at: write.capturedAt,
|
|
2134
|
+
ver: write.sourceVersionHash
|
|
2135
|
+
}));
|
|
2136
|
+
return {
|
|
2137
|
+
...write,
|
|
2138
|
+
recordHash,
|
|
2139
|
+
split: "train"
|
|
2140
|
+
};
|
|
2141
|
+
}
|
|
2142
|
+
pathForSource(source) {
|
|
2143
|
+
return join(this.options.root, `${source}.jsonl`);
|
|
2144
|
+
}
|
|
2145
|
+
};
|
|
2146
|
+
const ALL_SOURCES = [
|
|
2147
|
+
"production-trace",
|
|
2148
|
+
"eval-run",
|
|
2149
|
+
"manual",
|
|
2150
|
+
"red-team",
|
|
2151
|
+
"synthetic"
|
|
2152
|
+
];
|
|
2153
|
+
function sourceFilterContains(filter, needle) {
|
|
2154
|
+
if (!filter) return false;
|
|
2155
|
+
if (Array.isArray(filter)) return filter.includes(needle);
|
|
2156
|
+
return filter === needle;
|
|
2157
|
+
}
|
|
2158
|
+
function matchesFilter(record, args, source) {
|
|
2159
|
+
if (args.split === "train" && record.capturedAt >= args.capturedBefore) return false;
|
|
2160
|
+
if (args.split === "test" && record.capturedAt < args.capturedBefore) return false;
|
|
2161
|
+
const f = args.filter;
|
|
2162
|
+
if (!f) return true;
|
|
2163
|
+
if (f.kind && record.scenario.kind !== f.kind) return false;
|
|
2164
|
+
if (f.source) {
|
|
2165
|
+
if (!(Array.isArray(f.source) ? f.source : [f.source]).includes(source)) return false;
|
|
2166
|
+
}
|
|
2167
|
+
if (f.minComposite !== void 0 || f.maxComposite !== void 0) {
|
|
2168
|
+
const composites = Object.values(record.judgeScores).map((s) => s.composite);
|
|
2169
|
+
const max = composites.length === 0 ? 0 : Math.max(...composites);
|
|
2170
|
+
if (f.minComposite !== void 0 && max < f.minComposite) return false;
|
|
2171
|
+
if (f.maxComposite !== void 0 && max > f.maxComposite) return false;
|
|
2172
|
+
}
|
|
2173
|
+
if (f.minTrust !== void 0 && labelTrustRank(record.labelTrust) < labelTrustRank(f.minTrust)) return false;
|
|
2174
|
+
return true;
|
|
2175
|
+
}
|
|
2176
|
+
function sha256$1(input) {
|
|
2177
|
+
return createHash("sha256").update(input).digest("hex").slice(0, 16);
|
|
2178
|
+
}
|
|
2179
|
+
function appendLine(path, line) {
|
|
2180
|
+
if (existsSync(path)) writeFileSync(path, readFileSync(path, "utf8") + line);
|
|
2181
|
+
else writeFileSync(path, line);
|
|
2182
|
+
}
|
|
2183
|
+
//#endregion
|
|
2184
|
+
//#region src/campaign/neutralize.ts
|
|
2185
|
+
/**
|
|
2186
|
+
* @module
|
|
2187
|
+
* Footprint-matched neutralization — the placebo control for content-vs-footprint
|
|
2188
|
+
* attribution in a promotion gate.
|
|
2189
|
+
*
|
|
2190
|
+
* A promoted surface can raise a held-out score two different ways:
|
|
2191
|
+
* 1. its CONTENT is informative (the thing we want to promote), or
|
|
2192
|
+
* 2. it merely added prompt/mount FOOTPRINT — more bytes, more lines, a longer
|
|
2193
|
+
* more authoritative-looking prompt — that the model spends attention on
|
|
2194
|
+
* regardless of what the bytes say.
|
|
2195
|
+
*
|
|
2196
|
+
* A held-out gate proves the candidate beat baseline; it cannot separate (1) from
|
|
2197
|
+
* (2). `neutralizeText` produces a variant that keeps the input's layout and
|
|
2198
|
+
* length while carrying ZERO information, so scoring it isolates the footprint
|
|
2199
|
+
* contribution (2). Feed the neutralized variant's scores to `neutralizationGate`:
|
|
2200
|
+
* any lift it still holds over baseline is decorative, and a candidate whose lift
|
|
2201
|
+
* survives neutralization is rejected however large its raw lift.
|
|
2202
|
+
*/
|
|
2203
|
+
/** Filler for blanked content. A single ASCII byte, so a run of it preserves an
|
|
2204
|
+
* ASCII source's exact byte length; for multibyte sources it preserves CHARACTER
|
|
2205
|
+
* count and layout (what the tokenizer footprint tracks), not raw byte count. */
|
|
2206
|
+
const FILLER = "#";
|
|
2207
|
+
/**
|
|
2208
|
+
* Blank every non-whitespace character to a 1-byte filler while preserving all
|
|
2209
|
+
* whitespace. Line count, indentation, and word/line lengths are unchanged — so
|
|
2210
|
+
* the neutralized variant has the same layout and (for ASCII) the same byte
|
|
2211
|
+
* footprint as the input, but no readable content. Whitespace is preserved
|
|
2212
|
+
* deliberately: collapsing it would change the token structure and stop the
|
|
2213
|
+
* variant from being a true footprint match.
|
|
2214
|
+
*/
|
|
2215
|
+
function neutralizeText(content) {
|
|
2216
|
+
return content.replace(/\S/g, FILLER);
|
|
2217
|
+
}
|
|
2218
|
+
//#endregion
|
|
2219
|
+
//#region src/campaign/presets/playback.ts
|
|
2220
|
+
/**
|
|
2221
|
+
* Adapt a `PlaybackDriver` into a `runProfileMatrix` dispatch. The artifact the
|
|
2222
|
+
* matrix scores is the `ProducedState` extracted from the driver's event
|
|
2223
|
+
* stream — grade it with `scoreUserStory` (or a judge wrapping it).
|
|
2224
|
+
*/
|
|
2225
|
+
function makePlaybackDispatch(driver) {
|
|
2226
|
+
return async (profile, scenario, ctx) => {
|
|
2227
|
+
return extractProducedState(await driver.run(scenario, {
|
|
2228
|
+
...ctx,
|
|
2229
|
+
profile
|
|
2230
|
+
}));
|
|
2231
|
+
};
|
|
2232
|
+
}
|
|
2233
|
+
/**
|
|
2234
|
+
* Score one story's produced state against its requirements. Thin wrapper over
|
|
2235
|
+
* `verifyCompletion` that builds the gold from the story and returns a
|
|
2236
|
+
* per-requirement PASS/FAIL verdict. `checkCorrectness` is injected — a
|
|
2237
|
+
* deterministic stub in tests, `createLlmCorrectnessChecker` in production.
|
|
2238
|
+
*/
|
|
2239
|
+
async function scoreUserStory(story, state, checkCorrectness) {
|
|
2240
|
+
return {
|
|
2241
|
+
...await verifyCompletion({
|
|
2242
|
+
taskId: story.id,
|
|
2243
|
+
requirements: story.requirements
|
|
2244
|
+
}, state, checkCorrectness),
|
|
2245
|
+
title: story.title
|
|
2246
|
+
};
|
|
2247
|
+
}
|
|
2248
|
+
/**
|
|
2249
|
+
* Flatten story verdicts into the per-requirement scoreboard — the literal
|
|
2250
|
+
* Jira tick-off: one row per (story, requirement) with PASS/FAIL and the
|
|
2251
|
+
* evidence behind the verdict.
|
|
2252
|
+
*/
|
|
2253
|
+
function userStoryScoreboard(verdicts) {
|
|
2254
|
+
const rows = [];
|
|
2255
|
+
for (const v of verdicts) for (const r of v.requirements) rows.push({
|
|
2256
|
+
storyId: v.taskId,
|
|
2257
|
+
storyTitle: v.title,
|
|
2258
|
+
reqId: r.reqId,
|
|
2259
|
+
reqTitle: r.title,
|
|
2260
|
+
status: r.satisfied ? "PASS" : "FAIL",
|
|
2261
|
+
evidence: r.evidence
|
|
2262
|
+
});
|
|
2263
|
+
return rows;
|
|
2264
|
+
}
|
|
2265
|
+
/** Roll the per-requirement rows up into the launch headline counts. */
|
|
2266
|
+
function scoreboardSummary(rows) {
|
|
2267
|
+
const byStory = /* @__PURE__ */ new Map();
|
|
2268
|
+
let passed = 0;
|
|
2269
|
+
for (const r of rows) {
|
|
2270
|
+
const s = byStory.get(r.storyId) ?? {
|
|
2271
|
+
total: 0,
|
|
2272
|
+
passed: 0
|
|
2273
|
+
};
|
|
2274
|
+
s.total++;
|
|
2275
|
+
if (r.status === "PASS") {
|
|
2276
|
+
s.passed++;
|
|
2277
|
+
passed++;
|
|
2278
|
+
}
|
|
2279
|
+
byStory.set(r.storyId, s);
|
|
2280
|
+
}
|
|
2281
|
+
let storiesFullyComplete = 0;
|
|
2282
|
+
for (const s of byStory.values()) if (s.total > 0 && s.passed === s.total) storiesFullyComplete++;
|
|
2283
|
+
return {
|
|
2284
|
+
stories: byStory.size,
|
|
2285
|
+
storiesFullyComplete,
|
|
2286
|
+
requirements: rows.length,
|
|
2287
|
+
passed,
|
|
2288
|
+
failed: rows.length - passed,
|
|
2289
|
+
passRate: rows.length === 0 ? 0 : passed / rows.length
|
|
2290
|
+
};
|
|
2291
|
+
}
|
|
2292
|
+
function escapeCell(s) {
|
|
2293
|
+
return s.replace(/\|/g, "\\|").replace(/\r?\n/g, " ");
|
|
2294
|
+
}
|
|
2295
|
+
function truncate(s, max) {
|
|
2296
|
+
return s.length <= max ? s : `${s.slice(0, Math.max(0, max - 1))}…`;
|
|
2297
|
+
}
|
|
2298
|
+
/**
|
|
2299
|
+
* Render the scoreboard as a launch-readiness Markdown document — the literal
|
|
2300
|
+
* "tick off every user story" artifact: a headline roll-up, the open tickets
|
|
2301
|
+
* (FAIL rows) up top as the launch blockers, then a per-story table of
|
|
2302
|
+
* requirement → PASS/FAIL with the evidence behind each verdict. Pure: same
|
|
2303
|
+
* rows in, same bytes out (no clock/random), so it is safe to snapshot.
|
|
2304
|
+
*/
|
|
2305
|
+
function renderScoreboardMarkdown(rows, opts = {}) {
|
|
2306
|
+
const maxEv = opts.maxEvidenceChars ?? 160;
|
|
2307
|
+
const sum = scoreboardSummary(rows);
|
|
2308
|
+
const pct = (n) => `${Math.round(n * 100)}%`;
|
|
2309
|
+
const ev = (e) => escapeCell(truncate(e.join("; "), maxEv)) || "—";
|
|
2310
|
+
const out = [`# ${opts.title ?? "Product-flow playback scoreboard"}`, ""];
|
|
2311
|
+
if (opts.meta) {
|
|
2312
|
+
for (const [k, v] of Object.entries(opts.meta)) out.push(`- **${k}:** ${v}`);
|
|
2313
|
+
out.push("");
|
|
2314
|
+
}
|
|
2315
|
+
out.push(`**${sum.storiesFullyComplete}/${sum.stories}** user stories fully shipped · **${sum.passed}/${sum.requirements}** requirements passing (${pct(sum.passRate)}) · **${sum.failed}** open`, "");
|
|
2316
|
+
const fails = rows.filter((r) => r.status === "FAIL");
|
|
2317
|
+
if (fails.length > 0) {
|
|
2318
|
+
out.push("## Open tickets", "", "| Story | Requirement | Evidence |", "| --- | --- | --- |");
|
|
2319
|
+
for (const r of fails) out.push(`| ${escapeCell(r.storyTitle)} | ${escapeCell(r.reqTitle)} | ${ev(r.evidence)} |`);
|
|
2320
|
+
out.push("");
|
|
2321
|
+
} else out.push("_All requirements passing — no open tickets._", "");
|
|
2322
|
+
out.push("## Per-story tick-off", "");
|
|
2323
|
+
for (const storyId of [...new Set(rows.map((r) => r.storyId))]) {
|
|
2324
|
+
const storyRows = rows.filter((r) => r.storyId === storyId);
|
|
2325
|
+
const passed = storyRows.filter((r) => r.status === "PASS").length;
|
|
2326
|
+
const mark = passed === storyRows.length ? "✅" : "⚠️";
|
|
2327
|
+
out.push(`### ${escapeCell(storyRows[0].storyTitle)} — ${passed}/${storyRows.length} ${mark}`, "", "| Requirement | Status | Evidence |", "| --- | --- | --- |");
|
|
2328
|
+
for (const r of storyRows) out.push(`| ${escapeCell(r.reqTitle)} | ${r.status === "PASS" ? "✅ PASS" : "❌ FAIL"} | ${ev(r.evidence)} |`);
|
|
2329
|
+
out.push("");
|
|
2330
|
+
}
|
|
2331
|
+
return out.join("\n");
|
|
2332
|
+
}
|
|
2333
|
+
//#endregion
|
|
2334
|
+
//#region src/campaign/presets/run-profile-matrix.ts
|
|
2335
|
+
/**
|
|
2336
|
+
* `runProfileMatrix` — the missing keystone between `runAgentMatrix` and the
|
|
2337
|
+
* backend-integrity guard.
|
|
2338
|
+
*
|
|
2339
|
+
* The gap it closes: `runAgentMatrix` is a topology-opaque scheduler whose
|
|
2340
|
+
* cells return a bare `{ output, verdict, costUsd }` — no `tokenUsage`, not a
|
|
2341
|
+
* `RunRecord`. `assertRealBackend` / `summarizeBackendIntegrity` key on
|
|
2342
|
+
* `RunRecord.tokenUsage`, so they cannot run on a raw matrix result. Every
|
|
2343
|
+
* consumer therefore hand-writes the same bridge: fan a profile × scenario
|
|
2344
|
+
* cartesian, call dispatch, fabricate a `RunRecord` with token usage, thread it
|
|
2345
|
+
* back, run the integrity guard. That hand-rolled bridge is exactly the pile of
|
|
2346
|
+
* bespoke `eval:*` scripts the adoption skills keep trying (and failing) to
|
|
2347
|
+
* forbid.
|
|
2348
|
+
*
|
|
2349
|
+
* `runProfileMatrix` IS that bridge, once:
|
|
2350
|
+
*
|
|
2351
|
+
* - axis 3 (PROFILE) = `profiles: AgentProfile[]`
|
|
2352
|
+
* - axis 1 (PERSONA/SCENARIO) = `scenarios: Scenario[]` (each scenario carries
|
|
2353
|
+
* its persona; `personaOf` groups them for the `byPersona` pivot)
|
|
2354
|
+
* - the scoring axis = `judges`
|
|
2355
|
+
*
|
|
2356
|
+
* It runs `runCampaign` once per profile (reusing its seeds, reps, bootstrap
|
|
2357
|
+
* CIs, resumability, and the `LabeledScenarioStore` capture flywheel), maps
|
|
2358
|
+
* every cell to a validated `RunRecord` carrying the real `tokenUsage` the
|
|
2359
|
+
* dispatch committed via `ctx.cost.runPaidCall`, and runs `assertRealBackend`
|
|
2360
|
+
* BY CONSTRUCTION before returning — so a stub-backend run fails loudly instead
|
|
2361
|
+
* of reporting a clean 0/N leaderboard.
|
|
2362
|
+
*
|
|
2363
|
+
* Dispatch contract: a dispatch that calls an LLM MUST report usage via
|
|
2364
|
+
* `ctx.cost.runPaidCall({ execute, receipt })`.
|
|
2365
|
+
* A dispatch that reports zero tokens is indistinguishable from a stub and the
|
|
2366
|
+
* integrity guard treats it as one.
|
|
2367
|
+
*/
|
|
2368
|
+
/** Thrown when the matrix is misconfigured (no profiles, a profile whose model
|
|
2369
|
+
* lacks a snapshot version, etc.). Distinct from `BackendIntegrityError`,
|
|
2370
|
+
* which signals a stub backend at run time. */
|
|
2371
|
+
var ProfileMatrixError = class extends AgentEvalError {
|
|
2372
|
+
constructor(message) {
|
|
2373
|
+
super("profile_matrix", message);
|
|
2374
|
+
}
|
|
2375
|
+
};
|
|
2376
|
+
function sanitize(id) {
|
|
2377
|
+
return id.replace(/[^a-zA-Z0-9_-]/g, "_");
|
|
2378
|
+
}
|
|
2379
|
+
function sha(input) {
|
|
2380
|
+
return createHash("sha256").update(JSON.stringify(input)).digest("hex");
|
|
2381
|
+
}
|
|
2382
|
+
function mean(xs) {
|
|
2383
|
+
return xs.length === 0 ? 0 : xs.reduce((a, b) => a + b, 0) / xs.length;
|
|
2384
|
+
}
|
|
2385
|
+
/**
|
|
2386
|
+
* Resolve the concrete, snapshot-bearing model for a cell whose profile
|
|
2387
|
+
* declared the `HARNESS_NATIVE_MODEL` sentinel (a vendor-locked harness that
|
|
2388
|
+
* resolves its model at runtime). The dispatch must have reported it via
|
|
2389
|
+
* the model in a paid-call receipt — surfaced as `cell.resolvedModel`. Throws when it is
|
|
2390
|
+
* missing or lacks a snapshot, so a provenance-broken row can never be
|
|
2391
|
+
* recorded as the bare sentinel.
|
|
2392
|
+
*/
|
|
2393
|
+
function requireResolvedModel(cell, profileId) {
|
|
2394
|
+
const resolved = cell.resolvedModel?.trim();
|
|
2395
|
+
if (!resolved) throw new ProfileMatrixError(`profile '${profileId}' declared the '${HARNESS_NATIVE_MODEL}' runtime-resolved model but its dispatch reported no resolved model for cell '${cell.cellId}' — return it in the ctx.cost.runPaidCall receipt so the RunRecord pins the real model (never records '${HARNESS_NATIVE_MODEL}')`);
|
|
2396
|
+
if (!modelHasSnapshot(resolved)) throw new ProfileMatrixError(`profile '${profileId}' resolved to model '${resolved}' for cell '${cell.cellId}', which lacks a snapshot version — pin it (name@YYYY-MM-DD or name-YYYYMMDD) in the paid-call receipt`);
|
|
2397
|
+
return resolved;
|
|
2398
|
+
}
|
|
2399
|
+
function buildRunRecord(args) {
|
|
2400
|
+
const { cell, profile, profileHash, configHash, experimentId, splitTag, commitSha, matrixId } = args;
|
|
2401
|
+
const profileId = agentProfileId(profile);
|
|
2402
|
+
const declaredModel = agentProfileModelId(profile);
|
|
2403
|
+
const model = declaredModel === "default" ? requireResolvedModel(cell, profileId) : declaredModel;
|
|
2404
|
+
const record = campaignCellToRunRecord(cell, {
|
|
2405
|
+
runId: `${matrixId}:${profileId}:${cell.cellId}`,
|
|
2406
|
+
experimentId,
|
|
2407
|
+
candidateId: profileId,
|
|
2408
|
+
model,
|
|
2409
|
+
promptHash: profileHash,
|
|
2410
|
+
configHash,
|
|
2411
|
+
commitSha,
|
|
2412
|
+
splitTag,
|
|
2413
|
+
agentProfile: args.agentProfileCell
|
|
2414
|
+
});
|
|
2415
|
+
if (args.corpusText && args.scenario) try {
|
|
2416
|
+
const text = args.corpusText(cell.artifact, args.scenario);
|
|
2417
|
+
if (text && typeof text.prompt === "string" && typeof text.completion === "string") {
|
|
2418
|
+
record.prompt = text.prompt;
|
|
2419
|
+
record.completion = text.completion;
|
|
2420
|
+
}
|
|
2421
|
+
} catch {}
|
|
2422
|
+
return record;
|
|
2423
|
+
}
|
|
2424
|
+
/**
|
|
2425
|
+
* Profile × scenario matrix runner: fan N agent profiles across M scenarios, project each cell to a validated `RunRecord` with real token usage, and enforce the backend-integrity guard before returning.
|
|
2426
|
+
*/
|
|
2427
|
+
async function runProfileMatrix(opts) {
|
|
2428
|
+
if (opts.profiles.length === 0) throw new ProfileMatrixError("profiles must not be empty");
|
|
2429
|
+
if (opts.scenarios.length === 0) throw new ProfileMatrixError("scenarios must not be empty");
|
|
2430
|
+
const splitTag = opts.splitTag ?? "search";
|
|
2431
|
+
const seed = opts.seed ?? 42;
|
|
2432
|
+
const validate = opts.validate ?? true;
|
|
2433
|
+
const integrityMode = opts.integrity ?? "assert";
|
|
2434
|
+
const profileIds = opts.profiles.map(agentProfileId);
|
|
2435
|
+
const experimentId = opts.experimentId ?? `pm_${sha({
|
|
2436
|
+
profileIds,
|
|
2437
|
+
scenarios: opts.scenarios.map((s) => s.id)
|
|
2438
|
+
}).slice(0, 16)}`;
|
|
2439
|
+
const matrixId = `mtx_${sha({
|
|
2440
|
+
experimentId,
|
|
2441
|
+
profileIds,
|
|
2442
|
+
seed,
|
|
2443
|
+
splitTag
|
|
2444
|
+
}).slice(0, 16)}`;
|
|
2445
|
+
const scenarioById = new Map(opts.scenarios.map((s) => [s.id, s]));
|
|
2446
|
+
for (const profile of opts.profiles) {
|
|
2447
|
+
const profileHash = agentProfileHash(profile);
|
|
2448
|
+
const profileId = agentProfileId(profile);
|
|
2449
|
+
const declaredModel = agentProfileModelId(profile);
|
|
2450
|
+
const model = declaredModel === "default" ? `${HARNESS_NATIVE_MODEL}@runtime-resolved` : declaredModel;
|
|
2451
|
+
try {
|
|
2452
|
+
validateRunRecord({
|
|
2453
|
+
runId: `${matrixId}:${profileId}:probe`,
|
|
2454
|
+
experimentId,
|
|
2455
|
+
candidateId: profileId,
|
|
2456
|
+
seed,
|
|
2457
|
+
model,
|
|
2458
|
+
promptHash: profileHash,
|
|
2459
|
+
configHash: profileHash,
|
|
2460
|
+
commitSha: opts.commitSha,
|
|
2461
|
+
wallMs: 0,
|
|
2462
|
+
costUsd: null,
|
|
2463
|
+
costProvenance: {
|
|
2464
|
+
kind: "uncaptured",
|
|
2465
|
+
usd: null
|
|
2466
|
+
},
|
|
2467
|
+
tokenUsage: {
|
|
2468
|
+
input: 0,
|
|
2469
|
+
output: 0
|
|
2470
|
+
},
|
|
2471
|
+
terminalOutcome: "succeeded",
|
|
2472
|
+
outcome: { raw: { execution_error_count: 0 } },
|
|
2473
|
+
splitTag,
|
|
2474
|
+
scenarioId: "recordability-probe"
|
|
2475
|
+
});
|
|
2476
|
+
} catch (err) {
|
|
2477
|
+
throw new ProfileMatrixError(`profile '${profileId}' is not recordable: ${err instanceof Error ? err.message : String(err)}`);
|
|
2478
|
+
}
|
|
2479
|
+
}
|
|
2480
|
+
const records = [];
|
|
2481
|
+
const campaigns = {};
|
|
2482
|
+
const byProfile = {};
|
|
2483
|
+
for (const profile of opts.profiles) {
|
|
2484
|
+
const profileHash = agentProfileHash(profile);
|
|
2485
|
+
const profileId = agentProfileId(profile);
|
|
2486
|
+
const declaredModel = agentProfileModelId(profile);
|
|
2487
|
+
const configHash = sha({
|
|
2488
|
+
profile: profileHash,
|
|
2489
|
+
judges: (opts.judges ?? []).map((j) => j.name),
|
|
2490
|
+
seed,
|
|
2491
|
+
splitTag
|
|
2492
|
+
});
|
|
2493
|
+
const dispatch = (scenario, ctx) => opts.dispatch(profile, scenario, ctx);
|
|
2494
|
+
Object.defineProperty(dispatch, "name", { value: `profile_${sanitize(profileId)}` });
|
|
2495
|
+
const campaign = await runCampaign({
|
|
2496
|
+
scenarios: opts.scenarios,
|
|
2497
|
+
dispatch,
|
|
2498
|
+
judges: opts.judges,
|
|
2499
|
+
seed,
|
|
2500
|
+
reps: opts.reps,
|
|
2501
|
+
maxConcurrency: opts.maxConcurrency,
|
|
2502
|
+
costCeiling: opts.costCeiling,
|
|
2503
|
+
labeledStore: opts.labeledStore,
|
|
2504
|
+
captureSource: opts.captureSource,
|
|
2505
|
+
storage: opts.storage,
|
|
2506
|
+
now: opts.now,
|
|
2507
|
+
runDir: join(opts.runDir, sanitize(profileId))
|
|
2508
|
+
});
|
|
2509
|
+
const axis = harnessAxisOf(profile);
|
|
2510
|
+
const buildCellIdentity = (cellModel) => buildAgentProfileCell({
|
|
2511
|
+
profileId,
|
|
2512
|
+
sourceProfile: {
|
|
2513
|
+
kind: "agent-interface-profile",
|
|
2514
|
+
hash: profileHash
|
|
2515
|
+
},
|
|
2516
|
+
model: cellModel,
|
|
2517
|
+
...axis ? { harness: { id: axis.harness } } : {}
|
|
2518
|
+
});
|
|
2519
|
+
const sharedCellIdentity = declaredModel === "default" ? void 0 : await buildCellIdentity(declaredModel);
|
|
2520
|
+
const profileRecords = [];
|
|
2521
|
+
for (const cell of campaign.cells) {
|
|
2522
|
+
const agentProfileCell = sharedCellIdentity ?? await buildCellIdentity(requireResolvedModel(cell, profileId));
|
|
2523
|
+
const record = buildRunRecord({
|
|
2524
|
+
cell,
|
|
2525
|
+
profile,
|
|
2526
|
+
profileHash,
|
|
2527
|
+
configHash,
|
|
2528
|
+
experimentId,
|
|
2529
|
+
splitTag,
|
|
2530
|
+
commitSha: opts.commitSha,
|
|
2531
|
+
matrixId,
|
|
2532
|
+
agentProfileCell,
|
|
2533
|
+
scenario: scenarioById.get(cell.scenarioId),
|
|
2534
|
+
corpusText: opts.corpusText
|
|
2535
|
+
});
|
|
2536
|
+
if (validate) validateRunRecord(record);
|
|
2537
|
+
profileRecords.push(record);
|
|
2538
|
+
records.push(record);
|
|
2539
|
+
}
|
|
2540
|
+
const totalCostUsd = campaign.aggregates.cost.totalCostUsd;
|
|
2541
|
+
campaigns[profileId] = campaign;
|
|
2542
|
+
byProfile[profileId] = {
|
|
2543
|
+
profileId,
|
|
2544
|
+
profileHash,
|
|
2545
|
+
model: declaredModel === "default" ? profileRecords[0]?.model ?? declaredModel : declaredModel,
|
|
2546
|
+
records: profileRecords.length,
|
|
2547
|
+
meanComposite: meanOrNull(profileRecords.map(scoreOf).filter((score) => score !== void 0)),
|
|
2548
|
+
totalCostUsd,
|
|
2549
|
+
integrity: summarizeBackendIntegrity(profileRecords)
|
|
2550
|
+
};
|
|
2551
|
+
}
|
|
2552
|
+
const integrity = summarizeBackendIntegrity(records);
|
|
2553
|
+
if (integrityMode === "assert") assertRealBackend(records, { allowMixed: opts.allowMixed ?? true });
|
|
2554
|
+
else if (integrityMode === "warn" && integrity.verdict !== "real") console.warn(`[runProfileMatrix] backend integrity: ${integrity.verdict} — ${integrity.diagnosis}`);
|
|
2555
|
+
return {
|
|
2556
|
+
matrixId,
|
|
2557
|
+
experimentId,
|
|
2558
|
+
records,
|
|
2559
|
+
byProfile,
|
|
2560
|
+
byScenario: rollup(records, (r) => r.scenarioId),
|
|
2561
|
+
byPersona: opts.personaOf ? rollupByPersona(records, opts.scenarios, opts.personaOf) : void 0,
|
|
2562
|
+
integrity,
|
|
2563
|
+
campaigns
|
|
2564
|
+
};
|
|
2565
|
+
}
|
|
2566
|
+
/** Score for a produced RunRecord, absent when the campaign cell was unscored.
|
|
2567
|
+
* Ungated (`runTaskScore` is raw) — a matrix pivot reports measured scores;
|
|
2568
|
+
* it is not a training input. */
|
|
2569
|
+
function scoreOf(r) {
|
|
2570
|
+
return runTaskScore(r);
|
|
2571
|
+
}
|
|
2572
|
+
function meanOrNull(values) {
|
|
2573
|
+
return values.length === 0 ? null : mean(values);
|
|
2574
|
+
}
|
|
2575
|
+
function rollup(records, keyOf) {
|
|
2576
|
+
const groups = /* @__PURE__ */ new Map();
|
|
2577
|
+
for (const r of records) {
|
|
2578
|
+
const key = keyOf(r);
|
|
2579
|
+
if (key === void 0) continue;
|
|
2580
|
+
const score = scoreOf(r);
|
|
2581
|
+
if (score === void 0) continue;
|
|
2582
|
+
const arr = groups.get(key) ?? [];
|
|
2583
|
+
arr.push(score);
|
|
2584
|
+
groups.set(key, arr);
|
|
2585
|
+
}
|
|
2586
|
+
const out = {};
|
|
2587
|
+
for (const [key, xs] of groups) out[key] = {
|
|
2588
|
+
meanComposite: mean(xs),
|
|
2589
|
+
n: xs.length
|
|
2590
|
+
};
|
|
2591
|
+
return out;
|
|
2592
|
+
}
|
|
2593
|
+
function rollupByPersona(records, scenarios, personaOf) {
|
|
2594
|
+
const personaByScenarioId = /* @__PURE__ */ new Map();
|
|
2595
|
+
for (const s of scenarios) personaByScenarioId.set(s.id, personaOf(s));
|
|
2596
|
+
return rollup(records, (r) => r.scenarioId ? personaByScenarioId.get(r.scenarioId) : void 0);
|
|
2597
|
+
}
|
|
2598
|
+
//#endregion
|
|
2599
|
+
//#region src/campaign/scenario-selection.ts
|
|
2600
|
+
const DEFAULT_SATURATION_CEILING = .999;
|
|
2601
|
+
/** Variance below this is treated as "every candidate scored the same". */
|
|
2602
|
+
const VARIANCE_EPSILON = 1e-9;
|
|
2603
|
+
/** Population mean + variance of the candidate scores. Empty ⇒ zeros (a signal
|
|
2604
|
+
* with no observations cannot discriminate). */
|
|
2605
|
+
function moments(scores) {
|
|
2606
|
+
const n = scores.length;
|
|
2607
|
+
if (n === 0) return {
|
|
2608
|
+
meanScore: 0,
|
|
2609
|
+
variance: 0
|
|
2610
|
+
};
|
|
2611
|
+
let sum = 0;
|
|
2612
|
+
for (const s of scores) sum += s;
|
|
2613
|
+
const meanScore = sum / n;
|
|
2614
|
+
let sqDev = 0;
|
|
2615
|
+
for (const s of scores) {
|
|
2616
|
+
const d = s - meanScore;
|
|
2617
|
+
sqDev += d * d;
|
|
2618
|
+
}
|
|
2619
|
+
return {
|
|
2620
|
+
meanScore,
|
|
2621
|
+
variance: sqDev / n
|
|
2622
|
+
};
|
|
2623
|
+
}
|
|
2624
|
+
/** Deterministic ordering: discrimination desc, then meanScore asc (more
|
|
2625
|
+
* headroom first), then scenarioId asc. */
|
|
2626
|
+
function compareDiscrimination(a, b) {
|
|
2627
|
+
if (b.discrimination !== a.discrimination) return b.discrimination - a.discrimination;
|
|
2628
|
+
if (a.meanScore !== b.meanScore) return a.meanScore - b.meanScore;
|
|
2629
|
+
return a.scenarioId < b.scenarioId ? -1 : a.scenarioId > b.scenarioId ? 1 : 0;
|
|
2630
|
+
}
|
|
2631
|
+
/**
|
|
2632
|
+
* Rank scenarios by how well they DISCRIMINATE candidates.
|
|
2633
|
+
*
|
|
2634
|
+
* `discrimination = variance` (spread of the candidate scores) — kept simple on
|
|
2635
|
+
* purpose; the headroom term (`saturationCeiling - meanScore`) only breaks ties
|
|
2636
|
+
* so that, among equally spread scenarios, the one with more room to improve
|
|
2637
|
+
* ranks first. Returned sorted by the deterministic order above.
|
|
2638
|
+
*/
|
|
2639
|
+
function scoreDiscrimination(signals, opts) {
|
|
2640
|
+
const saturationCeiling = opts?.saturationCeiling ?? DEFAULT_SATURATION_CEILING;
|
|
2641
|
+
return signals.map((signal) => {
|
|
2642
|
+
const { meanScore, variance } = moments(signal.scores);
|
|
2643
|
+
const tied = variance < VARIANCE_EPSILON && meanScore >= saturationCeiling;
|
|
2644
|
+
return {
|
|
2645
|
+
scenarioId: signal.scenarioId,
|
|
2646
|
+
discrimination: variance,
|
|
2647
|
+
meanScore,
|
|
2648
|
+
variance,
|
|
2649
|
+
tied
|
|
2650
|
+
};
|
|
2651
|
+
}).sort(compareDiscrimination);
|
|
2652
|
+
}
|
|
2653
|
+
/**
|
|
2654
|
+
* Select the top-`k` most discriminative scenario ids for a holdout, EXCLUDING
|
|
2655
|
+
* fully saturated ties when enough non-tied scenarios exist (a tie in the
|
|
2656
|
+
* holdout wastes a paired cell).
|
|
2657
|
+
*
|
|
2658
|
+
* Prefers non-tied scenarios; if fewer than `k` non-tied exist, fills with the
|
|
2659
|
+
* least-saturated tied ones (tied scenarios are already ordered least-saturated
|
|
2660
|
+
* first by `meanScore` asc). Deterministic. Throws if `k < 1`. If
|
|
2661
|
+
* `signals.length <= k`, returns all ids in discrimination order.
|
|
2662
|
+
*/
|
|
2663
|
+
function selectDiscriminative(signals, k, opts) {
|
|
2664
|
+
if (k < 1) throw new Error(`selectDiscriminative: k must be >= 1 (got ${k})`);
|
|
2665
|
+
const ranked = scoreDiscrimination(signals, opts);
|
|
2666
|
+
if (ranked.length <= k) return ranked.map((s) => s.scenarioId);
|
|
2667
|
+
const nonTied = ranked.filter((s) => !s.tied);
|
|
2668
|
+
if (nonTied.length >= k) return nonTied.slice(0, k).map((s) => s.scenarioId);
|
|
2669
|
+
const fill = ranked.filter((s) => s.tied).slice(0, k - nonTied.length);
|
|
2670
|
+
return [...nonTied, ...fill].map((s) => s.scenarioId);
|
|
2671
|
+
}
|
|
2672
|
+
//#endregion
|
|
2673
|
+
//#region src/campaign/search-ledger.ts
|
|
2674
|
+
/**
|
|
2675
|
+
* Durable append-only audit log for improvement searches.
|
|
2676
|
+
*
|
|
2677
|
+
* Existing campaign artifacts keep their own rich records: `RunRecord` owns a
|
|
2678
|
+
* measured run and `CostLedger` owns per-call accounting. This ledger does not
|
|
2679
|
+
* copy those structures. It binds their immutable ids and receipts into one replayable event stream so a
|
|
2680
|
+
* search can answer, after a crash, exactly which candidates and task attempts
|
|
2681
|
+
* existed, which surfaces actually fired, what they cost, and why they were
|
|
2682
|
+
* selected or rejected.
|
|
2683
|
+
*
|
|
2684
|
+
* The file format is canonical JSONL with a SHA-256 hash chain. Every append is
|
|
2685
|
+
* serialized across processes, fsynced before acknowledgement, and idempotent
|
|
2686
|
+
* by `eventId`. A malformed, non-canonical, truncated, reordered, or conflicting
|
|
2687
|
+
* log fails loudly; the implementation never skips a bad row.
|
|
2688
|
+
*
|
|
2689
|
+
* The journal machinery itself (hash chain, locking, fsync, idempotent append)
|
|
2690
|
+
* is the generic `ledger-core` journal; this module supplies the campaign
|
|
2691
|
+
* codec: event schemas, canonical event ordering, and the search state machine.
|
|
2692
|
+
*/
|
|
2693
|
+
const SEARCH_LEDGER_SCHEMA = "tangle.search-ledger.v1";
|
|
2694
|
+
const NON_EMPTY = z.string().min(1).refine((value) => value.trim() === value, "must not contain surrounding whitespace");
|
|
2695
|
+
const HASH = z.string().regex(/^sha256:[a-f0-9]{64}$/);
|
|
2696
|
+
const LINEAGE_NODE_ID = z.string().regex(/^[a-f0-9]{16}$/);
|
|
2697
|
+
const IMMUTABLE_REVISION = z.string().regex(/^(?:[a-f0-9]{40}|[a-f0-9]{64}|sha256:[a-f0-9]{64}|sha512:[A-Za-z0-9+/=]+)$/);
|
|
2698
|
+
const ISO_TIMESTAMP = z.string().regex(/^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(?:\.\d{3})?Z$/).refine((value) => Number.isFinite(Date.parse(value)), "invalid timestamp");
|
|
2699
|
+
const NON_NEGATIVE_INT = z.number().int().nonnegative().safe();
|
|
2700
|
+
const FINITE_NUMBER = z.number().finite();
|
|
2701
|
+
const ArtifactRefSchema = z.object({
|
|
2702
|
+
role: NON_EMPTY,
|
|
2703
|
+
uri: NON_EMPTY,
|
|
2704
|
+
sha256: HASH,
|
|
2705
|
+
byteLength: NON_NEGATIVE_INT
|
|
2706
|
+
}).strict();
|
|
2707
|
+
const SourceRefSchema = z.object({
|
|
2708
|
+
uri: NON_EMPTY,
|
|
2709
|
+
revision: IMMUTABLE_REVISION
|
|
2710
|
+
}).strict();
|
|
2711
|
+
const FailureReasonSchema = z.object({
|
|
2712
|
+
code: NON_EMPTY,
|
|
2713
|
+
message: NON_EMPTY
|
|
2714
|
+
}).strict();
|
|
2715
|
+
const EventBaseShape = {
|
|
2716
|
+
eventId: NON_EMPTY,
|
|
2717
|
+
occurredAt: ISO_TIMESTAMP,
|
|
2718
|
+
artifacts: z.array(ArtifactRefSchema).min(1)
|
|
2719
|
+
};
|
|
2720
|
+
const OperationKindSchema = z.enum([
|
|
2721
|
+
"candidate-generation",
|
|
2722
|
+
"analysis",
|
|
2723
|
+
"selection",
|
|
2724
|
+
"judge",
|
|
2725
|
+
"other"
|
|
2726
|
+
]);
|
|
2727
|
+
const SearchPlannedSchema = z.object({
|
|
2728
|
+
...EventBaseShape,
|
|
2729
|
+
kind: z.literal("search-planned"),
|
|
2730
|
+
plan: z.object({
|
|
2731
|
+
candidateSlots: z.array(z.object({
|
|
2732
|
+
slotId: NON_EMPTY,
|
|
2733
|
+
generationOperationId: NON_EMPTY
|
|
2734
|
+
}).strict()).min(1),
|
|
2735
|
+
tasks: z.array(z.object({
|
|
2736
|
+
taskId: NON_EMPTY,
|
|
2737
|
+
source: SourceRefSchema,
|
|
2738
|
+
benchmark: SourceRefSchema,
|
|
2739
|
+
maxAttempts: z.number().int().positive().safe()
|
|
2740
|
+
}).strict()).min(1),
|
|
2741
|
+
operations: z.array(z.object({
|
|
2742
|
+
operationId: NON_EMPTY,
|
|
2743
|
+
kind: OperationKindSchema
|
|
2744
|
+
}).strict()).min(1)
|
|
2745
|
+
}).strict()
|
|
2746
|
+
}).strict();
|
|
2747
|
+
const CandidateRegisteredSchema = z.object({
|
|
2748
|
+
...EventBaseShape,
|
|
2749
|
+
kind: z.literal("candidate-registered"),
|
|
2750
|
+
slotId: NON_EMPTY,
|
|
2751
|
+
generationOperationId: NON_EMPTY,
|
|
2752
|
+
candidateId: NON_EMPTY,
|
|
2753
|
+
lineage: z.object({
|
|
2754
|
+
lineageNodeId: LINEAGE_NODE_ID,
|
|
2755
|
+
parentCandidateIds: z.array(NON_EMPTY),
|
|
2756
|
+
generation: NON_NEGATIVE_INT,
|
|
2757
|
+
proposer: NON_EMPTY,
|
|
2758
|
+
proposerSource: SourceRefSchema
|
|
2759
|
+
}).strict(),
|
|
2760
|
+
surfaces: z.array(z.object({
|
|
2761
|
+
surfaceId: NON_EMPTY,
|
|
2762
|
+
kind: z.enum([
|
|
2763
|
+
"prompt",
|
|
2764
|
+
"tool-contract",
|
|
2765
|
+
"runtime-config",
|
|
2766
|
+
"memory",
|
|
2767
|
+
"knowledge",
|
|
2768
|
+
"agent-profile",
|
|
2769
|
+
"code",
|
|
2770
|
+
"deployment"
|
|
2771
|
+
]),
|
|
2772
|
+
artifact: ArtifactRefSchema
|
|
2773
|
+
}).strict()).min(1)
|
|
2774
|
+
}).strict();
|
|
2775
|
+
const CandidateSlotClosedSchema = z.object({
|
|
2776
|
+
...EventBaseShape,
|
|
2777
|
+
kind: z.literal("candidate-slot-closed"),
|
|
2778
|
+
slotId: NON_EMPTY,
|
|
2779
|
+
generationOperationId: NON_EMPTY,
|
|
2780
|
+
reason: FailureReasonSchema
|
|
2781
|
+
}).strict();
|
|
2782
|
+
const KnownTokensSchema = z.object({
|
|
2783
|
+
status: z.literal("known"),
|
|
2784
|
+
inputTokens: NON_NEGATIVE_INT,
|
|
2785
|
+
outputTokens: NON_NEGATIVE_INT,
|
|
2786
|
+
cachedTokens: NON_NEGATIVE_INT
|
|
2787
|
+
}).strict();
|
|
2788
|
+
const UnknownSchema = z.object({
|
|
2789
|
+
status: z.literal("unknown"),
|
|
2790
|
+
reason: NON_EMPTY
|
|
2791
|
+
}).strict();
|
|
2792
|
+
const KnownCostSchema = z.object({
|
|
2793
|
+
status: z.literal("known"),
|
|
2794
|
+
usd: z.number().finite().nonnegative(),
|
|
2795
|
+
source: z.enum([
|
|
2796
|
+
"provider",
|
|
2797
|
+
"pricing-table",
|
|
2798
|
+
"free"
|
|
2799
|
+
])
|
|
2800
|
+
}).strict().superRefine((cost, ctx) => {
|
|
2801
|
+
if (cost.source === "free" && cost.usd !== 0) ctx.addIssue({
|
|
2802
|
+
code: "custom",
|
|
2803
|
+
message: "free cost source must have usd 0"
|
|
2804
|
+
});
|
|
2805
|
+
});
|
|
2806
|
+
const UnknownCostSchema = z.object({
|
|
2807
|
+
status: z.literal("unknown"),
|
|
2808
|
+
knownLowerBoundUsd: z.number().finite().nonnegative(),
|
|
2809
|
+
reason: NON_EMPTY
|
|
2810
|
+
}).strict();
|
|
2811
|
+
const AccountingSchema = z.object({
|
|
2812
|
+
tokens: z.discriminatedUnion("status", [KnownTokensSchema, UnknownSchema]),
|
|
2813
|
+
cost: z.discriminatedUnion("status", [KnownCostSchema, UnknownCostSchema])
|
|
2814
|
+
}).strict();
|
|
2815
|
+
const MetricsSchema = z.record(NON_EMPTY, FINITE_NUMBER).superRefine((metrics, ctx) => {
|
|
2816
|
+
for (const key of Object.keys(metrics)) if (key === "__proto__" || key === "constructor" || key === "prototype") ctx.addIssue({
|
|
2817
|
+
code: "custom",
|
|
2818
|
+
message: `unsafe metric key ${key}`
|
|
2819
|
+
});
|
|
2820
|
+
});
|
|
2821
|
+
const OutcomeSchema = z.discriminatedUnion("status", [
|
|
2822
|
+
z.object({
|
|
2823
|
+
status: z.literal("passed"),
|
|
2824
|
+
score: FINITE_NUMBER,
|
|
2825
|
+
metrics: MetricsSchema
|
|
2826
|
+
}).strict(),
|
|
2827
|
+
z.object({
|
|
2828
|
+
status: z.literal("failed"),
|
|
2829
|
+
score: FINITE_NUMBER,
|
|
2830
|
+
metrics: MetricsSchema,
|
|
2831
|
+
failure: FailureReasonSchema
|
|
2832
|
+
}).strict(),
|
|
2833
|
+
z.object({
|
|
2834
|
+
status: z.literal("errored"),
|
|
2835
|
+
metrics: MetricsSchema,
|
|
2836
|
+
error: FailureReasonSchema.extend({ retryable: z.boolean() }).strict()
|
|
2837
|
+
}).strict()
|
|
2838
|
+
]);
|
|
2839
|
+
const EffectSchema = z.discriminatedUnion("status", [z.object({
|
|
2840
|
+
status: z.literal("measured"),
|
|
2841
|
+
metric: NON_EMPTY,
|
|
2842
|
+
baselineValue: FINITE_NUMBER,
|
|
2843
|
+
candidateValue: FINITE_NUMBER,
|
|
2844
|
+
delta: FINITE_NUMBER
|
|
2845
|
+
}).strict().superRefine((effect, ctx) => {
|
|
2846
|
+
const expected = effect.candidateValue - effect.baselineValue;
|
|
2847
|
+
const tolerance = Number.EPSILON * Math.max(1, Math.abs(expected), Math.abs(effect.delta)) * 8;
|
|
2848
|
+
if (Math.abs(effect.delta - expected) > tolerance) ctx.addIssue({
|
|
2849
|
+
code: "custom",
|
|
2850
|
+
message: "delta must equal candidateValue - baselineValue"
|
|
2851
|
+
});
|
|
2852
|
+
}), z.object({
|
|
2853
|
+
status: z.literal("not-measured"),
|
|
2854
|
+
reason: NON_EMPTY
|
|
2855
|
+
}).strict()]);
|
|
2856
|
+
const SurfaceEvidenceSchema = z.object({
|
|
2857
|
+
surfaceId: NON_EMPTY,
|
|
2858
|
+
fired: z.boolean(),
|
|
2859
|
+
firingCount: NON_NEGATIVE_INT,
|
|
2860
|
+
effect: EffectSchema,
|
|
2861
|
+
evidence: z.array(ArtifactRefSchema).min(1)
|
|
2862
|
+
}).strict().superRefine((evidence, ctx) => {
|
|
2863
|
+
if (evidence.fired && evidence.firingCount === 0) ctx.addIssue({
|
|
2864
|
+
code: "custom",
|
|
2865
|
+
message: "a fired surface must have firingCount >= 1"
|
|
2866
|
+
});
|
|
2867
|
+
if (!evidence.fired && evidence.firingCount !== 0) ctx.addIssue({
|
|
2868
|
+
code: "custom",
|
|
2869
|
+
message: "a surface that did not fire must have firingCount 0"
|
|
2870
|
+
});
|
|
2871
|
+
if (!evidence.fired && evidence.effect.status === "measured" && evidence.effect.delta !== 0) ctx.addIssue({
|
|
2872
|
+
code: "custom",
|
|
2873
|
+
message: "a surface that did not fire cannot claim non-zero effect"
|
|
2874
|
+
});
|
|
2875
|
+
});
|
|
2876
|
+
const TaskAttemptedSchema = z.object({
|
|
2877
|
+
...EventBaseShape,
|
|
2878
|
+
kind: z.literal("task-attempted"),
|
|
2879
|
+
candidateId: NON_EMPTY,
|
|
2880
|
+
runId: NON_EMPTY,
|
|
2881
|
+
attemptIndex: NON_NEGATIVE_INT,
|
|
2882
|
+
task: z.object({
|
|
2883
|
+
taskId: NON_EMPTY,
|
|
2884
|
+
source: SourceRefSchema
|
|
2885
|
+
}).strict(),
|
|
2886
|
+
identity: z.object({
|
|
2887
|
+
model: z.object({
|
|
2888
|
+
provider: NON_EMPTY,
|
|
2889
|
+
snapshot: NON_EMPTY.refine(modelHasSnapshot, "model must include an immutable snapshot")
|
|
2890
|
+
}).strict(),
|
|
2891
|
+
agent: SourceRefSchema,
|
|
2892
|
+
benchmark: SourceRefSchema
|
|
2893
|
+
}).strict(),
|
|
2894
|
+
outcome: OutcomeSchema,
|
|
2895
|
+
accounting: AccountingSchema,
|
|
2896
|
+
surfaceEvidence: z.array(SurfaceEvidenceSchema).min(1)
|
|
2897
|
+
}).strict();
|
|
2898
|
+
const SearchOperationRecordedSchema = z.object({
|
|
2899
|
+
...EventBaseShape,
|
|
2900
|
+
kind: z.literal("search-operation-recorded"),
|
|
2901
|
+
operationId: NON_EMPTY,
|
|
2902
|
+
operationKind: OperationKindSchema,
|
|
2903
|
+
execution: z.discriminatedUnion("kind", [z.object({
|
|
2904
|
+
kind: z.literal("model"),
|
|
2905
|
+
model: z.object({
|
|
2906
|
+
provider: NON_EMPTY,
|
|
2907
|
+
snapshot: NON_EMPTY.refine(modelHasSnapshot, "model must include an immutable snapshot")
|
|
2908
|
+
}).strict(),
|
|
2909
|
+
source: SourceRefSchema
|
|
2910
|
+
}).strict(), z.object({
|
|
2911
|
+
kind: z.literal("deterministic"),
|
|
2912
|
+
source: SourceRefSchema
|
|
2913
|
+
}).strict()]),
|
|
2914
|
+
outcome: z.discriminatedUnion("status", [
|
|
2915
|
+
z.object({ status: z.literal("completed") }).strict(),
|
|
2916
|
+
z.object({
|
|
2917
|
+
status: z.literal("partial"),
|
|
2918
|
+
failure: FailureReasonSchema
|
|
2919
|
+
}).strict(),
|
|
2920
|
+
z.object({
|
|
2921
|
+
status: z.literal("failed"),
|
|
2922
|
+
failure: FailureReasonSchema
|
|
2923
|
+
}).strict()
|
|
2924
|
+
]),
|
|
2925
|
+
accounting: AccountingSchema
|
|
2926
|
+
}).strict();
|
|
2927
|
+
const CandidateDecidedSchema = z.object({
|
|
2928
|
+
...EventBaseShape,
|
|
2929
|
+
kind: z.literal("candidate-decided"),
|
|
2930
|
+
candidateId: NON_EMPTY,
|
|
2931
|
+
decision: z.discriminatedUnion("status", [z.object({ status: z.literal("selected") }).strict(), z.object({
|
|
2932
|
+
status: z.literal("rejected"),
|
|
2933
|
+
reason: FailureReasonSchema
|
|
2934
|
+
}).strict()])
|
|
2935
|
+
}).strict();
|
|
2936
|
+
const SearchCompletedSchema = z.object({
|
|
2937
|
+
...EventBaseShape,
|
|
2938
|
+
kind: z.literal("search-completed"),
|
|
2939
|
+
result: z.discriminatedUnion("status", [z.object({
|
|
2940
|
+
status: z.literal("selected"),
|
|
2941
|
+
candidateId: NON_EMPTY
|
|
2942
|
+
}).strict(), z.object({
|
|
2943
|
+
status: z.literal("all-rejected"),
|
|
2944
|
+
reason: FailureReasonSchema
|
|
2945
|
+
}).strict()])
|
|
2946
|
+
}).strict();
|
|
2947
|
+
const EventSchema = z.discriminatedUnion("kind", [
|
|
2948
|
+
SearchPlannedSchema,
|
|
2949
|
+
CandidateRegisteredSchema,
|
|
2950
|
+
CandidateSlotClosedSchema,
|
|
2951
|
+
TaskAttemptedSchema,
|
|
2952
|
+
SearchOperationRecordedSchema,
|
|
2953
|
+
CandidateDecidedSchema,
|
|
2954
|
+
SearchCompletedSchema
|
|
2955
|
+
]);
|
|
2956
|
+
const EntrySchema = z.object({
|
|
2957
|
+
schema: z.literal(SEARCH_LEDGER_SCHEMA),
|
|
2958
|
+
campaignId: NON_EMPTY,
|
|
2959
|
+
sequence: NON_NEGATIVE_INT,
|
|
2960
|
+
previousHash: z.union([HASH, z.null()]),
|
|
2961
|
+
event: EventSchema,
|
|
2962
|
+
entryHash: HASH
|
|
2963
|
+
}).strict();
|
|
2964
|
+
/** Validate and return a canonical copy. Arrays whose order is not semantic are
|
|
2965
|
+
* sorted so retries from different processes produce byte-identical events. */
|
|
2966
|
+
function validateSearchLedgerEvent(input) {
|
|
2967
|
+
const parsed = EventSchema.safeParse(input);
|
|
2968
|
+
if (!parsed.success) throw new SearchLedgerError(`invalid search ledger event: ${formatZodError(parsed.error)}`);
|
|
2969
|
+
return normalizeEvent(parsed.data);
|
|
2970
|
+
}
|
|
2971
|
+
/** Open a durable filesystem search ledger. Construction performs no I/O; the
|
|
2972
|
+
* first `append` or `replay` validates the complete existing file. */
|
|
2973
|
+
function openSearchLedger(options) {
|
|
2974
|
+
if (options.path.trim().length === 0) throw new SearchLedgerError("ledger path is empty");
|
|
2975
|
+
return new FileSearchLedger(options.path, options.campaignId);
|
|
2976
|
+
}
|
|
2977
|
+
function searchLedgerCodec(campaignId) {
|
|
2978
|
+
return {
|
|
2979
|
+
...SEARCH_LEDGER_FILE_CONTEXT,
|
|
2980
|
+
header: {
|
|
2981
|
+
schema: SEARCH_LEDGER_SCHEMA,
|
|
2982
|
+
campaignId
|
|
2983
|
+
},
|
|
2984
|
+
conflictError: (message) => new SearchLedgerConflictError(message),
|
|
2985
|
+
parseEntry: parseSearchLedgerEntry,
|
|
2986
|
+
checkEntryHeader: (entry, index) => {
|
|
2987
|
+
if (entry.campaignId !== campaignId) throw new SearchLedgerIntegrityError(`entry ${index} belongs to campaign ${entry.campaignId}, expected ${campaignId}`);
|
|
2988
|
+
},
|
|
2989
|
+
createProjector: () => createSearchLedgerProjector(campaignId)
|
|
2990
|
+
};
|
|
2991
|
+
}
|
|
2992
|
+
/** Append-only file-backed search ledger with idempotent writes and replay. */
|
|
2993
|
+
var FileSearchLedger = class {
|
|
2994
|
+
path;
|
|
2995
|
+
campaignId;
|
|
2996
|
+
journal;
|
|
2997
|
+
constructor(path, campaignId) {
|
|
2998
|
+
if (path.trim().length === 0) throw new SearchLedgerError("ledger path is empty");
|
|
2999
|
+
if (campaignId.length === 0) throw new SearchLedgerError("campaignId is empty");
|
|
3000
|
+
if (campaignId.trim() !== campaignId) throw new SearchLedgerError("campaignId must not contain surrounding whitespace");
|
|
3001
|
+
this.campaignId = campaignId;
|
|
3002
|
+
this.journal = new FileLedgerJournal(path, searchLedgerCodec(campaignId));
|
|
3003
|
+
this.path = this.journal.path;
|
|
3004
|
+
}
|
|
3005
|
+
async replay() {
|
|
3006
|
+
return (await this.journal.replay()).projection;
|
|
3007
|
+
}
|
|
3008
|
+
async append(input) {
|
|
3009
|
+
const event = validateSearchLedgerEvent(input);
|
|
3010
|
+
const { entry, appended, projection } = await this.journal.append(event);
|
|
3011
|
+
return {
|
|
3012
|
+
entry,
|
|
3013
|
+
appended,
|
|
3014
|
+
replay: projection
|
|
3015
|
+
};
|
|
3016
|
+
}
|
|
3017
|
+
};
|
|
3018
|
+
function parseSearchLedgerEntry(raw, context) {
|
|
3019
|
+
const parsed = EntrySchema.safeParse(raw);
|
|
3020
|
+
if (!parsed.success) throw new SearchLedgerIntegrityError(`search ledger ${context.path} has a malformed entry at line ${context.line}: ${formatZodError(parsed.error)}`);
|
|
3021
|
+
const entry = parsed.data;
|
|
3022
|
+
if (canonicalString(validateSearchLedgerEvent(entry.event)) !== canonicalString(entry.event)) throw new SearchLedgerIntegrityError(`search ledger ${context.path} has non-canonical event ordering at line ${context.line}`);
|
|
3023
|
+
return entry;
|
|
3024
|
+
}
|
|
3025
|
+
/** Replay the campaign search state machine over chain-verified entries. The
|
|
3026
|
+
* generic journal owns sequence, hash, and eventId-uniqueness checks; this
|
|
3027
|
+
* projector owns every campaign invariant and builds the replay projection. */
|
|
3028
|
+
function createSearchLedgerProjector(campaignId) {
|
|
3029
|
+
const candidates = /* @__PURE__ */ new Map();
|
|
3030
|
+
const candidateBySlot = /* @__PURE__ */ new Map();
|
|
3031
|
+
const closedSlots = /* @__PURE__ */ new Map();
|
|
3032
|
+
const lineageNodes = /* @__PURE__ */ new Map();
|
|
3033
|
+
const runIds = /* @__PURE__ */ new Set();
|
|
3034
|
+
const attemptKeys = /* @__PURE__ */ new Set();
|
|
3035
|
+
const candidateEvents = [];
|
|
3036
|
+
const closedSlotEvents = [];
|
|
3037
|
+
const attempts = [];
|
|
3038
|
+
const operationEvents = [];
|
|
3039
|
+
const operationsById = /* @__PURE__ */ new Map();
|
|
3040
|
+
const decisions = [];
|
|
3041
|
+
let planEvent = null;
|
|
3042
|
+
let completion = null;
|
|
3043
|
+
let previousOccurredAt = Number.NEGATIVE_INFINITY;
|
|
3044
|
+
const apply = (entry, index) => {
|
|
3045
|
+
const event = entry.event;
|
|
3046
|
+
if (completion) throw new SearchLedgerIntegrityError(`event ${event.eventId} appears after terminal event ${completion.eventId}`);
|
|
3047
|
+
const occurredAt = Date.parse(event.occurredAt);
|
|
3048
|
+
if (occurredAt < previousOccurredAt) throw new SearchLedgerIntegrityError(`event ${event.eventId} occurred before the preceding durable event`);
|
|
3049
|
+
previousOccurredAt = occurredAt;
|
|
3050
|
+
assertUnique(event.artifacts.map(artifactKey), "artifact receipt", event.eventId);
|
|
3051
|
+
if (event.kind === "search-planned") {
|
|
3052
|
+
if (index !== 0 || planEvent) throw new SearchLedgerIntegrityError("search plan must be the first and only plan event");
|
|
3053
|
+
assertUnique(event.plan.candidateSlots.map((slot) => slot.slotId), "candidate slot", event.eventId);
|
|
3054
|
+
assertUnique(event.plan.tasks.map((task) => task.taskId), "planned taskId", event.eventId);
|
|
3055
|
+
assertUnique(event.plan.operations.map((operation) => operation.operationId), "planned operationId", event.eventId);
|
|
3056
|
+
for (const slot of event.plan.candidateSlots) if (event.plan.operations.find((operation) => operation.operationId === slot.generationOperationId)?.kind !== "candidate-generation") throw new SearchLedgerIntegrityError(`candidate slot ${slot.slotId} references unplanned candidate-generation operation ${slot.generationOperationId}`);
|
|
3057
|
+
planEvent = event;
|
|
3058
|
+
return;
|
|
3059
|
+
}
|
|
3060
|
+
if (!planEvent) throw new SearchLedgerIntegrityError(`event ${event.eventId} appears before the required search plan`);
|
|
3061
|
+
if (event.kind === "candidate-registered") {
|
|
3062
|
+
if (candidates.has(event.candidateId)) throw new SearchLedgerIntegrityError(`candidate ${event.candidateId} was registered twice`);
|
|
3063
|
+
const plannedSlot = planEvent.plan.candidateSlots.find((slot) => slot.slotId === event.slotId);
|
|
3064
|
+
if (!plannedSlot) throw new SearchLedgerIntegrityError(`candidate ${event.candidateId} binds unknown slot ${event.slotId}`);
|
|
3065
|
+
if (candidateBySlot.has(event.slotId)) throw new SearchLedgerIntegrityError(`candidate slot ${event.slotId} was bound twice`);
|
|
3066
|
+
if (closedSlots.has(event.slotId)) throw new SearchLedgerIntegrityError(`candidate slot ${event.slotId} was already closed`);
|
|
3067
|
+
if (event.generationOperationId !== plannedSlot.generationOperationId) throw new SearchLedgerIntegrityError(`candidate ${event.candidateId} generation operation ${event.generationOperationId} does not match slot ${event.slotId} plan ${plannedSlot.generationOperationId}`);
|
|
3068
|
+
const generationOperation = operationsById.get(event.generationOperationId);
|
|
3069
|
+
if (!generationOperation) throw new SearchLedgerIntegrityError(`candidate ${event.candidateId} precedes generation operation ${event.generationOperationId}`);
|
|
3070
|
+
if (generationOperation.outcome.status === "failed") throw new SearchLedgerIntegrityError(`candidate ${event.candidateId} cannot bind failed generation operation ${event.generationOperationId}`);
|
|
3071
|
+
const previousCandidate = lineageNodes.get(event.lineage.lineageNodeId);
|
|
3072
|
+
if (previousCandidate) throw new SearchLedgerIntegrityError(`lineage node ${event.lineage.lineageNodeId} is already bound to ${previousCandidate}`);
|
|
3073
|
+
assertUnique(event.lineage.parentCandidateIds, "parentCandidateId", event.eventId);
|
|
3074
|
+
const parents = event.lineage.parentCandidateIds.map((id) => {
|
|
3075
|
+
const parent = candidates.get(id);
|
|
3076
|
+
if (!parent) throw new SearchLedgerIntegrityError(`candidate ${event.candidateId} references unknown parent ${id}`);
|
|
3077
|
+
return parent;
|
|
3078
|
+
});
|
|
3079
|
+
const expectedGeneration = parents.length === 0 ? 0 : Math.max(...parents.map((parent) => parent.registered.lineage.generation)) + 1;
|
|
3080
|
+
if (event.lineage.generation !== expectedGeneration) throw new SearchLedgerIntegrityError(`candidate ${event.candidateId} generation ${event.lineage.generation} does not follow its parents (expected ${expectedGeneration})`);
|
|
3081
|
+
assertUnique(event.surfaces.map((surface) => surface.surfaceId), "surfaceId", event.eventId);
|
|
3082
|
+
candidates.set(event.candidateId, {
|
|
3083
|
+
registered: event,
|
|
3084
|
+
attempts: [],
|
|
3085
|
+
decision: null
|
|
3086
|
+
});
|
|
3087
|
+
candidateBySlot.set(event.slotId, event.candidateId);
|
|
3088
|
+
lineageNodes.set(event.lineage.lineageNodeId, event.candidateId);
|
|
3089
|
+
candidateEvents.push(event);
|
|
3090
|
+
return;
|
|
3091
|
+
}
|
|
3092
|
+
if (event.kind === "task-attempted") {
|
|
3093
|
+
const candidate = candidates.get(event.candidateId);
|
|
3094
|
+
if (!candidate) throw new SearchLedgerIntegrityError(`attempt ${event.eventId} references unknown candidate ${event.candidateId}`);
|
|
3095
|
+
if (candidate.decision) throw new SearchLedgerIntegrityError(`attempt ${event.eventId} appears after candidate ${event.candidateId} was decided`);
|
|
3096
|
+
const plannedTask = planEvent.plan.tasks.find((task) => task.taskId === event.task.taskId);
|
|
3097
|
+
if (!plannedTask) throw new SearchLedgerIntegrityError(`attempt ${event.eventId} references unplanned task ${event.task.taskId}`);
|
|
3098
|
+
if (canonicalString(plannedTask.source) !== canonicalString(event.task.source) || canonicalString(plannedTask.benchmark) !== canonicalString(event.identity.benchmark)) throw new SearchLedgerIntegrityError(`task ${event.task.taskId} does not match its planned source identity`);
|
|
3099
|
+
if (event.attemptIndex >= plannedTask.maxAttempts) throw new SearchLedgerIntegrityError(`task ${event.task.taskId} attempt ${event.attemptIndex} exceeds planned maxAttempts ${plannedTask.maxAttempts}`);
|
|
3100
|
+
if (runIds.has(event.runId)) throw new SearchLedgerIntegrityError(`runId ${event.runId} was recorded twice`);
|
|
3101
|
+
runIds.add(event.runId);
|
|
3102
|
+
const attemptKey = canonicalString([
|
|
3103
|
+
event.candidateId,
|
|
3104
|
+
event.task.taskId,
|
|
3105
|
+
event.attemptIndex
|
|
3106
|
+
]);
|
|
3107
|
+
if (attemptKeys.has(attemptKey)) throw new SearchLedgerIntegrityError(`candidate ${event.candidateId} task ${event.task.taskId} attempt ${event.attemptIndex} was recorded twice`);
|
|
3108
|
+
const expectedAttemptIndex = candidate.attempts.filter((attempt) => attempt.task.taskId === event.task.taskId).length;
|
|
3109
|
+
if (event.attemptIndex !== expectedAttemptIndex) throw new SearchLedgerIntegrityError(`candidate ${event.candidateId} task ${event.task.taskId} attempt index ${event.attemptIndex} is not contiguous (expected ${expectedAttemptIndex})`);
|
|
3110
|
+
const previousAttempt = candidate.attempts.find((attempt) => attempt.task.taskId === event.task.taskId);
|
|
3111
|
+
if (previousAttempt?.outcome.status !== void 0 && previousAttempt.outcome.status !== "errored") throw new SearchLedgerIntegrityError(`task ${event.task.taskId} was retried after a measured outcome`);
|
|
3112
|
+
if (previousAttempt && canonicalString({
|
|
3113
|
+
task: previousAttempt.task,
|
|
3114
|
+
identity: previousAttempt.identity
|
|
3115
|
+
}) !== canonicalString({
|
|
3116
|
+
task: event.task,
|
|
3117
|
+
identity: event.identity
|
|
3118
|
+
})) throw new SearchLedgerIntegrityError(`candidate ${event.candidateId} task ${event.task.taskId} changed immutable execution identity between attempts`);
|
|
3119
|
+
attemptKeys.add(attemptKey);
|
|
3120
|
+
const declared = candidate.registered.surfaces.map((surface) => surface.surfaceId).sort();
|
|
3121
|
+
const observed = event.surfaceEvidence.map((surface) => surface.surfaceId).sort();
|
|
3122
|
+
assertUnique(observed, "surface evidence", event.eventId);
|
|
3123
|
+
for (const evidence of event.surfaceEvidence) assertUnique(evidence.evidence.map(artifactKey), "surface evidence receipt", event.eventId);
|
|
3124
|
+
if (canonicalString(declared) !== canonicalString(observed)) throw new SearchLedgerIntegrityError(`attempt ${event.eventId} surface evidence does not exactly cover candidate ${event.candidateId}`);
|
|
3125
|
+
candidate.attempts.push(event);
|
|
3126
|
+
attempts.push(event);
|
|
3127
|
+
return;
|
|
3128
|
+
}
|
|
3129
|
+
if (event.kind === "search-operation-recorded") {
|
|
3130
|
+
const plannedOperation = planEvent.plan.operations.find((operation) => operation.operationId === event.operationId);
|
|
3131
|
+
if (!plannedOperation) throw new SearchLedgerIntegrityError(`operation ${event.operationId} was not declared in the search plan`);
|
|
3132
|
+
if (plannedOperation.kind !== event.operationKind) throw new SearchLedgerIntegrityError(`operation ${event.operationId} kind ${event.operationKind} does not match planned ${plannedOperation.kind}`);
|
|
3133
|
+
if (operationsById.has(event.operationId)) throw new SearchLedgerIntegrityError(`operation ${event.operationId} was recorded twice`);
|
|
3134
|
+
operationsById.set(event.operationId, event);
|
|
3135
|
+
operationEvents.push(event);
|
|
3136
|
+
return;
|
|
3137
|
+
}
|
|
3138
|
+
if (event.kind === "candidate-slot-closed") {
|
|
3139
|
+
const plannedSlot = planEvent.plan.candidateSlots.find((slot) => slot.slotId === event.slotId);
|
|
3140
|
+
if (!plannedSlot) throw new SearchLedgerIntegrityError(`candidate slot closure ${event.eventId} references unknown slot ${event.slotId}`);
|
|
3141
|
+
if (event.generationOperationId !== plannedSlot.generationOperationId) throw new SearchLedgerIntegrityError(`candidate slot closure ${event.eventId} generation operation ${event.generationOperationId} does not match slot ${event.slotId} plan ${plannedSlot.generationOperationId}`);
|
|
3142
|
+
if (candidateBySlot.has(event.slotId)) throw new SearchLedgerIntegrityError(`candidate slot ${event.slotId} was already bound to a candidate`);
|
|
3143
|
+
if (closedSlots.has(event.slotId)) throw new SearchLedgerIntegrityError(`candidate slot ${event.slotId} was closed twice`);
|
|
3144
|
+
const operation = operationsById.get(event.generationOperationId);
|
|
3145
|
+
if (!operation) throw new SearchLedgerIntegrityError(`candidate slot closure ${event.eventId} precedes operation ${event.generationOperationId}`);
|
|
3146
|
+
if (operation.outcome.status === "completed") throw new SearchLedgerIntegrityError(`candidate slot ${event.slotId} cannot close from completed operation ${event.generationOperationId}`);
|
|
3147
|
+
closedSlots.set(event.slotId, event);
|
|
3148
|
+
closedSlotEvents.push(event);
|
|
3149
|
+
return;
|
|
3150
|
+
}
|
|
3151
|
+
if (event.kind === "candidate-decided") {
|
|
3152
|
+
const candidate = candidates.get(event.candidateId);
|
|
3153
|
+
if (!candidate) throw new SearchLedgerIntegrityError(`decision ${event.eventId} references unknown candidate ${event.candidateId}`);
|
|
3154
|
+
if (candidate.decision) throw new SearchLedgerIntegrityError(`candidate ${event.candidateId} was decided twice`);
|
|
3155
|
+
if (event.decision.status === "selected") {
|
|
3156
|
+
if (!candidate.attempts.some((attempt) => attempt.outcome.status !== "errored")) throw new SearchLedgerIntegrityError(`candidate ${event.candidateId} cannot be selected without a measured task outcome`);
|
|
3157
|
+
if (decisions.some((decision) => decision.decision.status === "selected")) throw new SearchLedgerIntegrityError("more than one candidate was selected");
|
|
3158
|
+
}
|
|
3159
|
+
candidate.decision = event;
|
|
3160
|
+
decisions.push(event);
|
|
3161
|
+
return;
|
|
3162
|
+
}
|
|
3163
|
+
const missingCandidateSlots = planEvent.plan.candidateSlots.filter((slot) => !candidateBySlot.has(slot.slotId) && !closedSlots.has(slot.slotId)).map((slot) => slot.slotId);
|
|
3164
|
+
if (missingCandidateSlots.length > 0) throw new SearchLedgerIntegrityError(`search completed with missing candidate slots: ${missingCandidateSlots.join(", ")}`);
|
|
3165
|
+
const missingTaskOutcomes = plannedTaskOutcomeKeys(planEvent, candidates);
|
|
3166
|
+
if (missingTaskOutcomes.length > 0) throw new SearchLedgerIntegrityError(`search completed with missing task outcomes: ${missingTaskOutcomes.join(", ")}`);
|
|
3167
|
+
const missingOperations = planEvent.plan.operations.filter((operation) => !operationsById.has(operation.operationId)).map((operation) => operation.operationId);
|
|
3168
|
+
if (missingOperations.length > 0) throw new SearchLedgerIntegrityError(`search completed with missing search operations: ${missingOperations.join(", ")}`);
|
|
3169
|
+
for (const operation of planEvent.plan.operations) {
|
|
3170
|
+
if (operation.kind !== "candidate-generation") continue;
|
|
3171
|
+
const generationOutcome = operationsById.get(operation.operationId).outcome.status;
|
|
3172
|
+
const slots = planEvent.plan.candidateSlots.filter((slot) => slot.generationOperationId === operation.operationId);
|
|
3173
|
+
if (slots.length === 0) continue;
|
|
3174
|
+
const registeredCount = slots.filter((slot) => candidateBySlot.has(slot.slotId)).length;
|
|
3175
|
+
const closedCount = slots.filter((slot) => closedSlots.has(slot.slotId)).length;
|
|
3176
|
+
if (generationOutcome === "completed" && closedCount > 0) throw new SearchLedgerIntegrityError(`completed generation operation ${operation.operationId} contains ${closedCount} closed slot(s)`);
|
|
3177
|
+
if (generationOutcome === "failed" && registeredCount > 0) throw new SearchLedgerIntegrityError(`failed generation operation ${operation.operationId} contains ${registeredCount} registered candidate(s)`);
|
|
3178
|
+
if (generationOutcome === "partial" && (registeredCount === 0 || closedCount === 0)) throw new SearchLedgerIntegrityError(`partial generation operation ${operation.operationId} must contain both a registered candidate and a closed slot`);
|
|
3179
|
+
}
|
|
3180
|
+
const pending = [...candidates.values()].filter((candidate) => candidate.decision === null);
|
|
3181
|
+
if (pending.length > 0) throw new SearchLedgerIntegrityError(`search completed with ${pending.length} candidate decision(s) missing`);
|
|
3182
|
+
const selected = decisions.filter((decision) => decision.decision.status === "selected");
|
|
3183
|
+
if (event.result.status === "selected") {
|
|
3184
|
+
if (selected.length !== 1 || selected[0].candidateId !== event.result.candidateId) throw new SearchLedgerIntegrityError(`search completion winner ${event.result.candidateId} does not match candidate decisions`);
|
|
3185
|
+
} else if (selected.length !== 0) throw new SearchLedgerIntegrityError("all-rejected completion contains a selected candidate");
|
|
3186
|
+
completion = event;
|
|
3187
|
+
};
|
|
3188
|
+
const finish = (entries) => {
|
|
3189
|
+
const selectedDecisions = decisions.filter((decision) => decision.decision.status === "selected");
|
|
3190
|
+
const rejectedDecisions = decisions.filter((decision) => decision.decision.status === "rejected");
|
|
3191
|
+
const outcomeCounts = {
|
|
3192
|
+
passed: 0,
|
|
3193
|
+
failed: 0,
|
|
3194
|
+
errored: 0
|
|
3195
|
+
};
|
|
3196
|
+
const operationOutcomeCounts = {
|
|
3197
|
+
completed: 0,
|
|
3198
|
+
partial: 0,
|
|
3199
|
+
failed: 0
|
|
3200
|
+
};
|
|
3201
|
+
let inputTokens = 0;
|
|
3202
|
+
let outputTokens = 0;
|
|
3203
|
+
let cachedTokens = 0;
|
|
3204
|
+
let costUsd = 0;
|
|
3205
|
+
const unknownTokenEventIds = [];
|
|
3206
|
+
const unknownCostEventIds = [];
|
|
3207
|
+
for (const attempt of attempts) outcomeCounts[attempt.outcome.status] += 1;
|
|
3208
|
+
for (const operation of operationEvents) operationOutcomeCounts[operation.outcome.status] += 1;
|
|
3209
|
+
for (const costedEvent of [...attempts, ...operationEvents]) {
|
|
3210
|
+
if (costedEvent.accounting.tokens.status === "known") {
|
|
3211
|
+
inputTokens += costedEvent.accounting.tokens.inputTokens;
|
|
3212
|
+
outputTokens += costedEvent.accounting.tokens.outputTokens;
|
|
3213
|
+
cachedTokens += costedEvent.accounting.tokens.cachedTokens;
|
|
3214
|
+
} else unknownTokenEventIds.push(costedEvent.eventId);
|
|
3215
|
+
if (costedEvent.accounting.cost.status === "known") costUsd += costedEvent.accounting.cost.usd;
|
|
3216
|
+
else {
|
|
3217
|
+
costUsd += costedEvent.accounting.cost.knownLowerBoundUsd;
|
|
3218
|
+
unknownCostEventIds.push(costedEvent.eventId);
|
|
3219
|
+
}
|
|
3220
|
+
}
|
|
3221
|
+
const accounting = unknownTokenEventIds.length === 0 && unknownCostEventIds.length === 0 ? {
|
|
3222
|
+
status: "known",
|
|
3223
|
+
inputTokens,
|
|
3224
|
+
outputTokens,
|
|
3225
|
+
cachedTokens,
|
|
3226
|
+
costUsd
|
|
3227
|
+
} : {
|
|
3228
|
+
status: "partial",
|
|
3229
|
+
knownInputTokens: inputTokens,
|
|
3230
|
+
knownOutputTokens: outputTokens,
|
|
3231
|
+
knownCachedTokens: cachedTokens,
|
|
3232
|
+
knownCostUsd: costUsd,
|
|
3233
|
+
unknownTokenEventIds,
|
|
3234
|
+
unknownCostEventIds
|
|
3235
|
+
};
|
|
3236
|
+
const selectedCandidateId = completion?.result.status === "selected" ? completion.result.candidateId : null;
|
|
3237
|
+
const status = completion?.result.status === "selected" ? "selected" : completion?.result.status === "all-rejected" ? "all-rejected" : "in-progress";
|
|
3238
|
+
const missingCandidateSlots = planEvent?.plan.candidateSlots.filter((slot) => !candidateBySlot.has(slot.slotId) && !closedSlots.has(slot.slotId)).map((slot) => slot.slotId) ?? [];
|
|
3239
|
+
const missingTaskOutcomes = planEvent ? plannedTaskOutcomeKeys(planEvent, candidates) : [];
|
|
3240
|
+
const missingOperations = planEvent?.plan.operations.filter((operation) => !operationsById.has(operation.operationId)).map((operation) => operation.operationId) ?? [];
|
|
3241
|
+
return {
|
|
3242
|
+
entries: [...entries],
|
|
3243
|
+
plan: planEvent,
|
|
3244
|
+
candidates: candidateEvents,
|
|
3245
|
+
closedCandidateSlots: closedSlotEvents,
|
|
3246
|
+
attempts,
|
|
3247
|
+
operations: operationEvents,
|
|
3248
|
+
decisions,
|
|
3249
|
+
completion,
|
|
3250
|
+
audit: {
|
|
3251
|
+
campaignId,
|
|
3252
|
+
eventCount: entries.length,
|
|
3253
|
+
candidateCount: candidates.size,
|
|
3254
|
+
closedCandidateSlotCount: closedSlots.size,
|
|
3255
|
+
attemptCount: attempts.length,
|
|
3256
|
+
operationCount: operationEvents.length,
|
|
3257
|
+
outcomes: outcomeCounts,
|
|
3258
|
+
operationOutcomes: operationOutcomeCounts,
|
|
3259
|
+
decisions: {
|
|
3260
|
+
selected: selectedDecisions.length,
|
|
3261
|
+
rejected: rejectedDecisions.length,
|
|
3262
|
+
pending: candidates.size - decisions.length
|
|
3263
|
+
},
|
|
3264
|
+
expected: {
|
|
3265
|
+
candidateSlots: planEvent?.plan.candidateSlots.length ?? 0,
|
|
3266
|
+
taskOutcomes: candidates.size * (planEvent?.plan.tasks.length ?? 0),
|
|
3267
|
+
operations: planEvent?.plan.operations.length ?? 0,
|
|
3268
|
+
missingCandidateSlots,
|
|
3269
|
+
missingTaskOutcomes,
|
|
3270
|
+
missingOperations
|
|
3271
|
+
},
|
|
3272
|
+
status,
|
|
3273
|
+
selectedCandidateId,
|
|
3274
|
+
accounting,
|
|
3275
|
+
headHash: entries.at(-1)?.entryHash ?? null
|
|
3276
|
+
}
|
|
3277
|
+
};
|
|
3278
|
+
};
|
|
3279
|
+
return {
|
|
3280
|
+
apply,
|
|
3281
|
+
finish
|
|
3282
|
+
};
|
|
3283
|
+
}
|
|
3284
|
+
function plannedTaskOutcomeKeys(planEvent, candidates) {
|
|
3285
|
+
const missing = [];
|
|
3286
|
+
const registeredCandidates = [...candidates.values()].sort((a, b) => compareStrings(a.registered.slotId, b.registered.slotId));
|
|
3287
|
+
for (const candidate of registeredCandidates) {
|
|
3288
|
+
const slotId = candidate.registered.slotId;
|
|
3289
|
+
for (const task of planEvent.plan.tasks) if (!candidate.attempts.some((attempt) => attempt.task.taskId === task.taskId && attempt.outcome.status !== "errored")) missing.push(`${slotId}/${task.taskId}`);
|
|
3290
|
+
}
|
|
3291
|
+
return missing;
|
|
3292
|
+
}
|
|
3293
|
+
function normalizeEvent(event) {
|
|
3294
|
+
const artifacts = sortArtifacts(event.artifacts);
|
|
3295
|
+
if (event.kind === "search-planned") return {
|
|
3296
|
+
...event,
|
|
3297
|
+
artifacts,
|
|
3298
|
+
plan: {
|
|
3299
|
+
candidateSlots: [...event.plan.candidateSlots].sort((a, b) => compareStrings(a.slotId, b.slotId)),
|
|
3300
|
+
tasks: [...event.plan.tasks].sort((a, b) => compareStrings(a.taskId, b.taskId)),
|
|
3301
|
+
operations: [...event.plan.operations].sort((a, b) => compareStrings(a.operationId, b.operationId))
|
|
3302
|
+
}
|
|
3303
|
+
};
|
|
3304
|
+
if (event.kind === "candidate-registered") return {
|
|
3305
|
+
...event,
|
|
3306
|
+
artifacts,
|
|
3307
|
+
lineage: {
|
|
3308
|
+
...event.lineage,
|
|
3309
|
+
parentCandidateIds: sortedStrings(event.lineage.parentCandidateIds)
|
|
3310
|
+
},
|
|
3311
|
+
surfaces: [...event.surfaces].map((surface) => ({
|
|
3312
|
+
...surface,
|
|
3313
|
+
artifact: { ...surface.artifact }
|
|
3314
|
+
})).sort((a, b) => compareStrings(a.surfaceId, b.surfaceId))
|
|
3315
|
+
};
|
|
3316
|
+
if (event.kind === "task-attempted") return {
|
|
3317
|
+
...event,
|
|
3318
|
+
artifacts,
|
|
3319
|
+
surfaceEvidence: [...event.surfaceEvidence].map((evidence) => ({
|
|
3320
|
+
...evidence,
|
|
3321
|
+
evidence: sortArtifacts(evidence.evidence)
|
|
3322
|
+
})).sort((a, b) => compareStrings(a.surfaceId, b.surfaceId))
|
|
3323
|
+
};
|
|
3324
|
+
return {
|
|
3325
|
+
...event,
|
|
3326
|
+
artifacts
|
|
3327
|
+
};
|
|
3328
|
+
}
|
|
3329
|
+
function sortArtifacts(artifacts) {
|
|
3330
|
+
return [...artifacts].map((artifact) => ({ ...artifact })).sort((a, b) => compareStrings(artifactKey(a), artifactKey(b)));
|
|
3331
|
+
}
|
|
3332
|
+
function artifactKey(artifact) {
|
|
3333
|
+
return canonicalString(artifact);
|
|
3334
|
+
}
|
|
3335
|
+
function compareStrings(a, b) {
|
|
3336
|
+
return a < b ? -1 : a > b ? 1 : 0;
|
|
3337
|
+
}
|
|
3338
|
+
function sortedStrings(values) {
|
|
3339
|
+
return [...values].sort();
|
|
3340
|
+
}
|
|
3341
|
+
function assertUnique(values, label, eventId) {
|
|
3342
|
+
if (new Set(values).size !== values.length) throw new SearchLedgerIntegrityError(`event ${eventId} contains duplicate ${label} values`);
|
|
3343
|
+
}
|
|
3344
|
+
function formatZodError(error) {
|
|
3345
|
+
return error.issues.map((issue) => `${issue.path.length > 0 ? issue.path.join(".") : "<root>"}: ${issue.message}`).join("; ");
|
|
3346
|
+
}
|
|
3347
|
+
//#endregion
|
|
3348
|
+
//#region src/campaign/transient-failure.ts
|
|
3349
|
+
const BASE_TRANSIENT = /\b50[234]\b|no stream output|produced no stream|admission timed out|admission_rejected|queue_timeout|fetch failed|ECONNRESET|This operation was aborted/i;
|
|
3350
|
+
const TIMEOUT_PATTERN = /timeout after \d+ ?ms|cli-bridge timeout/i;
|
|
3351
|
+
/**
|
|
3352
|
+
* True when the error text describes an infrastructure hiccup that should be
|
|
3353
|
+
* retried rather than scored. Empty/undefined input is not transient.
|
|
3354
|
+
*/
|
|
3355
|
+
function isTransientTransportFailure(message, opts = {}) {
|
|
3356
|
+
if (!message) return false;
|
|
3357
|
+
if (BASE_TRANSIENT.test(message)) return true;
|
|
3358
|
+
if ((opts.retryFullDurationTimeouts ?? false) && TIMEOUT_PATTERN.test(message)) return true;
|
|
3359
|
+
for (const p of opts.extraPatterns ?? []) if (p.test(message)) return true;
|
|
3360
|
+
return false;
|
|
3361
|
+
}
|
|
3362
|
+
//#endregion
|
|
3363
|
+
//#region src/campaign/worktree/index.ts
|
|
3364
|
+
/**
|
|
3365
|
+
* VCS-pluggable worktree adapter. One improvement = one worktree, PR-like
|
|
3366
|
+
* (multiple commits allowed). A code-tier proposer's `propose()` creates a
|
|
3367
|
+
* worktree, an agent commits the change into it, and `finalize()` returns a
|
|
3368
|
+
* content-addressed `CodeSurface` the measurement verifies before running.
|
|
3369
|
+
* On promotion the worktree becomes the PR branch.
|
|
3370
|
+
*
|
|
3371
|
+
* The interface is VCS-agnostic so a future `jj` ([jj-vcs](https://github.com/jj-vcs/jj))
|
|
3372
|
+
* adapter can slot in without touching proposer code. Only the git adapter
|
|
3373
|
+
* ships today. See `docs/design/loop-taxonomy.md`.
|
|
3374
|
+
*/
|
|
3375
|
+
const MAX_GIT_OUTPUT_BYTES = 256 * 1024 * 1024;
|
|
3376
|
+
const FILE_HASH_CHUNK_BYTES = 1024 * 1024;
|
|
3377
|
+
const GIT_REPOSITORY_ENV = /* @__PURE__ */ new Set([
|
|
3378
|
+
"GIT_ALTERNATE_OBJECT_DIRECTORIES",
|
|
3379
|
+
"GIT_COMMON_DIR",
|
|
3380
|
+
"GIT_DIR",
|
|
3381
|
+
"GIT_INDEX_FILE",
|
|
3382
|
+
"GIT_NAMESPACE",
|
|
3383
|
+
"GIT_OBJECT_DIRECTORY",
|
|
3384
|
+
"GIT_PREFIX",
|
|
3385
|
+
"GIT_QUARANTINE_PATH",
|
|
3386
|
+
"GIT_WORK_TREE"
|
|
3387
|
+
]);
|
|
3388
|
+
/** Typed failure from a `WorktreeAdapter` operation (create/finalize/discard) — wraps the underlying git error as `cause`. */
|
|
3389
|
+
var WorktreeAdapterError = class extends Error {
|
|
3390
|
+
cause;
|
|
3391
|
+
constructor(message, cause) {
|
|
3392
|
+
super(message);
|
|
3393
|
+
this.cause = cause;
|
|
3394
|
+
this.name = "WorktreeAdapterError";
|
|
3395
|
+
}
|
|
3396
|
+
};
|
|
3397
|
+
function defaultGit(args, cwd, overrides) {
|
|
3398
|
+
try {
|
|
3399
|
+
const env = { ...process.env };
|
|
3400
|
+
for (const key of Object.keys(env)) if (GIT_REPOSITORY_ENV.has(key) || key === "GIT_CONFIG" || key.startsWith("GIT_CONFIG_") || key.startsWith("GIT_ATTR_") || key === "GIT_DIFF_OPTS" || key === "GIT_EXTERNAL_DIFF") delete env[key];
|
|
3401
|
+
Object.assign(env, overrides);
|
|
3402
|
+
env.GIT_NO_REPLACE_OBJECTS = "1";
|
|
3403
|
+
env.LC_ALL = "C";
|
|
3404
|
+
env.LANG = "C";
|
|
3405
|
+
return execFileSync("git", args, {
|
|
3406
|
+
cwd,
|
|
3407
|
+
env,
|
|
3408
|
+
maxBuffer: MAX_GIT_OUTPUT_BYTES
|
|
3409
|
+
});
|
|
3410
|
+
} catch (err) {
|
|
3411
|
+
const stderr = err && typeof err === "object" && "stderr" in err ? String(err.stderr) : "";
|
|
3412
|
+
throw new WorktreeAdapterError(`git ${args.join(" ")} failed: ${stderr || String(err)}`, err);
|
|
3413
|
+
}
|
|
3414
|
+
}
|
|
3415
|
+
function gitBytes(git, args, cwd, env) {
|
|
3416
|
+
try {
|
|
3417
|
+
const output = git(args, cwd, env);
|
|
3418
|
+
return typeof output === "string" ? Buffer.from(output, "utf8") : Buffer.from(output);
|
|
3419
|
+
} catch (err) {
|
|
3420
|
+
if (err instanceof WorktreeAdapterError) throw err;
|
|
3421
|
+
throw new WorktreeAdapterError(`git ${args.join(" ")} failed: ${String(err)}`, err);
|
|
3422
|
+
}
|
|
3423
|
+
}
|
|
3424
|
+
function gitText(git, args, cwd, env) {
|
|
3425
|
+
return gitBytes(git, args, cwd, env).toString("utf8").trim();
|
|
3426
|
+
}
|
|
3427
|
+
function hasRegisteredWorktree(git, repoRoot, path) {
|
|
3428
|
+
const expected = Buffer.from(`worktree ${resolve(path)}`, "utf8");
|
|
3429
|
+
const records = gitBytes(git, [
|
|
3430
|
+
"worktree",
|
|
3431
|
+
"list",
|
|
3432
|
+
"--porcelain",
|
|
3433
|
+
"-z"
|
|
3434
|
+
], repoRoot);
|
|
3435
|
+
let start = 0;
|
|
3436
|
+
while (start < records.length) {
|
|
3437
|
+
const end = records.indexOf(0, start);
|
|
3438
|
+
if (end < 0) throw new WorktreeAdapterError("Git worktree list output was not NUL-terminated");
|
|
3439
|
+
if (records.subarray(start, end).equals(expected)) return true;
|
|
3440
|
+
start = end + 1;
|
|
3441
|
+
}
|
|
3442
|
+
return false;
|
|
3443
|
+
}
|
|
3444
|
+
function hasLocalBranch(git, repoRoot, branch) {
|
|
3445
|
+
const ref = `refs/heads/${branch}`;
|
|
3446
|
+
return gitText(git, [
|
|
3447
|
+
"for-each-ref",
|
|
3448
|
+
"--format=%(refname)",
|
|
3449
|
+
"--",
|
|
3450
|
+
ref
|
|
3451
|
+
], repoRoot).split("\n").some((candidate) => candidate === ref);
|
|
3452
|
+
}
|
|
3453
|
+
function reconcileAbsent(exists, remove) {
|
|
3454
|
+
try {
|
|
3455
|
+
if (!exists()) return void 0;
|
|
3456
|
+
} catch (err) {
|
|
3457
|
+
return err;
|
|
3458
|
+
}
|
|
3459
|
+
try {
|
|
3460
|
+
remove();
|
|
3461
|
+
return;
|
|
3462
|
+
} catch (removeError) {
|
|
3463
|
+
try {
|
|
3464
|
+
if (!exists()) return void 0;
|
|
3465
|
+
} catch (recheckError) {
|
|
3466
|
+
return new AggregateError([removeError, recheckError], "Removal failed and the resulting resource state could not be checked");
|
|
3467
|
+
}
|
|
3468
|
+
return removeError;
|
|
3469
|
+
}
|
|
3470
|
+
}
|
|
3471
|
+
function sha256(bytes) {
|
|
3472
|
+
return `sha256:${createHash("sha256").update(bytes).digest("hex")}`;
|
|
3473
|
+
}
|
|
3474
|
+
const PATCH_ARGS = [
|
|
3475
|
+
"-c",
|
|
3476
|
+
"core.compression=0",
|
|
3477
|
+
"-c",
|
|
3478
|
+
`core.attributesFile=${devNull}`,
|
|
3479
|
+
"-c",
|
|
3480
|
+
"core.quotePath=true",
|
|
3481
|
+
"-c",
|
|
3482
|
+
"diff.suppressBlankEmpty=false",
|
|
3483
|
+
"diff",
|
|
3484
|
+
"--unified=3",
|
|
3485
|
+
"--inter-hunk-context=0",
|
|
3486
|
+
`-O${devNull}`,
|
|
3487
|
+
"--binary",
|
|
3488
|
+
"--full-index",
|
|
3489
|
+
"--no-color",
|
|
3490
|
+
"--no-ext-diff",
|
|
3491
|
+
"--no-textconv",
|
|
3492
|
+
"--no-renames",
|
|
3493
|
+
"--diff-algorithm=myers",
|
|
3494
|
+
"--no-indent-heuristic",
|
|
3495
|
+
"--src-prefix=a/",
|
|
3496
|
+
"--dst-prefix=b/"
|
|
3497
|
+
];
|
|
3498
|
+
const CANONICAL_GIT_ENV = {
|
|
3499
|
+
GIT_ATTR_GLOBAL: devNull,
|
|
3500
|
+
GIT_ATTR_NOSYSTEM: "1",
|
|
3501
|
+
GIT_CONFIG_GLOBAL: devNull,
|
|
3502
|
+
GIT_CONFIG_NOSYSTEM: "1",
|
|
3503
|
+
GIT_CONFIG_SYSTEM: devNull
|
|
3504
|
+
};
|
|
3505
|
+
/** Render the transport patch from immutable objects through fresh Git metadata.
|
|
3506
|
+
* Source-repository config, info attributes, templates, and caller environment
|
|
3507
|
+
* therefore cannot alter bytes for the same two trees. */
|
|
3508
|
+
function patchBytes(git, cwd, baseCommit, candidateCommit) {
|
|
3509
|
+
const scratch = mkdtempSync(join(tmpdir(), "agent-eval-patch-"));
|
|
3510
|
+
const bareRepo = join(scratch, "repo.git");
|
|
3511
|
+
const emptyTemplate = join(scratch, "empty-template");
|
|
3512
|
+
mkdirSync(emptyTemplate);
|
|
3513
|
+
try {
|
|
3514
|
+
const objectFormat = gitObjectHashAlgorithm(candidateCommit);
|
|
3515
|
+
const sourceObjects = realpathSync(gitText(git, [
|
|
3516
|
+
"rev-parse",
|
|
3517
|
+
"--git-path",
|
|
3518
|
+
"objects"
|
|
3519
|
+
], cwd));
|
|
3520
|
+
gitText(git, [
|
|
3521
|
+
"init",
|
|
3522
|
+
"--bare",
|
|
3523
|
+
"--quiet",
|
|
3524
|
+
...objectFormat === "sha256" ? ["--object-format=sha256"] : [],
|
|
3525
|
+
`--template=${emptyTemplate}`,
|
|
3526
|
+
bareRepo
|
|
3527
|
+
], scratch, CANONICAL_GIT_ENV);
|
|
3528
|
+
return gitBytes(git, [
|
|
3529
|
+
`--git-dir=${bareRepo}`,
|
|
3530
|
+
...PATCH_ARGS,
|
|
3531
|
+
baseCommit,
|
|
3532
|
+
candidateCommit,
|
|
3533
|
+
"--"
|
|
3534
|
+
], scratch, {
|
|
3535
|
+
...CANONICAL_GIT_ENV,
|
|
3536
|
+
GIT_ALTERNATE_OBJECT_DIRECTORIES: sourceObjects
|
|
3537
|
+
});
|
|
3538
|
+
} finally {
|
|
3539
|
+
rmSync(scratch, {
|
|
3540
|
+
recursive: true,
|
|
3541
|
+
force: true
|
|
3542
|
+
});
|
|
3543
|
+
}
|
|
3544
|
+
}
|
|
3545
|
+
function resolveCommit(git, cwd, ref) {
|
|
3546
|
+
return gitText(git, [
|
|
3547
|
+
"rev-parse",
|
|
3548
|
+
"--verify",
|
|
3549
|
+
`${ref}^{commit}`
|
|
3550
|
+
], cwd);
|
|
3551
|
+
}
|
|
3552
|
+
function unresolvedWorktreePath(surface, worktreeDir) {
|
|
3553
|
+
if (isAbsolute(surface.worktreeRef)) return surface.worktreeRef;
|
|
3554
|
+
if (worktreeDir) return join(worktreeDir, basename(surface.worktreeRef));
|
|
3555
|
+
return surface.worktreeRef;
|
|
3556
|
+
}
|
|
3557
|
+
function displayGitPath(path) {
|
|
3558
|
+
return JSON.stringify(path);
|
|
3559
|
+
}
|
|
3560
|
+
function parseGitTreeEntries(bytes) {
|
|
3561
|
+
const input = Buffer.from(bytes);
|
|
3562
|
+
const entries = [];
|
|
3563
|
+
let start = 0;
|
|
3564
|
+
while (start < input.length) {
|
|
3565
|
+
const end = input.indexOf(0, start);
|
|
3566
|
+
if (end < 0) throw new WorktreeAdapterError("Git tree output was not NUL-terminated");
|
|
3567
|
+
const record = input.subarray(start, end);
|
|
3568
|
+
start = end + 1;
|
|
3569
|
+
if (record.length === 0) continue;
|
|
3570
|
+
const tab = record.indexOf(9);
|
|
3571
|
+
if (tab < 0) throw new WorktreeAdapterError("Git tree entry did not contain a path");
|
|
3572
|
+
const metadata = record.subarray(0, tab).toString("ascii").split(" ");
|
|
3573
|
+
if (metadata.length !== 3) throw new WorktreeAdapterError("Git tree entry was malformed");
|
|
3574
|
+
const mode = metadata[0];
|
|
3575
|
+
const type = metadata[1];
|
|
3576
|
+
const objectId = metadata[2];
|
|
3577
|
+
if (!mode || !type || !objectId) throw new WorktreeAdapterError("Git tree entry was malformed");
|
|
3578
|
+
const pathBytes = record.subarray(tab + 1);
|
|
3579
|
+
const path = pathBytes.toString("utf8");
|
|
3580
|
+
if (!Buffer.from(path, "utf8").equals(pathBytes)) throw new WorktreeAdapterError("CodeSurface paths must be valid UTF-8");
|
|
3581
|
+
if (type !== "blob" && type !== "commit") throw new WorktreeAdapterError(`CodeSurface contains unsupported Git object type ${type} at ${displayGitPath(path)}`);
|
|
3582
|
+
entries.push({
|
|
3583
|
+
mode,
|
|
3584
|
+
objectId,
|
|
3585
|
+
path
|
|
3586
|
+
});
|
|
3587
|
+
}
|
|
3588
|
+
return entries;
|
|
3589
|
+
}
|
|
3590
|
+
function parseGitPathList(bytes) {
|
|
3591
|
+
const input = Buffer.from(bytes);
|
|
3592
|
+
const paths = [];
|
|
3593
|
+
let start = 0;
|
|
3594
|
+
while (start < input.length) {
|
|
3595
|
+
const end = input.indexOf(0, start);
|
|
3596
|
+
if (end < 0) throw new WorktreeAdapterError("Git path output was not NUL-terminated");
|
|
3597
|
+
const pathBytes = input.subarray(start, end);
|
|
3598
|
+
start = end + 1;
|
|
3599
|
+
if (pathBytes.length === 0) continue;
|
|
3600
|
+
const path = pathBytes.toString("utf8");
|
|
3601
|
+
if (!Buffer.from(path, "utf8").equals(pathBytes)) throw new WorktreeAdapterError("CodeSurface paths must be valid UTF-8");
|
|
3602
|
+
paths.push(path);
|
|
3603
|
+
}
|
|
3604
|
+
return paths;
|
|
3605
|
+
}
|
|
3606
|
+
function isWithinRoot(root, candidate) {
|
|
3607
|
+
const rel = relative(root, candidate);
|
|
3608
|
+
return rel === "" || !isAbsolute(rel) && rel !== ".." && !rel.startsWith(`..${sep}`);
|
|
3609
|
+
}
|
|
3610
|
+
function assertSafeRelativePath(root, path) {
|
|
3611
|
+
if (path.length === 0 || isAbsolute(path)) throw new WorktreeAdapterError(`CodeSurface contains unsafe path ${displayGitPath(path)}`);
|
|
3612
|
+
const absolutePath = resolve(root, path);
|
|
3613
|
+
if (!isWithinRoot(root, absolutePath) || absolutePath === root) throw new WorktreeAdapterError(`CodeSurface path escapes its worktree: ${displayGitPath(path)}`);
|
|
3614
|
+
const segments = path.split(/[\\/]/u);
|
|
3615
|
+
let parent = root;
|
|
3616
|
+
for (const segment of segments.slice(0, -1)) {
|
|
3617
|
+
if (segment.length === 0 || segment === "." || segment === "..") throw new WorktreeAdapterError(`CodeSurface contains unsafe path ${displayGitPath(path)}`);
|
|
3618
|
+
parent = join(parent, segment);
|
|
3619
|
+
const stat = lstatSync(parent);
|
|
3620
|
+
if (!stat.isDirectory() || stat.isSymbolicLink()) throw new WorktreeAdapterError(`CodeSurface path traverses a non-directory or symbolic link: ${displayGitPath(path)}`);
|
|
3621
|
+
}
|
|
3622
|
+
return absolutePath;
|
|
3623
|
+
}
|
|
3624
|
+
function gitObjectHashAlgorithm(objectId) {
|
|
3625
|
+
if (/^[a-f0-9]{40}$/.test(objectId)) return "sha1";
|
|
3626
|
+
if (/^[a-f0-9]{64}$/.test(objectId)) return "sha256";
|
|
3627
|
+
throw new WorktreeAdapterError(`Unsupported Git object id: ${objectId}`);
|
|
3628
|
+
}
|
|
3629
|
+
function hashGitBlobBytes(bytes, objectId) {
|
|
3630
|
+
const hash = createHash(gitObjectHashAlgorithm(objectId));
|
|
3631
|
+
hash.update(`blob ${bytes.byteLength}\0`);
|
|
3632
|
+
hash.update(bytes);
|
|
3633
|
+
return hash.digest("hex");
|
|
3634
|
+
}
|
|
3635
|
+
function hashGitBlobFile(path, objectId) {
|
|
3636
|
+
const before = lstatSync(path);
|
|
3637
|
+
if (!before.isFile() || before.isSymbolicLink()) throw new WorktreeAdapterError(`CodeSurface expected a regular file at ${displayGitPath(path)}`);
|
|
3638
|
+
const noFollow = process.platform === "win32" ? 0 : constants.O_NOFOLLOW;
|
|
3639
|
+
const fd = openSync(path, constants.O_RDONLY | noFollow);
|
|
3640
|
+
try {
|
|
3641
|
+
const opened = fstatSync(fd);
|
|
3642
|
+
if (!opened.isFile() || opened.dev !== before.dev || opened.ino !== before.ino || opened.size !== before.size) throw new WorktreeAdapterError(`CodeSurface file changed while it was being verified: ${displayGitPath(path)}`);
|
|
3643
|
+
const hash = createHash(gitObjectHashAlgorithm(objectId));
|
|
3644
|
+
hash.update(`blob ${opened.size}\0`);
|
|
3645
|
+
const chunk = Buffer.allocUnsafe(FILE_HASH_CHUNK_BYTES);
|
|
3646
|
+
let total = 0;
|
|
3647
|
+
while (true) {
|
|
3648
|
+
const count = readSync(fd, chunk, 0, chunk.length, null);
|
|
3649
|
+
if (count === 0) break;
|
|
3650
|
+
total += count;
|
|
3651
|
+
hash.update(chunk.subarray(0, count));
|
|
3652
|
+
}
|
|
3653
|
+
if (total !== opened.size) throw new WorktreeAdapterError(`CodeSurface file changed while it was being verified: ${displayGitPath(path)}`);
|
|
3654
|
+
return {
|
|
3655
|
+
hash: hash.digest("hex"),
|
|
3656
|
+
executable: (opened.mode & 73) !== 0
|
|
3657
|
+
};
|
|
3658
|
+
} finally {
|
|
3659
|
+
closeSync(fd);
|
|
3660
|
+
}
|
|
3661
|
+
}
|
|
3662
|
+
function assertSymlinkTargetIsBound(root, linkPath, trackedPaths) {
|
|
3663
|
+
const targetBytes = readlinkSync(linkPath, { encoding: "buffer" });
|
|
3664
|
+
const target = targetBytes.toString("utf8");
|
|
3665
|
+
if (!Buffer.from(target, "utf8").equals(targetBytes)) throw new WorktreeAdapterError(`CodeSurface symbolic link has an unsafe target at ${displayGitPath(linkPath)}`);
|
|
3666
|
+
if (isAbsolute(target)) throw new WorktreeAdapterError(`CodeSurface symbolic link escapes its worktree at ${displayGitPath(linkPath)}`);
|
|
3667
|
+
if (!isWithinRoot(root, resolve(dirname(linkPath), target))) throw new WorktreeAdapterError(`CodeSurface symbolic link escapes its worktree at ${displayGitPath(linkPath)}`);
|
|
3668
|
+
let resolvedTarget;
|
|
3669
|
+
try {
|
|
3670
|
+
resolvedTarget = realpathSync(linkPath);
|
|
3671
|
+
} catch (err) {
|
|
3672
|
+
throw new WorktreeAdapterError(`CodeSurface symbolic link target is missing or cyclic at ${displayGitPath(linkPath)}`, err);
|
|
3673
|
+
}
|
|
3674
|
+
if (!isWithinRoot(root, resolvedTarget)) throw new WorktreeAdapterError(`CodeSurface symbolic link escapes its worktree at ${displayGitPath(linkPath)}`);
|
|
3675
|
+
const targetRelative = relative(root, resolvedTarget).split(sep).join("/");
|
|
3676
|
+
if (!(trackedPaths.has(targetRelative) || [...trackedPaths].some((trackedPath) => trackedPath.startsWith(`${targetRelative}/`)))) throw new WorktreeAdapterError(`CodeSurface symbolic link resolves to untracked content at ${displayGitPath(linkPath)}`);
|
|
3677
|
+
}
|
|
3678
|
+
function assertRawTreeMatchesWorktree(git, root, candidateCommit) {
|
|
3679
|
+
const entries = parseGitTreeEntries(gitBytes(git, [
|
|
3680
|
+
"ls-tree",
|
|
3681
|
+
"-r",
|
|
3682
|
+
"-z",
|
|
3683
|
+
"--full-tree",
|
|
3684
|
+
candidateCommit
|
|
3685
|
+
], root));
|
|
3686
|
+
const trackedPaths = new Set(entries.map((entry) => entry.path));
|
|
3687
|
+
for (const entry of entries) {
|
|
3688
|
+
const absolutePath = assertSafeRelativePath(root, entry.path);
|
|
3689
|
+
if (entry.mode === "160000" || entry.mode === "040000") throw new WorktreeAdapterError(`CodeSurface contains a Git submodule whose executable bytes are not bound: ${displayGitPath(entry.path)}`);
|
|
3690
|
+
if (entry.mode === "120000") {
|
|
3691
|
+
if (!lstatSync(absolutePath).isSymbolicLink()) throw new WorktreeAdapterError(`CodeSurface expected a symbolic link at ${displayGitPath(entry.path)}`);
|
|
3692
|
+
if (hashGitBlobBytes(readlinkSync(absolutePath, { encoding: "buffer" }), entry.objectId) !== entry.objectId) throw new WorktreeAdapterError(`CodeSurface raw symbolic-link bytes differ from the candidate tree at ${displayGitPath(entry.path)}`);
|
|
3693
|
+
assertSymlinkTargetIsBound(root, absolutePath, trackedPaths);
|
|
3694
|
+
continue;
|
|
3695
|
+
}
|
|
3696
|
+
if (entry.mode !== "100644" && entry.mode !== "100755") throw new WorktreeAdapterError(`CodeSurface contains unsupported Git mode ${entry.mode} at ${displayGitPath(entry.path)}`);
|
|
3697
|
+
const actual = hashGitBlobFile(absolutePath, entry.objectId);
|
|
3698
|
+
if (actual.hash !== entry.objectId) throw new WorktreeAdapterError(`CodeSurface raw tracked bytes differ from the candidate tree because the worktree changed after finalization at ${displayGitPath(entry.path)}`);
|
|
3699
|
+
if (process.platform !== "win32" && actual.executable !== (entry.mode === "100755")) throw new WorktreeAdapterError(`CodeSurface executable mode differs from the candidate tree at ${displayGitPath(entry.path)}`);
|
|
3700
|
+
}
|
|
3701
|
+
}
|
|
3702
|
+
function verifyCodeSurfaceWithGit(surface, path, git) {
|
|
3703
|
+
assertCodeSurfaceIdentity(surface);
|
|
3704
|
+
if (!existsSync(path)) throw new WorktreeAdapterError(`CodeSurface worktree does not exist: ${path}`);
|
|
3705
|
+
const lexicalRoot = resolve(path);
|
|
3706
|
+
const canonicalRoot = realpathSync(path);
|
|
3707
|
+
if (lstatSync(lexicalRoot).isSymbolicLink() || process.platform !== "win32" && lexicalRoot !== canonicalRoot) throw new WorktreeAdapterError(`CodeSurface worktree locator must not contain a symbolic link: ${path}`);
|
|
3708
|
+
const repoRoot = gitText(git, ["rev-parse", "--show-toplevel"], path);
|
|
3709
|
+
const canonicalRepoRoot = realpathSync(repoRoot);
|
|
3710
|
+
if (canonicalRepoRoot !== canonicalRoot) throw new WorktreeAdapterError(`CodeSurface worktree locator is not the repository root: expected ${repoRoot}, got ${path}`);
|
|
3711
|
+
const indexFlags = gitText(git, ["ls-files", "-v"], path).split("\n").filter((line) => line.length > 0 && (line.startsWith("S ") || /^[a-z]/.test(line)));
|
|
3712
|
+
if (indexFlags.length > 0) throw new WorktreeAdapterError(`CodeSurface worktree uses hidden index entries that cannot be verified: ${path}\n${indexFlags.join("\n")}`);
|
|
3713
|
+
const extraPaths = [...parseGitPathList(gitBytes(git, [
|
|
3714
|
+
"ls-files",
|
|
3715
|
+
"--others",
|
|
3716
|
+
"--exclude-standard",
|
|
3717
|
+
"-z"
|
|
3718
|
+
], path)), ...parseGitPathList(gitBytes(git, [
|
|
3719
|
+
"ls-files",
|
|
3720
|
+
"--others",
|
|
3721
|
+
"--ignored",
|
|
3722
|
+
"--exclude-standard",
|
|
3723
|
+
"-z"
|
|
3724
|
+
], path))];
|
|
3725
|
+
if (extraPaths.length > 0) throw new WorktreeAdapterError(`CodeSurface worktree changed after finalization: ${path}\n${extraPaths.join("\n")}`);
|
|
3726
|
+
const candidateCommit = resolveCommit(git, path, "HEAD");
|
|
3727
|
+
if (candidateCommit !== surface.candidateCommit) throw new WorktreeAdapterError(`CodeSurface candidate commit mismatch: expected ${surface.candidateCommit}, got ${candidateCommit}`);
|
|
3728
|
+
const baseCommit = resolveCommit(git, path, surface.baseCommit);
|
|
3729
|
+
if (baseCommit !== surface.baseCommit) throw new WorktreeAdapterError(`CodeSurface base commit mismatch: expected ${surface.baseCommit}, got ${baseCommit}`);
|
|
3730
|
+
const baseTree = gitText(git, [
|
|
3731
|
+
"rev-parse",
|
|
3732
|
+
"--verify",
|
|
3733
|
+
`${surface.baseCommit}^{tree}`
|
|
3734
|
+
], path);
|
|
3735
|
+
if (baseTree !== surface.baseTree) throw new WorktreeAdapterError(`CodeSurface base tree mismatch: expected ${surface.baseTree}, got ${baseTree}`);
|
|
3736
|
+
const candidateTree = gitText(git, [
|
|
3737
|
+
"rev-parse",
|
|
3738
|
+
"--verify",
|
|
3739
|
+
`${surface.candidateCommit}^{tree}`
|
|
3740
|
+
], path);
|
|
3741
|
+
if (candidateTree !== surface.candidateTree) throw new WorktreeAdapterError(`CodeSurface tree mismatch: expected ${surface.candidateTree}, got ${candidateTree}`);
|
|
3742
|
+
try {
|
|
3743
|
+
gitText(git, [
|
|
3744
|
+
"diff-index",
|
|
3745
|
+
"--cached",
|
|
3746
|
+
"--quiet",
|
|
3747
|
+
"--no-ext-diff",
|
|
3748
|
+
"--no-textconv",
|
|
3749
|
+
surface.candidateCommit,
|
|
3750
|
+
"--"
|
|
3751
|
+
], path);
|
|
3752
|
+
} catch (err) {
|
|
3753
|
+
throw new WorktreeAdapterError(`CodeSurface index differs from finalized candidate: ${path}`, err);
|
|
3754
|
+
}
|
|
3755
|
+
try {
|
|
3756
|
+
assertRawTreeMatchesWorktree(git, canonicalRepoRoot, surface.candidateCommit);
|
|
3757
|
+
} catch (err) {
|
|
3758
|
+
if (err instanceof WorktreeAdapterError) throw err;
|
|
3759
|
+
throw new WorktreeAdapterError(`CodeSurface raw tree verification failed: ${path}`, err);
|
|
3760
|
+
}
|
|
3761
|
+
if (gitText(git, [
|
|
3762
|
+
"merge-base",
|
|
3763
|
+
surface.baseCommit,
|
|
3764
|
+
surface.candidateCommit
|
|
3765
|
+
], path) !== surface.baseCommit) throw new WorktreeAdapterError(`CodeSurface candidate ${surface.candidateCommit} does not descend from base ${surface.baseCommit}`);
|
|
3766
|
+
const patch = patchBytes(git, path, surface.baseCommit, surface.candidateCommit);
|
|
3767
|
+
const actualPatchHash = sha256(patch);
|
|
3768
|
+
if (actualPatchHash !== surface.patch.sha256 || patch.byteLength !== surface.patch.byteLength) throw new WorktreeAdapterError(`CodeSurface patch mismatch: expected ${surface.patch.sha256}/${surface.patch.byteLength} bytes, got ${actualPatchHash}/${patch.byteLength} bytes`);
|
|
3769
|
+
return {
|
|
3770
|
+
path,
|
|
3771
|
+
repoRoot: canonicalRepoRoot,
|
|
3772
|
+
contentHash: surfaceContentHash(surface),
|
|
3773
|
+
patchBytes: new Uint8Array(patch)
|
|
3774
|
+
};
|
|
3775
|
+
}
|
|
3776
|
+
/** Slugify a label into a branch-safe segment. */
|
|
3777
|
+
function slug(label) {
|
|
3778
|
+
return label.toLowerCase().replace(/[^a-z0-9]+/g, "-").replace(/^-+|-+$/g, "").slice(0, 48) || "candidate";
|
|
3779
|
+
}
|
|
3780
|
+
/**
|
|
3781
|
+
* Git-backed `WorktreeAdapter`: creates isolated worktrees on fresh branches, commits agent changes, and discards losers.
|
|
3782
|
+
*/
|
|
3783
|
+
function gitWorktreeAdapter(opts) {
|
|
3784
|
+
const git = opts.git ?? defaultGit;
|
|
3785
|
+
const worktreeDir = opts.worktreeDir ?? join(opts.repoRoot, ".worktrees");
|
|
3786
|
+
const branchPrefix = opts.branchPrefix ?? "improve";
|
|
3787
|
+
return {
|
|
3788
|
+
async create({ baseRef, label }) {
|
|
3789
|
+
const id = `${slug(label)}-${Date.now().toString(36)}-${Math.random().toString(36).slice(2, 6)}`;
|
|
3790
|
+
const branch = `${branchPrefix}/${id}`;
|
|
3791
|
+
const path = join(worktreeDir, id);
|
|
3792
|
+
const baseCommit = resolveCommit(git, opts.repoRoot, baseRef);
|
|
3793
|
+
const baseTree = gitText(git, [
|
|
3794
|
+
"rev-parse",
|
|
3795
|
+
"--verify",
|
|
3796
|
+
`${baseCommit}^{tree}`
|
|
3797
|
+
], opts.repoRoot);
|
|
3798
|
+
gitText(git, [
|
|
3799
|
+
"worktree",
|
|
3800
|
+
"add",
|
|
3801
|
+
"-b",
|
|
3802
|
+
branch,
|
|
3803
|
+
path,
|
|
3804
|
+
baseCommit
|
|
3805
|
+
], opts.repoRoot);
|
|
3806
|
+
return {
|
|
3807
|
+
path,
|
|
3808
|
+
branch,
|
|
3809
|
+
baseRef,
|
|
3810
|
+
baseCommit,
|
|
3811
|
+
baseTree
|
|
3812
|
+
};
|
|
3813
|
+
},
|
|
3814
|
+
async finalize(worktree, summary) {
|
|
3815
|
+
if (gitText(git, [
|
|
3816
|
+
"status",
|
|
3817
|
+
"--porcelain=v1",
|
|
3818
|
+
"--untracked-files=all"
|
|
3819
|
+
], worktree.path).length > 0) {
|
|
3820
|
+
gitText(git, ["add", "-A"], worktree.path);
|
|
3821
|
+
gitText(git, [
|
|
3822
|
+
"commit",
|
|
3823
|
+
"-m",
|
|
3824
|
+
summary
|
|
3825
|
+
], worktree.path);
|
|
3826
|
+
}
|
|
3827
|
+
const candidateCommit = resolveCommit(git, worktree.path, "HEAD");
|
|
3828
|
+
const candidateTree = gitText(git, [
|
|
3829
|
+
"rev-parse",
|
|
3830
|
+
"--verify",
|
|
3831
|
+
`${candidateCommit}^{tree}`
|
|
3832
|
+
], worktree.path);
|
|
3833
|
+
const patch = patchBytes(git, worktree.path, worktree.baseCommit, candidateCommit);
|
|
3834
|
+
const surface = {
|
|
3835
|
+
kind: "code",
|
|
3836
|
+
worktreeRef: worktree.path,
|
|
3837
|
+
baseRef: worktree.baseRef,
|
|
3838
|
+
baseCommit: worktree.baseCommit,
|
|
3839
|
+
baseTree: worktree.baseTree,
|
|
3840
|
+
candidateCommit,
|
|
3841
|
+
candidateTree,
|
|
3842
|
+
patch: {
|
|
3843
|
+
format: "git-diff-binary",
|
|
3844
|
+
sha256: sha256(patch),
|
|
3845
|
+
byteLength: patch.byteLength
|
|
3846
|
+
},
|
|
3847
|
+
summary
|
|
3848
|
+
};
|
|
3849
|
+
verifyCodeSurfaceWithGit(surface, worktree.path, git);
|
|
3850
|
+
return surface;
|
|
3851
|
+
},
|
|
3852
|
+
async discard(worktree) {
|
|
3853
|
+
const failures = [reconcileAbsent(() => hasRegisteredWorktree(git, opts.repoRoot, worktree.path), () => gitText(git, [
|
|
3854
|
+
"worktree",
|
|
3855
|
+
"remove",
|
|
3856
|
+
"--force",
|
|
3857
|
+
"--",
|
|
3858
|
+
worktree.path
|
|
3859
|
+
], opts.repoRoot)), reconcileAbsent(() => hasLocalBranch(git, opts.repoRoot, worktree.branch), () => gitText(git, [
|
|
3860
|
+
"branch",
|
|
3861
|
+
"-D",
|
|
3862
|
+
"--",
|
|
3863
|
+
worktree.branch
|
|
3864
|
+
], opts.repoRoot))].filter((failure) => failure !== void 0);
|
|
3865
|
+
if (failures.length > 0) {
|
|
3866
|
+
const cause = failures.length === 1 ? failures[0] : new AggregateError(failures, "Multiple Git resources could not be removed");
|
|
3867
|
+
throw new WorktreeAdapterError(`Failed to discard worktree ${worktree.path} and branch ${worktree.branch}`, cause);
|
|
3868
|
+
}
|
|
3869
|
+
}
|
|
3870
|
+
};
|
|
3871
|
+
}
|
|
3872
|
+
/** Verify a finalized code surface against its current checkout. This rejects
|
|
3873
|
+
* dirty/ignored files, moved refs, missing Git objects, raw byte/mode
|
|
3874
|
+
* mismatches, external symlinks, and submodules. */
|
|
3875
|
+
function verifyCodeSurface(surface, worktreeDir) {
|
|
3876
|
+
return verifyCodeSurfaceWithGit(surface, unresolvedWorktreePath(surface, worktreeDir), defaultGit);
|
|
3877
|
+
}
|
|
3878
|
+
/** Resolve a code candidate for evaluation only after verifying its immutable
|
|
3879
|
+
* identity against the checkout at `worktreeRef`. */
|
|
3880
|
+
function resolveWorktreePath(surface, worktreeDir) {
|
|
3881
|
+
return verifyCodeSurface(surface, worktreeDir).path;
|
|
3882
|
+
}
|
|
3883
|
+
//#endregion
|
|
3884
|
+
export { planEvalFixtureRun as A, harnessAxisOf as B, rolloutArgumentDiff as C, discoverEvalFixtures as D, neutralizationGate as E, HARNESS_NATIVE_MODEL as F, parseCorrectnessResponse as G, completionVerdict as H, agentProfileHash as I, verifyCompletion as K, agentProfileId as L, buildAnalystSurfaceDispatch as M, failureModeRecallJudge as N, loadEvalFixture as O, CODING_HARNESSES as P, agentProfileModelId as R, classifyUngroundedLiterals as S, sequentialPairedGate as T, createLlmCorrectnessChecker as U, extractProducedState as V, createTokenRecallChecker as W, scoreboardSummary as _, isTransientTransportFailure as a, FsLabeledScenarioStore as b, openSearchLedger as c, selectDiscriminative as d, ProfileMatrixError as f, scoreUserStory as g, renderScoreboardMarkdown as h, verifyCodeSurface as i, analyzeCrossSurfaceInteractions as j, loadEvalFixtureScenarios as k, validateSearchLedgerEvent as l, makePlaybackDispatch as m, gitWorktreeAdapter as n, FileSearchLedger as o, runProfileMatrix as p, resolveWorktreePath as r, SEARCH_LEDGER_SCHEMA as s, WorktreeAdapterError as t, scoreDiscrimination as u, userStoryScoreboard as v, sequentialDecide as w, LabeledScenarioStoreError as x, neutralizeText as y, expandProfileAxes as z };
|
|
3885
|
+
|
|
3886
|
+
//# sourceMappingURL=campaign-aKJt6emI.js.map
|