@tangle-network/agent-eval 0.128.2 → 0.130.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +279 -0
- package/README.md +19 -0
- package/dist/active-curriculum-C4mk67HP.js +214 -0
- package/dist/active-curriculum-C4mk67HP.js.map +1 -0
- package/dist/adversarial-smnADNFS.d.ts +21 -0
- package/dist/adversarial-smnADNFS.d.ts.map +1 -0
- package/dist/analyst/index.d.ts +83 -2932
- package/dist/analyst/index.d.ts.map +1 -0
- package/dist/analyst/index.js +319 -364
- package/dist/analyst/index.js.map +1 -1
- package/dist/analyst-BkTS3C58.d.ts +89 -0
- package/dist/analyst-BkTS3C58.d.ts.map +1 -0
- package/dist/analyst-LsnNpSkm.js +152 -0
- package/dist/analyst-LsnNpSkm.js.map +1 -0
- package/dist/analyze-runs-C1CavBMk.js +1067 -0
- package/dist/analyze-runs-C1CavBMk.js.map +1 -0
- package/dist/analyze-runs-FsgYCinh.d.ts +72 -0
- package/dist/analyze-runs-FsgYCinh.d.ts.map +1 -0
- package/dist/attribute-vocabulary-DLJ6303h.d.ts +54 -0
- package/dist/attribute-vocabulary-DLJ6303h.d.ts.map +1 -0
- package/dist/authenticity/index.d.ts +81 -79
- package/dist/authenticity/index.d.ts.map +1 -0
- package/dist/authenticity/index.js +209 -193
- package/dist/authenticity/index.js.map +1 -1
- package/dist/baseline-HsBvw_dk.js +550 -0
- package/dist/baseline-HsBvw_dk.js.map +1 -0
- package/dist/baseline-hG3K85h4.d.ts +125 -0
- package/dist/baseline-hG3K85h4.d.ts.map +1 -0
- package/dist/belief-state/index.d.ts +448 -1205
- package/dist/belief-state/index.d.ts.map +1 -0
- package/dist/belief-state/index.js +1617 -1710
- package/dist/belief-state/index.js.map +1 -1
- package/dist/benchmarks/index.d.ts +2 -894
- package/dist/benchmarks/index.js +2 -59
- package/dist/benchmarks-DviOvUNr.js +754 -0
- package/dist/benchmarks-DviOvUNr.js.map +1 -0
- package/dist/builder-eval/index.d.ts +150 -662
- package/dist/builder-eval/index.d.ts.map +1 -0
- package/dist/builder-eval/index.js +356 -345
- package/dist/builder-eval/index.js.map +1 -1
- package/dist/calibration-CNWWA6K8.js +94 -0
- package/dist/calibration-CNWWA6K8.js.map +1 -0
- package/dist/campaign/index.d.ts +5 -6390
- package/dist/campaign/index.js +3 -212
- package/dist/campaign-CBKZvQ1H.js +3885 -0
- package/dist/campaign-CBKZvQ1H.js.map +1 -0
- package/dist/cli.d.ts +1 -1
- package/dist/cli.js +164 -174
- package/dist/cli.js.map +1 -1
- package/dist/client-C97NMzqi.d.ts +581 -0
- package/dist/client-C97NMzqi.d.ts.map +1 -0
- package/dist/client-CYzbdJOZ.js +637 -0
- package/dist/client-CYzbdJOZ.js.map +1 -0
- package/dist/code-agent-session-BjkMTQ7H.js +1390 -0
- package/dist/code-agent-session-BjkMTQ7H.js.map +1 -0
- package/dist/code-agent-session-DqqgOJaz.d.ts +143 -0
- package/dist/code-agent-session-DqqgOJaz.d.ts.map +1 -0
- package/dist/concurrency-DIxRZF_J.js +85 -0
- package/dist/concurrency-DIxRZF_J.js.map +1 -0
- package/dist/contract/index.d.ts +645 -5605
- package/dist/contract/index.d.ts.map +1 -0
- package/dist/contract/index.js +1730 -1937
- package/dist/contract/index.js.map +1 -1
- package/dist/control.d.ts +3 -1030
- package/dist/control.js +2 -32
- package/dist/cost-ledger-DIgQUFZZ.js +801 -0
- package/dist/cost-ledger-DIgQUFZZ.js.map +1 -0
- package/dist/cost-ledger-Dye6jCgg.d.ts +236 -0
- package/dist/cost-ledger-Dye6jCgg.d.ts.map +1 -0
- package/dist/dataset-BvtnC8Dc.d.ts +115 -0
- package/dist/dataset-BvtnC8Dc.d.ts.map +1 -0
- package/dist/default-registry-C-vFCSEc.js +2579 -0
- package/dist/default-registry-C-vFCSEc.js.map +1 -0
- package/dist/default-registry-CNPo-Vsb.d.ts +540 -0
- package/dist/default-registry-CNPo-Vsb.d.ts.map +1 -0
- package/dist/emitter-CPBAhxum.js +266 -0
- package/dist/emitter-CPBAhxum.js.map +1 -0
- package/dist/emitter-DGQGoLyj.d.ts +113 -0
- package/dist/emitter-DGQGoLyj.d.ts.map +1 -0
- package/dist/errors-8YnH8WlF.js +64 -0
- package/dist/errors-8YnH8WlF.js.map +1 -0
- package/dist/errors-CEk209JS.d.ts +76 -0
- package/dist/errors-CEk209JS.d.ts.map +1 -0
- package/dist/eval-campaign-DEm6c8ru.js +349 -0
- package/dist/eval-campaign-DEm6c8ru.js.map +1 -0
- package/dist/execution-tracks-CpgFPpS5.js +93 -0
- package/dist/execution-tracks-CpgFPpS5.js.map +1 -0
- package/dist/exporters-q9iL-2Jf.js +148 -0
- package/dist/exporters-q9iL-2Jf.js.map +1 -0
- package/dist/extract-usage-BrQ8mCLX.js +155 -0
- package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
- package/dist/failure-cluster-CqcvCcdR.d.ts +59 -0
- package/dist/failure-cluster-CqcvCcdR.d.ts.map +1 -0
- package/dist/feedback-trajectory-CVaeREXV.d.ts +340 -0
- package/dist/feedback-trajectory-CVaeREXV.d.ts.map +1 -0
- package/dist/fuzz.d.ts +320 -646
- package/dist/fuzz.d.ts.map +1 -0
- package/dist/fuzz.js +670 -617
- package/dist/fuzz.js.map +1 -1
- package/dist/hf-dataset-DBJXXoY1.js +763 -0
- package/dist/hf-dataset-DBJXXoY1.js.map +1 -0
- package/dist/hosted/index.d.ts +11 -831
- package/dist/hosted/index.d.ts.map +1 -0
- package/dist/hosted/index.js +2 -37
- package/dist/index-2JJSA6-r2.d.ts +926 -0
- package/dist/index-2JJSA6-r2.d.ts.map +1 -0
- package/dist/index-6N0aYmpW.d.ts +217 -0
- package/dist/index-6N0aYmpW.d.ts.map +1 -0
- package/dist/index-BAvgST_9.d.ts +131 -0
- package/dist/index-BAvgST_9.d.ts.map +1 -0
- package/dist/index-BvnJuTGD.d.ts +68 -0
- package/dist/index-BvnJuTGD.d.ts.map +1 -0
- package/dist/index-C61Wi7yg.d.ts +547 -0
- package/dist/index-C61Wi7yg.d.ts.map +1 -0
- package/dist/index-CAPUUKaM.d.ts +335 -0
- package/dist/index-CAPUUKaM.d.ts.map +1 -0
- package/dist/index-DE5fb3EC.d.ts +2244 -0
- package/dist/index-DE5fb3EC.d.ts.map +1 -0
- package/dist/index-DSC51roc.d.ts +102 -0
- package/dist/index-DSC51roc.d.ts.map +1 -0
- package/dist/index.d.ts +3776 -15120
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +11185 -11191
- package/dist/index.js.map +1 -1
- package/dist/integrity-BzRbCHzi.js +424 -0
- package/dist/integrity-BzRbCHzi.js.map +1 -0
- package/dist/integrity-rmVhXWA7.d.ts +61 -0
- package/dist/integrity-rmVhXWA7.d.ts.map +1 -0
- package/dist/judge-calibration-DFtEMlde.d.ts +146 -0
- package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
- package/dist/ledger-core/index.d.ts +2 -0
- package/dist/ledger-core/index.js +2 -0
- package/dist/ledger-core-DtZz1RG0.js +388 -0
- package/dist/ledger-core-DtZz1RG0.js.map +1 -0
- package/dist/llm-client--GR4JbZE.js +687 -0
- package/dist/llm-client--GR4JbZE.js.map +1 -0
- package/dist/llm-client-B_nIBlYo.d.ts +290 -0
- package/dist/llm-client-B_nIBlYo.d.ts.map +1 -0
- package/dist/matrix/index.d.ts +3 -155
- package/dist/matrix/index.js +2 -8
- package/dist/matrix-BzQnu2S6.js +270 -0
- package/dist/matrix-BzQnu2S6.js.map +1 -0
- package/dist/meta-eval/index.d.ts +4 -1027
- package/dist/meta-eval/index.js +393 -390
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/metrics-C9YY1OcL.js +239 -0
- package/dist/metrics-C9YY1OcL.js.map +1 -0
- package/dist/mint-yN2M2eh0.js +201 -0
- package/dist/mint-yN2M2eh0.js.map +1 -0
- package/dist/multi-layer-verifier-BHY1gWAc.d.ts +138 -0
- package/dist/multi-layer-verifier-BHY1gWAc.d.ts.map +1 -0
- package/dist/multishot/index.d.ts +273 -481
- package/dist/multishot/index.d.ts.map +1 -0
- package/dist/multishot/index.js +595 -548
- package/dist/multishot/index.js.map +1 -1
- package/dist/off-policy-DvgzvtIx.js +220 -0
- package/dist/off-policy-DvgzvtIx.js.map +1 -0
- package/dist/off-policy-mskQw8Mb.d.ts +153 -0
- package/dist/off-policy-mskQw8Mb.d.ts.map +1 -0
- package/dist/openapi.json +1 -1
- package/dist/opencode-sqlite-BGrHeDu3.js +318 -0
- package/dist/opencode-sqlite-BGrHeDu3.js.map +1 -0
- package/dist/outcome-store-BYHIuO0e.d.ts +65 -0
- package/dist/outcome-store-BYHIuO0e.d.ts.map +1 -0
- package/dist/outcome-store-ChBKlTd_.js +75 -0
- package/dist/outcome-store-ChBKlTd_.js.map +1 -0
- package/dist/paired-arms-D9D0wXj2.js +260 -0
- package/dist/paired-arms-D9D0wXj2.js.map +1 -0
- package/dist/pipelines/index.d.ts +95 -532
- package/dist/pipelines/index.d.ts.map +1 -0
- package/dist/pipelines/index.js +497 -478
- package/dist/pipelines/index.js.map +1 -1
- package/dist/pre-registration-DakwTRXk.js +96 -0
- package/dist/pre-registration-DakwTRXk.js.map +1 -0
- package/dist/propose-review-control-LhIWGzHl.js +1458 -0
- package/dist/propose-review-control-LhIWGzHl.js.map +1 -0
- package/dist/query-CJ_DX8vl.d.ts +28 -0
- package/dist/query-CJ_DX8vl.d.ts.map +1 -0
- package/dist/query-Di7eEQ79.js +83 -0
- package/dist/query-Di7eEQ79.js.map +1 -0
- package/dist/raw-provider-sink-BQd7mzyT.js +201 -0
- package/dist/raw-provider-sink-BQd7mzyT.js.map +1 -0
- package/dist/raw-provider-sink-BU29Sh8h.d.ts +134 -0
- package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +1 -0
- package/dist/redact-7Aq1ukl-.js +107 -0
- package/dist/redact-7Aq1ukl-.js.map +1 -0
- package/dist/release-report-Cz9NKH39.js +603 -0
- package/dist/release-report-Cz9NKH39.js.map +1 -0
- package/dist/release-report-mvB2G4_J.d.ts +244 -0
- package/dist/release-report-mvB2G4_J.d.ts.map +1 -0
- package/dist/replay-GnyotH0J.js +1741 -0
- package/dist/replay-GnyotH0J.js.map +1 -0
- package/dist/replay-OoidtG1E.d.ts +749 -0
- package/dist/replay-OoidtG1E.d.ts.map +1 -0
- package/dist/reporting.d.ts +6 -1298
- package/dist/reporting.js +6 -50
- package/dist/researcher-CwTdwXG1.d.ts +314 -0
- package/dist/researcher-CwTdwXG1.d.ts.map +1 -0
- package/dist/reward-hacking-eAnOsynk.d.ts +224 -0
- package/dist/reward-hacking-eAnOsynk.d.ts.map +1 -0
- package/dist/reward-hacking-qipEpKvY.js +596 -0
- package/dist/reward-hacking-qipEpKvY.js.map +1 -0
- package/dist/reward-nw2xZGZG.js +137 -0
- package/dist/reward-nw2xZGZG.js.map +1 -0
- package/dist/rl.d.ts +916 -3596
- package/dist/rl.d.ts.map +1 -0
- package/dist/rl.js +2362 -1751
- package/dist/rl.js.map +1 -1
- package/dist/rolldown-runtime-8H4AJuhK.js +14 -0
- package/dist/rollout/index.d.ts +3 -1048
- package/dist/rollout/index.js +8 -110
- package/dist/rollout-BOYjemfR.js +624 -0
- package/dist/rollout-BOYjemfR.js.map +1 -0
- package/dist/rubric-predictive-validity-B3xmbmS1.js +141 -0
- package/dist/rubric-predictive-validity-B3xmbmS1.js.map +1 -0
- package/dist/rubric-predictive-validity-Ku_clp_1.d.ts +74 -0
- package/dist/rubric-predictive-validity-Ku_clp_1.d.ts.map +1 -0
- package/dist/run-evidence-ClRX_8A9.d.ts +225 -0
- package/dist/run-evidence-ClRX_8A9.d.ts.map +1 -0
- package/dist/run-record-BuoE80Dq.js +467 -0
- package/dist/run-record-BuoE80Dq.js.map +1 -0
- package/dist/run-record-CnZu_gjl.d.ts +357 -0
- package/dist/run-record-CnZu_gjl.d.ts.map +1 -0
- package/dist/run-score-iEEAWiBY.js +41 -0
- package/dist/run-score-iEEAWiBY.js.map +1 -0
- package/dist/runtime-trajectory-1gyaTOoC.js +93 -0
- package/dist/runtime-trajectory-1gyaTOoC.js.map +1 -0
- package/dist/runtime-trajectory-BvSZcCHD.d.ts +50 -0
- package/dist/runtime-trajectory-BvSZcCHD.d.ts.map +1 -0
- package/dist/schema-BtVldJ3T.d.ts +206 -0
- package/dist/schema-BtVldJ3T.d.ts.map +1 -0
- package/dist/schema-C6DW4ZHR.js +821 -0
- package/dist/schema-C6DW4ZHR.js.map +1 -0
- package/dist/schema-CRhEY1SO.js +69 -0
- package/dist/schema-CRhEY1SO.js.map +1 -0
- package/dist/schema-Cef2cFmb.d.ts +408 -0
- package/dist/schema-Cef2cFmb.d.ts.map +1 -0
- package/dist/semantic-concept-judge-B6cWNJ2K.js +725 -0
- package/dist/semantic-concept-judge-B6cWNJ2K.js.map +1 -0
- package/dist/sequential-Br0mAPHA.js +148 -0
- package/dist/sequential-Br0mAPHA.js.map +1 -0
- package/dist/sequential-CYwq6Ff_.d.ts +141 -0
- package/dist/sequential-CYwq6Ff_.d.ts.map +1 -0
- package/dist/series-convergence-CjO2QdRW.js +43 -0
- package/dist/series-convergence-CjO2QdRW.js.map +1 -0
- package/dist/series-convergence-ofsqPWhs.d.ts +35 -0
- package/dist/series-convergence-ofsqPWhs.d.ts.map +1 -0
- package/dist/server-m5D9cvnG.js +1040 -0
- package/dist/server-m5D9cvnG.js.map +1 -0
- package/dist/skill-usage-C_pXm7lP.d.ts +534 -0
- package/dist/skill-usage-C_pXm7lP.d.ts.map +1 -0
- package/dist/skillopt-optimization-method-B7wX7XkF.d.ts +1739 -0
- package/dist/skillopt-optimization-method-B7wX7XkF.d.ts.map +1 -0
- package/dist/skillopt-optimization-method-D4ODwFVV.js +7712 -0
- package/dist/skillopt-optimization-method-D4ODwFVV.js.map +1 -0
- package/dist/statistics-Cmj6nynr.d.ts +514 -0
- package/dist/statistics-Cmj6nynr.d.ts.map +1 -0
- package/dist/statistics-CnnxdpOg.js +1437 -0
- package/dist/statistics-CnnxdpOg.js.map +1 -0
- package/dist/store-CT9YIIve.d.ts +117 -0
- package/dist/store-CT9YIIve.d.ts.map +1 -0
- package/dist/store-CxJry_cs.d.ts +229 -0
- package/dist/store-CxJry_cs.d.ts.map +1 -0
- package/dist/storyboard/index.d.ts +84 -203
- package/dist/storyboard/index.d.ts.map +1 -0
- package/dist/storyboard/index.js +609 -542
- package/dist/storyboard/index.js.map +1 -1
- package/dist/summary-report-BNs5nmXI.js +862 -0
- package/dist/summary-report-BNs5nmXI.js.map +1 -0
- package/dist/summary-report-Cj9gdw4i.d.ts +407 -0
- package/dist/summary-report-Cj9gdw4i.d.ts.map +1 -0
- package/dist/supervisor-run/index.d.ts +2 -849
- package/dist/supervisor-run/index.js +2 -64
- package/dist/supervisor-run-_lnTLM3z.js +1679 -0
- package/dist/supervisor-run-_lnTLM3z.js.map +1 -0
- package/dist/task-failure-attributes-CQZlB3et.js +311 -0
- package/dist/task-failure-attributes-CQZlB3et.js.map +1 -0
- package/dist/test-graded-scenario-BsqWLmPt.js +318 -0
- package/dist/test-graded-scenario-BsqWLmPt.js.map +1 -0
- package/dist/test-graded-scenario-D1TaI2va.d.ts +141 -0
- package/dist/test-graded-scenario-D1TaI2va.d.ts.map +1 -0
- package/dist/tools-BmuN627J.js +1085 -0
- package/dist/tools-BmuN627J.js.map +1 -0
- package/dist/trace-attributes.d.ts +2 -52
- package/dist/trace-attributes.js +131 -61
- package/dist/trace-attributes.js.map +1 -1
- package/dist/traces.d.ts +12 -2365
- package/dist/traces.js +12 -251
- package/dist/trajectory-D_7rLrvE.js +56 -0
- package/dist/trajectory-D_7rLrvE.js.map +1 -0
- package/dist/types-DGsxbAEd.d.ts +387 -0
- package/dist/types-DGsxbAEd.d.ts.map +1 -0
- package/dist/types-k9tZGKUg.d.ts +640 -0
- package/dist/types-k9tZGKUg.d.ts.map +1 -0
- package/dist/verdict-Dps8_okt.d.ts +37 -0
- package/dist/verdict-Dps8_okt.d.ts.map +1 -0
- package/dist/wire/index.d.ts +702 -1174
- package/dist/wire/index.d.ts.map +1 -0
- package/dist/wire/index.js +2 -81
- package/docs/feature-guide.md +1 -1
- package/docs/rollout.md +116 -2
- package/package.json +18 -10
- package/dist/benchmarks/index.js.map +0 -1
- package/dist/campaign/index.js.map +0 -1
- package/dist/chunk-2JX3CFMB.js +0 -695
- package/dist/chunk-2JX3CFMB.js.map +0 -1
- package/dist/chunk-2MKQIFS4.js +0 -183
- package/dist/chunk-2MKQIFS4.js.map +0 -1
- package/dist/chunk-3RF76KTD.js +0 -84
- package/dist/chunk-3RF76KTD.js.map +0 -1
- package/dist/chunk-5DTSBUL2.js +0 -159
- package/dist/chunk-5DTSBUL2.js.map +0 -1
- package/dist/chunk-7ZZMD7UK.js +0 -386
- package/dist/chunk-7ZZMD7UK.js.map +0 -1
- package/dist/chunk-BOD4O7OF.js +0 -40
- package/dist/chunk-BOD4O7OF.js.map +0 -1
- package/dist/chunk-BYT7ELPS.js +0 -1553
- package/dist/chunk-BYT7ELPS.js.map +0 -1
- package/dist/chunk-DJKY2TSY.js +0 -2428
- package/dist/chunk-DJKY2TSY.js.map +0 -1
- package/dist/chunk-DPUHNQLN.js +0 -232
- package/dist/chunk-DPUHNQLN.js.map +0 -1
- package/dist/chunk-DRYIUNWY.js +0 -622
- package/dist/chunk-DRYIUNWY.js.map +0 -1
- package/dist/chunk-EJGRPCO3.js +0 -617
- package/dist/chunk-EJGRPCO3.js.map +0 -1
- package/dist/chunk-EOSZT7PL.js +0 -2001
- package/dist/chunk-EOSZT7PL.js.map +0 -1
- package/dist/chunk-EZJEIH2R.js +0 -1559
- package/dist/chunk-EZJEIH2R.js.map +0 -1
- package/dist/chunk-GGE4NNQT.js +0 -65
- package/dist/chunk-GGE4NNQT.js.map +0 -1
- package/dist/chunk-HHWE3POT.js +0 -94
- package/dist/chunk-HHWE3POT.js.map +0 -1
- package/dist/chunk-IHQDPH7D.js +0 -171
- package/dist/chunk-IHQDPH7D.js.map +0 -1
- package/dist/chunk-JHCHEVET.js +0 -274
- package/dist/chunk-JHCHEVET.js.map +0 -1
- package/dist/chunk-K4DBDHLK.js +0 -158
- package/dist/chunk-K4DBDHLK.js.map +0 -1
- package/dist/chunk-K6N6XJJX.js +0 -306
- package/dist/chunk-K6N6XJJX.js.map +0 -1
- package/dist/chunk-MA6HLL3S.js +0 -65
- package/dist/chunk-MA6HLL3S.js.map +0 -1
- package/dist/chunk-MAZ26DC7.js +0 -99
- package/dist/chunk-MAZ26DC7.js.map +0 -1
- package/dist/chunk-MHELPNRP.js +0 -1212
- package/dist/chunk-MHELPNRP.js.map +0 -1
- package/dist/chunk-NACAGYSY.js +0 -1040
- package/dist/chunk-NACAGYSY.js.map +0 -1
- package/dist/chunk-NKAGIDE2.js +0 -7633
- package/dist/chunk-NKAGIDE2.js.map +0 -1
- package/dist/chunk-NPCTHQIO.js +0 -91
- package/dist/chunk-NPCTHQIO.js.map +0 -1
- package/dist/chunk-NYLOYM6N.js +0 -332
- package/dist/chunk-NYLOYM6N.js.map +0 -1
- package/dist/chunk-ONWEPEDO.js +0 -57
- package/dist/chunk-ONWEPEDO.js.map +0 -1
- package/dist/chunk-P5W7RQKK.js +0 -576
- package/dist/chunk-P5W7RQKK.js.map +0 -1
- package/dist/chunk-P6FYH6K4.js +0 -1161
- package/dist/chunk-P6FYH6K4.js.map +0 -1
- package/dist/chunk-PBE2LOSS.js +0 -669
- package/dist/chunk-PBE2LOSS.js.map +0 -1
- package/dist/chunk-PC4UYEBM.js +0 -166
- package/dist/chunk-PC4UYEBM.js.map +0 -1
- package/dist/chunk-PXE2VKMX.js +0 -140
- package/dist/chunk-PXE2VKMX.js.map +0 -1
- package/dist/chunk-PZ5AY32C.js +0 -10
- package/dist/chunk-PZ5AY32C.js.map +0 -1
- package/dist/chunk-RZTMDUO7.js +0 -49
- package/dist/chunk-RZTMDUO7.js.map +0 -1
- package/dist/chunk-S5YLIBFX.js +0 -136
- package/dist/chunk-S5YLIBFX.js.map +0 -1
- package/dist/chunk-SZLVEKMJ.js +0 -1446
- package/dist/chunk-SZLVEKMJ.js.map +0 -1
- package/dist/chunk-T4SQEITX.js +0 -95
- package/dist/chunk-T4SQEITX.js.map +0 -1
- package/dist/chunk-TBL77AUT.js +0 -355
- package/dist/chunk-TBL77AUT.js.map +0 -1
- package/dist/chunk-TSN7JT6D.js +0 -1646
- package/dist/chunk-TSN7JT6D.js.map +0 -1
- package/dist/chunk-TT4KNT67.js +0 -124
- package/dist/chunk-TT4KNT67.js.map +0 -1
- package/dist/chunk-UB2LOJ6Q.js +0 -4461
- package/dist/chunk-UB2LOJ6Q.js.map +0 -1
- package/dist/chunk-UWZZKKU7.js +0 -237
- package/dist/chunk-UWZZKKU7.js.map +0 -1
- package/dist/chunk-VBQ3CRKH.js +0 -291
- package/dist/chunk-VBQ3CRKH.js.map +0 -1
- package/dist/chunk-VGRCHJON.js +0 -163
- package/dist/chunk-VGRCHJON.js.map +0 -1
- package/dist/chunk-VI2UW6B6.js +0 -162
- package/dist/chunk-VI2UW6B6.js.map +0 -1
- package/dist/chunk-VLOATJQ2.js +0 -908
- package/dist/chunk-VLOATJQ2.js.map +0 -1
- package/dist/chunk-VQMK5FMP.js +0 -247
- package/dist/chunk-VQMK5FMP.js.map +0 -1
- package/dist/chunk-VZSRQ272.js +0 -149
- package/dist/chunk-VZSRQ272.js.map +0 -1
- package/dist/chunk-WGXIEX7P.js +0 -116
- package/dist/chunk-WGXIEX7P.js.map +0 -1
- package/dist/chunk-WS3NZZQQ.js +0 -929
- package/dist/chunk-WS3NZZQQ.js.map +0 -1
- package/dist/chunk-XDWDC2MP.js +0 -695
- package/dist/chunk-XDWDC2MP.js.map +0 -1
- package/dist/chunk-XPRT64IE.js +0 -766
- package/dist/chunk-XPRT64IE.js.map +0 -1
- package/dist/chunk-YJBNWCAA.js +0 -1056
- package/dist/chunk-YJBNWCAA.js.map +0 -1
- package/dist/chunk-ZET2UAYW.js +0 -89
- package/dist/chunk-ZET2UAYW.js.map +0 -1
- package/dist/chunk-ZUUWPZCV.js +0 -752
- package/dist/chunk-ZUUWPZCV.js.map +0 -1
- package/dist/control.js.map +0 -1
- package/dist/hosted/index.js.map +0 -1
- package/dist/matrix/index.js.map +0 -1
- package/dist/reporting.js.map +0 -1
- package/dist/rollout/index.js.map +0 -1
- package/dist/run-campaign-ISHFZ7FJ.js +0 -17
- package/dist/run-campaign-ISHFZ7FJ.js.map +0 -1
- package/dist/supervisor-run/index.js.map +0 -1
- package/dist/traces.js.map +0 -1
- package/dist/wire/index.js.map +0 -1
|
@@ -0,0 +1,2244 @@
|
|
|
1
|
+
import { t as DefaultVerdict } from "./verdict-Dps8_okt.js";
|
|
2
|
+
import { c as ValidationError, t as AgentEvalError } from "./errors-CEk209JS.js";
|
|
3
|
+
import { a as RunRecord, s as RunSplitTag } from "./run-record-CnZu_gjl.js";
|
|
4
|
+
import { l as RawProviderSink } from "./raw-provider-sink-BU29Sh8h.js";
|
|
5
|
+
import { c as CostLedgerHandle, p as CostReceipt } from "./cost-ledger-Dye6jCgg.js";
|
|
6
|
+
import { A as ChatClient } from "./types-DGsxbAEd.js";
|
|
7
|
+
import { c as EProcessState, d as PairedBootstrapOptions, f as PairedBootstrapResult, h as RiskDifferenceResult, u as McNemarResult } from "./statistics-Cmj6nynr.js";
|
|
8
|
+
import { A as LabeledScenarioWrite, D as LabeledScenarioSampleArgs, E as LabeledScenarioRecord, O as LabeledScenarioSource, R as Scenario, S as JudgeConfig, T as LabelTrust, a as CampaignResult, b as GenerationRecord, d as DispatchContext, j as MutableSurface, k as LabeledScenarioStore, l as CodeSurface, p as Gate, u as ComponentSurface, w as JudgeScore } from "./types-k9tZGKUg.js";
|
|
9
|
+
import { rn as PlanCampaignRunOptions, sn as CampaignStorage, tn as CampaignRunPlan } from "./skillopt-optimization-method-B7wX7XkF.js";
|
|
10
|
+
import { n as AnalyzeTracesOptions, r as AnalyzeTracesResult, t as AnalyzeTracesInput } from "./analyst-BkTS3C58.js";
|
|
11
|
+
import { o as LedgerHash } from "./index-BAvgST_9.js";
|
|
12
|
+
import { AgentProfile, AgentProfile as AgentProfile$1, HarnessType, HarnessType as HarnessType$1 } from "@tangle-network/agent-interface";
|
|
13
|
+
//#region src/campaign/analyst-surface.d.ts
|
|
14
|
+
/**
|
|
15
|
+
* A labeled trace scenario: a FIXED trace corpus plus the failure modes a
|
|
16
|
+
* competent analyst MUST surface from it. The labels are ground truth — the
|
|
17
|
+
* objective failures that actually occurred — which is what makes optimizing
|
|
18
|
+
* the analyst prompt against them meaningful rather than circular.
|
|
19
|
+
*/
|
|
20
|
+
interface AnalystScenario extends Scenario {
|
|
21
|
+
kind: 'analyst-surface';
|
|
22
|
+
/** OTLP-JSONL path or an in-memory store of the traces to analyze. */
|
|
23
|
+
source: AnalyzeTracesOptions['source'];
|
|
24
|
+
/** The domain question handed to the analyst (framing lives here, not in
|
|
25
|
+
* the surface under optimization). */
|
|
26
|
+
question: string;
|
|
27
|
+
/**
|
|
28
|
+
* Ground-truth failure modes a good analyst must identify. A finding "hits"
|
|
29
|
+
* a mode when it contains ANY of the mode's case-insensitive cues. Derive
|
|
30
|
+
* these from objective signal (failed task + which step broke), never from
|
|
31
|
+
* the analyst's own prior output.
|
|
32
|
+
*/
|
|
33
|
+
expectedFailureModes: Array<{
|
|
34
|
+
id: string;
|
|
35
|
+
cues: string[];
|
|
36
|
+
}>;
|
|
37
|
+
/**
|
|
38
|
+
* Cues that mark a finding as HALLUCINATED / out-of-scope for this corpus —
|
|
39
|
+
* naming a tool, error, or failure that did not occur. Presence penalizes
|
|
40
|
+
* precision. Optional; omit to score recall only.
|
|
41
|
+
*/
|
|
42
|
+
forbiddenCues?: string[];
|
|
43
|
+
}
|
|
44
|
+
/** The analyst's output for one scenario — the artifact the judge scores. */
|
|
45
|
+
interface AnalystArtifact {
|
|
46
|
+
answer: string;
|
|
47
|
+
findings: string[];
|
|
48
|
+
/** The hardcoded-prompt version the analyst reported (provenance only; the
|
|
49
|
+
* optimized surface overrides the actual prompt text used). */
|
|
50
|
+
actorPromptVersion: string;
|
|
51
|
+
}
|
|
52
|
+
interface BuildAnalystSurfaceDispatchOptions {
|
|
53
|
+
/**
|
|
54
|
+
* Everything `analyzeTraces` needs EXCEPT `actorDescription` (supplied by the
|
|
55
|
+
* surface under optimization) and `source` (supplied by the scenario). `ai`
|
|
56
|
+
* (the AxAIService) is required for a live run.
|
|
57
|
+
*/
|
|
58
|
+
analystOptions: Omit<AnalyzeTracesOptions, 'actorDescription' | 'source'>;
|
|
59
|
+
/** Test seam: defaults to the real `analyzeTraces`. */
|
|
60
|
+
analyze?: (input: AnalyzeTracesInput, options: AnalyzeTracesOptions) => Promise<AnalyzeTracesResult>;
|
|
61
|
+
}
|
|
62
|
+
/**
|
|
63
|
+
* Build the `dispatchWithSurface(surface, scenario, ctx)` the improvement loop
|
|
64
|
+
* calls: run the analyst with `surface` as its actorDescription over the
|
|
65
|
+
* scenario's trace corpus and return its findings.
|
|
66
|
+
*/
|
|
67
|
+
declare function buildAnalystSurfaceDispatch(opts: BuildAnalystSurfaceDispatchOptions): (surface: MutableSurface, scenario: AnalystScenario, ctx: DispatchContext) => Promise<AnalystArtifact>;
|
|
68
|
+
interface FailureModeRecallJudgeOptions {
|
|
69
|
+
/** Weight on recall when precision is also scored (forbiddenCues present).
|
|
70
|
+
* Default 0.5 (equal). Recall-only when no forbiddenCues exist. */
|
|
71
|
+
recallWeight?: number;
|
|
72
|
+
}
|
|
73
|
+
/**
|
|
74
|
+
* Deterministic, ground-truth judge for analyst findings. Composite =
|
|
75
|
+
* recall of the scenario's `expectedFailureModes` (optionally blended with a
|
|
76
|
+
* precision term that penalizes findings tripping `forbiddenCues`). No LLM —
|
|
77
|
+
* the score is a function of the labels, so the analyst prompt is optimized
|
|
78
|
+
* toward surfacing real failures, not toward a judge it can flatter.
|
|
79
|
+
*/
|
|
80
|
+
declare function failureModeRecallJudge(opts?: FailureModeRecallJudgeOptions): JudgeConfig<AnalystArtifact, AnalystScenario>;
|
|
81
|
+
//#endregion
|
|
82
|
+
//#region src/paired-arms.d.ts
|
|
83
|
+
/** One arm observation of one work item. Structural on purpose: callers
|
|
84
|
+
* project their own record type (e.g. a `RunRecord`) into this shape. */
|
|
85
|
+
interface PairedArmRow {
|
|
86
|
+
/** Matching key — rows sharing a `pairKey` across both arms form pairs
|
|
87
|
+
* (typically the task/scenario/seed identity). */
|
|
88
|
+
pairKey: string;
|
|
89
|
+
/** Rep identity within a `pairKey` (e.g. a seed or rep number). Required on
|
|
90
|
+
* every row of a `pairKey` that has more than one rep in either arm; reps
|
|
91
|
+
* then pair only on exact (`pairKey`, `repKey`) match, never on outcome
|
|
92
|
+
* content. Optional when each arm has at most one rep of the item. */
|
|
93
|
+
repKey?: string;
|
|
94
|
+
/** Arm label this row was produced under. */
|
|
95
|
+
arm: string;
|
|
96
|
+
/** Binary outcome; omit when the comparison has no pass/fail notion. */
|
|
97
|
+
pass?: boolean;
|
|
98
|
+
/** Named numeric measurements (score, cost, latency, …). */
|
|
99
|
+
metrics?: Record<string, number>;
|
|
100
|
+
}
|
|
101
|
+
interface PairArmsOptions {
|
|
102
|
+
/** Arm treated as the control side of every pair. */
|
|
103
|
+
baselineArm: string;
|
|
104
|
+
/** Arm treated as the treatment side of every pair. */
|
|
105
|
+
treatmentArm: string;
|
|
106
|
+
}
|
|
107
|
+
/** One matched (baseline, treatment) observation of the same work item. */
|
|
108
|
+
interface MatchedPair {
|
|
109
|
+
pairKey: string;
|
|
110
|
+
/** 0-based position of this pair within its `pairKey`, ordered by sorted
|
|
111
|
+
* `repKey` (always 0 for a single-rep item). The rep identity itself is on
|
|
112
|
+
* the rows (`baseline.repKey` / `treatment.repKey`). */
|
|
113
|
+
repIndex: number;
|
|
114
|
+
baseline: PairedArmRow;
|
|
115
|
+
treatment: PairedArmRow;
|
|
116
|
+
}
|
|
117
|
+
interface PairArmsResult {
|
|
118
|
+
/** Matched pairs, ordered by (`pairKey`, `repIndex`). */
|
|
119
|
+
pairs: MatchedPair[];
|
|
120
|
+
/** Baseline rows left without a treatment counterpart — reported, never
|
|
121
|
+
* silently dropped. */
|
|
122
|
+
unpairedBaseline: PairedArmRow[];
|
|
123
|
+
/** Treatment rows left without a baseline counterpart. */
|
|
124
|
+
unpairedTreatment: PairedArmRow[];
|
|
125
|
+
}
|
|
126
|
+
/**
|
|
127
|
+
* Match rows across two arms into (baseline, treatment) pairs by `pairKey`.
|
|
128
|
+
*
|
|
129
|
+
* A `pairKey` with at most one row per arm pairs directly, no `repKey`
|
|
130
|
+
* needed. A `pairKey` with multiple reps in either arm requires `repKey` on
|
|
131
|
+
* every one of its rows, and reps pair only on exact (`pairKey`, `repKey`)
|
|
132
|
+
* match — pairing is keyed purely on row identity, never on outcome content
|
|
133
|
+
* (outcome-keyed matching deflates discordant counts and biases McNemar), and
|
|
134
|
+
* is therefore independent of input order. Reps whose `repKey` has no
|
|
135
|
+
* counterpart in the other arm, and items present in only one arm, land in
|
|
136
|
+
* the unpaired lists — reported, never truncated.
|
|
137
|
+
*
|
|
138
|
+
* Fail-loud: throws when either named arm has zero rows (an unknown arm
|
|
139
|
+
* name would otherwise read as "everything unpaired"), when the two arm
|
|
140
|
+
* names are equal, when a multi-rep `pairKey` has a row without `repKey`, or
|
|
141
|
+
* when a (`pairKey`, arm) group repeats a `repKey` (the match would be
|
|
142
|
+
* ambiguous).
|
|
143
|
+
*/
|
|
144
|
+
declare function pairArms(rows: readonly PairedArmRow[], opts: PairArmsOptions): PairArmsResult;
|
|
145
|
+
/** Paired pass/fail comparison over the pairs where BOTH sides carry `pass`. */
|
|
146
|
+
interface PairedCorrectness {
|
|
147
|
+
/** Discordant pairs where the treatment passed and the baseline failed. */
|
|
148
|
+
b10: number;
|
|
149
|
+
/** Discordant pairs where the baseline passed and the treatment failed. */
|
|
150
|
+
b01: number;
|
|
151
|
+
/** Exact McNemar significance over the paired outcomes (`b === b10`, `c === b01`). */
|
|
152
|
+
mcnemar: McNemarResult;
|
|
153
|
+
/** Paired effect size: p(treatment) − p(baseline) with a paired-variance CI. */
|
|
154
|
+
riskDifference: RiskDifferenceResult;
|
|
155
|
+
}
|
|
156
|
+
/** Paired delta summary for one named metric (delta = treatment − baseline). */
|
|
157
|
+
interface PairedMetricDelta {
|
|
158
|
+
name: string;
|
|
159
|
+
/** Pairs where BOTH sides carry a finite value for this metric. */
|
|
160
|
+
n: number;
|
|
161
|
+
/** Pairs where at least one side does not carry the metric. */
|
|
162
|
+
nMissing: number;
|
|
163
|
+
/** Median paired delta, or null when `n === 0`. */
|
|
164
|
+
medianDelta: number | null;
|
|
165
|
+
/** Mean paired delta, or null when `n === 0`. */
|
|
166
|
+
meanDelta: number | null;
|
|
167
|
+
/** Bootstrap CI on the paired delta (`pairedBootstrap`); null when
|
|
168
|
+
* `n === 0` — a zero-width [0, 0] interval on no data would read as a
|
|
169
|
+
* measured tight null. */
|
|
170
|
+
bootstrapCi: PairedBootstrapResult | null;
|
|
171
|
+
/** Wilcoxon signed-rank test on the paired deltas; null when `n === 0`. */
|
|
172
|
+
wilcoxon: {
|
|
173
|
+
w: number;
|
|
174
|
+
p: number;
|
|
175
|
+
} | null;
|
|
176
|
+
}
|
|
177
|
+
interface ComparePairedArmsOptions extends PairArmsOptions {
|
|
178
|
+
/** Metrics to compare. Default: every metric name observed on any matched
|
|
179
|
+
* pair, sorted. A name that appears on no pair is still reported (with
|
|
180
|
+
* `n = 0`) so a misspelled metric is visible instead of vanishing. */
|
|
181
|
+
metricNames?: string[];
|
|
182
|
+
/** Passed through to `pairedBootstrap` — set `seed` for reproducible CIs. */
|
|
183
|
+
bootstrap?: PairedBootstrapOptions;
|
|
184
|
+
}
|
|
185
|
+
interface PairedArmsComparison {
|
|
186
|
+
nPairs: number;
|
|
187
|
+
nUnpairedBaseline: number;
|
|
188
|
+
nUnpairedTreatment: number;
|
|
189
|
+
/** null when no matched pair carries `pass` on both sides — a pass/fail
|
|
190
|
+
* verdict over rows that never measured pass/fail would be fabricated. */
|
|
191
|
+
correctness: PairedCorrectness | null;
|
|
192
|
+
metricDeltas: PairedMetricDelta[];
|
|
193
|
+
}
|
|
194
|
+
/**
|
|
195
|
+
* Full matched-pair arm comparison: pair via {@link pairArms}, then compose
|
|
196
|
+
* the paired estimators from `statistics` over the matched pairs.
|
|
197
|
+
*
|
|
198
|
+
* Correctness uses only the pairs where both sides carry `pass` (`mcnemar.n`
|
|
199
|
+
* is that subset's size); each metric uses only the pairs where both sides
|
|
200
|
+
* carry a finite value for it, with the remainder counted in `nMissing`.
|
|
201
|
+
* Deltas are treatment − baseline throughout.
|
|
202
|
+
*
|
|
203
|
+
* Fail-loud: inherits {@link pairArms}'s unknown-arm throw, and throws on a
|
|
204
|
+
* non-finite metric value — silently treating corrupt telemetry as "metric
|
|
205
|
+
* absent" would misreport it as missing coverage.
|
|
206
|
+
*/
|
|
207
|
+
declare function comparePairedArms(rows: readonly PairedArmRow[], opts: ComparePairedArmsOptions): PairedArmsComparison;
|
|
208
|
+
interface MatchedRunRecordPair {
|
|
209
|
+
pairKey: string;
|
|
210
|
+
repKey: string;
|
|
211
|
+
baseline: RunRecord;
|
|
212
|
+
treatment: RunRecord;
|
|
213
|
+
}
|
|
214
|
+
interface PairRunRecordsResult {
|
|
215
|
+
pairs: MatchedRunRecordPair[];
|
|
216
|
+
unpairedBaseline: RunRecord[];
|
|
217
|
+
unpairedTreatment: RunRecord[];
|
|
218
|
+
}
|
|
219
|
+
/**
|
|
220
|
+
* Pair two RunRecord arms by the identity of the evaluated work:
|
|
221
|
+
* `(experimentId, scenarioId, seed)`.
|
|
222
|
+
*
|
|
223
|
+
* Falling back to array order, candidate id, or experiment id can compare
|
|
224
|
+
* different tasks and fabricate lift. Duplicate identities throw.
|
|
225
|
+
*/
|
|
226
|
+
declare function pairRunRecords(baselineRuns: readonly RunRecord[], treatmentRuns: readonly RunRecord[]): PairRunRecordsResult;
|
|
227
|
+
//#endregion
|
|
228
|
+
//#region src/campaign/cross-surface-types.d.ts
|
|
229
|
+
/** Whether one candidate attempt produced a usable executable outcome. */
|
|
230
|
+
type CrossSurfaceAttemptCompleteness = 'complete' | 'missing' | 'invalid';
|
|
231
|
+
/** One independently proposed change on one caller-defined surface. */
|
|
232
|
+
interface CrossSurfaceComponent {
|
|
233
|
+
componentId: string;
|
|
234
|
+
surfaceId: string;
|
|
235
|
+
/** Explicitly controls whether this component may anchor the best-single arm. */
|
|
236
|
+
bestSingleEligible: boolean;
|
|
237
|
+
}
|
|
238
|
+
/** Immutable identity for a single candidate or a materialized composition. */
|
|
239
|
+
interface CrossSurfaceCandidate {
|
|
240
|
+
candidateId: string;
|
|
241
|
+
componentIds: string[];
|
|
242
|
+
contentHash: string;
|
|
243
|
+
artifactBytes: number;
|
|
244
|
+
}
|
|
245
|
+
/** Per-component trace evidence captured during one task attempt. */
|
|
246
|
+
interface CrossSurfaceComponentEvidence {
|
|
247
|
+
componentId: string;
|
|
248
|
+
/** null means the trace could not establish whether the component fired. */
|
|
249
|
+
fired: boolean | null;
|
|
250
|
+
/** null means the trace could not establish whether the component changed behavior. */
|
|
251
|
+
effectObserved: boolean | null;
|
|
252
|
+
}
|
|
253
|
+
/**
|
|
254
|
+
* Canonical per-task input row. Consumers may extend this interface with
|
|
255
|
+
* receipt, trace, retry, or failure details; the report preserves the original
|
|
256
|
+
* row object rather than projecting those details away.
|
|
257
|
+
*/
|
|
258
|
+
interface CrossSurfaceTaskRow {
|
|
259
|
+
taskId: string;
|
|
260
|
+
candidateId: string;
|
|
261
|
+
/** Repeated here so every persisted row remains self-describing. */
|
|
262
|
+
componentIds: string[];
|
|
263
|
+
completeness: CrossSurfaceAttemptCompleteness;
|
|
264
|
+
pass: boolean | null;
|
|
265
|
+
score: number | null;
|
|
266
|
+
/**
|
|
267
|
+
* Per-attempt deployment measurements. Every declared metric must have a
|
|
268
|
+
* known, non-negative value. Proposal, analysis, and selection spend belongs
|
|
269
|
+
* in the search ledger rather than being spread across task cells.
|
|
270
|
+
*/
|
|
271
|
+
cost: Record<string, number | null>;
|
|
272
|
+
componentEvidence: CrossSurfaceComponentEvidence[];
|
|
273
|
+
/** Required for missing or invalid attempts; forbidden for complete attempts. */
|
|
274
|
+
rejectReason: string | null;
|
|
275
|
+
}
|
|
276
|
+
interface CrossSurfaceBootstrapPolicy {
|
|
277
|
+
seed: number;
|
|
278
|
+
resamples: number;
|
|
279
|
+
confidence: number;
|
|
280
|
+
}
|
|
281
|
+
/** Predeclared candidate eligibility and composition policy. */
|
|
282
|
+
interface CrossSurfaceSelectionPolicy {
|
|
283
|
+
minimumFiringTasks: number;
|
|
284
|
+
minimumEffectTasks: number;
|
|
285
|
+
requireObservedFiring: boolean;
|
|
286
|
+
requireObservedEffect: boolean;
|
|
287
|
+
/** Only named metrics are constrained; all declared metrics are still reported. */
|
|
288
|
+
maximumMedianCostRatioToBaseline: Record<string, number>;
|
|
289
|
+
/** A smaller terminal bundle is reported but cannot become the selected arm. */
|
|
290
|
+
minimumBundleComponents: number;
|
|
291
|
+
}
|
|
292
|
+
interface AnalyzeCrossSurfaceInteractionsInput<TRow extends CrossSurfaceTaskRow = CrossSurfaceTaskRow> {
|
|
293
|
+
components: readonly CrossSurfaceComponent[];
|
|
294
|
+
candidates: readonly CrossSurfaceCandidate[];
|
|
295
|
+
rows: readonly TRow[];
|
|
296
|
+
baselineCandidateId: string;
|
|
297
|
+
/** Exact shared task axis and its canonical output order. */
|
|
298
|
+
taskOrder: readonly string[];
|
|
299
|
+
/** Canonical materialization order for component sets and the naive stack. */
|
|
300
|
+
componentOrder: readonly string[];
|
|
301
|
+
/** Final deterministic tie-break; lower index wins. */
|
|
302
|
+
candidateOrder: readonly string[];
|
|
303
|
+
/** Declares every cost key and the order used for cost tie-breaks. */
|
|
304
|
+
costMetricOrder: readonly string[];
|
|
305
|
+
bootstrap: CrossSurfaceBootstrapPolicy;
|
|
306
|
+
selection: CrossSurfaceSelectionPolicy;
|
|
307
|
+
}
|
|
308
|
+
interface CrossSurfaceDistribution {
|
|
309
|
+
n: number;
|
|
310
|
+
min: number;
|
|
311
|
+
median: number;
|
|
312
|
+
mean: number;
|
|
313
|
+
max: number;
|
|
314
|
+
total: number;
|
|
315
|
+
}
|
|
316
|
+
interface CrossSurfaceEvidenceBreakdown {
|
|
317
|
+
componentId: string;
|
|
318
|
+
observedTaskIds: string[];
|
|
319
|
+
notObservedTaskIds: string[];
|
|
320
|
+
unobservedTaskIds: string[];
|
|
321
|
+
}
|
|
322
|
+
interface CrossSurfaceCandidateEvidence {
|
|
323
|
+
byComponent: CrossSurfaceEvidenceBreakdown[];
|
|
324
|
+
allObservedTaskIds: string[];
|
|
325
|
+
someObservedTaskIds: string[];
|
|
326
|
+
noneObservedTaskIds: string[];
|
|
327
|
+
unobservedTaskIds: string[];
|
|
328
|
+
}
|
|
329
|
+
type CrossSurfaceIneligibilityReason = 'missing_attempt' | 'invalid_attempt' | 'baseline_outcome_missing' | 'benefit_not_greater_than_regression' | 'firing_below_minimum' | 'firing_unobserved' | 'effect_below_minimum' | 'effect_unobserved' | 'cost_limit_exceeded';
|
|
330
|
+
interface CrossSurfaceEligibility {
|
|
331
|
+
eligible: boolean;
|
|
332
|
+
reasons: CrossSurfaceIneligibilityReason[];
|
|
333
|
+
}
|
|
334
|
+
interface CrossSurfaceCandidateOutcome {
|
|
335
|
+
resolvedTaskIds: string[];
|
|
336
|
+
failedTaskIds: string[];
|
|
337
|
+
missingTaskIds: string[];
|
|
338
|
+
invalidTaskIds: string[];
|
|
339
|
+
benefitTaskIds: string[];
|
|
340
|
+
regressionTaskIds: string[];
|
|
341
|
+
comparisonMissingTaskIds: string[];
|
|
342
|
+
netBenefit: number;
|
|
343
|
+
}
|
|
344
|
+
interface CrossSurfaceCandidateSummary {
|
|
345
|
+
candidate: CrossSurfaceCandidate;
|
|
346
|
+
outcome: CrossSurfaceCandidateOutcome;
|
|
347
|
+
score: CrossSurfaceDistribution | null;
|
|
348
|
+
costs: Record<string, CrossSurfaceDistribution>;
|
|
349
|
+
firing: CrossSurfaceCandidateEvidence;
|
|
350
|
+
effect: CrossSurfaceCandidateEvidence;
|
|
351
|
+
/** Reuses the package's paired McNemar/risk-difference/bootstrap statistics. */
|
|
352
|
+
comparisonToBaseline: PairedArmsComparison | null;
|
|
353
|
+
/** null only for the fixed baseline. */
|
|
354
|
+
eligibility: CrossSurfaceEligibility | null;
|
|
355
|
+
}
|
|
356
|
+
interface CrossSurfaceRelativeCost {
|
|
357
|
+
treatmentMedian: number;
|
|
358
|
+
comparatorMedian: number;
|
|
359
|
+
medianDelta: number;
|
|
360
|
+
/** null when the comparator median is zero but the treatment median is not. */
|
|
361
|
+
medianRatio: number | null;
|
|
362
|
+
}
|
|
363
|
+
interface CrossSurfaceCandidateComparison {
|
|
364
|
+
comparatorCandidateId: string;
|
|
365
|
+
treatmentCandidateId: string;
|
|
366
|
+
winsTaskIds: string[];
|
|
367
|
+
regressionTaskIds: string[];
|
|
368
|
+
missingTaskIds: string[];
|
|
369
|
+
paired: PairedArmsComparison;
|
|
370
|
+
relativeCost: Record<string, CrossSurfaceRelativeCost>;
|
|
371
|
+
}
|
|
372
|
+
interface CrossSurfacePairEvidence {
|
|
373
|
+
bothTaskIds: string[];
|
|
374
|
+
leftOnlyTaskIds: string[];
|
|
375
|
+
rightOnlyTaskIds: string[];
|
|
376
|
+
neitherTaskIds: string[];
|
|
377
|
+
unobservedTaskIds: string[];
|
|
378
|
+
}
|
|
379
|
+
interface CrossSurfaceInteractionTask {
|
|
380
|
+
taskId: string;
|
|
381
|
+
/** Composition minus the additive expectation from the baseline and singles. */
|
|
382
|
+
passInteraction: number | null;
|
|
383
|
+
scoreInteraction: number | null;
|
|
384
|
+
}
|
|
385
|
+
interface CrossSurfaceInteractionEffect {
|
|
386
|
+
perTask: CrossSurfaceInteractionTask[];
|
|
387
|
+
n: number;
|
|
388
|
+
nMissing: number;
|
|
389
|
+
meanPassInteraction: number | null;
|
|
390
|
+
meanScoreInteraction: number | null;
|
|
391
|
+
passBootstrap: PairedBootstrapResult | null;
|
|
392
|
+
scoreBootstrap: PairedBootstrapResult | null;
|
|
393
|
+
}
|
|
394
|
+
type CrossSurfacePairIncompatibilityReason = 'constituent_not_ready' | 'pair_incomplete' | 'baseline_regression' | 'interference' | 'no_incremental_resolution' | 'firing_below_minimum' | 'firing_unobserved' | 'effect_below_minimum' | 'effect_unobserved' | 'cost_limit_exceeded';
|
|
395
|
+
interface CrossSurfacePairCompatibility {
|
|
396
|
+
compatible: boolean;
|
|
397
|
+
reasons: CrossSurfacePairIncompatibilityReason[];
|
|
398
|
+
betterSingleCandidateId: string;
|
|
399
|
+
}
|
|
400
|
+
interface CrossSurfacePairwiseEntry {
|
|
401
|
+
componentIds: [string, string];
|
|
402
|
+
singleCandidateIds: [string, string];
|
|
403
|
+
compositionCandidateId: string;
|
|
404
|
+
benefitTaskIds: string[];
|
|
405
|
+
regressionTaskIds: string[];
|
|
406
|
+
synergyTaskIds: string[];
|
|
407
|
+
interferenceTaskIds: string[];
|
|
408
|
+
incrementalVsConstituents: [CrossSurfaceCandidateComparison, CrossSurfaceCandidateComparison];
|
|
409
|
+
relativeCostToBaseline: Record<string, CrossSurfaceRelativeCost>;
|
|
410
|
+
firing: CrossSurfacePairEvidence;
|
|
411
|
+
effect: CrossSurfacePairEvidence;
|
|
412
|
+
interaction: CrossSurfaceInteractionEffect;
|
|
413
|
+
compatibility: CrossSurfacePairCompatibility;
|
|
414
|
+
}
|
|
415
|
+
interface CrossSurfaceRankedSingle {
|
|
416
|
+
rank: number;
|
|
417
|
+
candidateId: string;
|
|
418
|
+
componentId: string;
|
|
419
|
+
}
|
|
420
|
+
interface CrossSurfaceBestSingleSelection {
|
|
421
|
+
candidateId: string;
|
|
422
|
+
componentId: string;
|
|
423
|
+
ranking: CrossSurfaceRankedSingle[];
|
|
424
|
+
}
|
|
425
|
+
interface CrossSurfaceNaiveStackSelection {
|
|
426
|
+
/** Every individually eligible single, stacked in canonical component order. */
|
|
427
|
+
candidateId: string;
|
|
428
|
+
componentIds: string[];
|
|
429
|
+
}
|
|
430
|
+
type CrossSurfaceAdditionRejectionReason = 'pair_incompatible' | 'full_bundle_not_evaluated' | 'bundle_incomplete' | 'baseline_regression' | 'no_incremental_resolution' | 'incremental_regression' | 'firing_below_minimum' | 'firing_unobserved' | 'effect_below_minimum' | 'effect_unobserved' | 'cost_limit_exceeded';
|
|
431
|
+
interface CrossSurfaceAdditionDecision {
|
|
432
|
+
additionCandidateId: string;
|
|
433
|
+
additionComponentId: string;
|
|
434
|
+
bundleCandidateId: string | null;
|
|
435
|
+
incrementalResolutionTaskIds: string[];
|
|
436
|
+
incrementalRegressionTaskIds: string[];
|
|
437
|
+
incrementalMedianCost: Record<string, number> | null;
|
|
438
|
+
eligible: boolean;
|
|
439
|
+
selected: boolean;
|
|
440
|
+
reasons: CrossSurfaceAdditionRejectionReason[];
|
|
441
|
+
}
|
|
442
|
+
interface CrossSurfaceCompositionStep {
|
|
443
|
+
fromCandidateId: string;
|
|
444
|
+
retainedComponentIds: string[];
|
|
445
|
+
considered: CrossSurfaceAdditionDecision[];
|
|
446
|
+
selectedCandidateId: string | null;
|
|
447
|
+
}
|
|
448
|
+
/** One deterministic growth path starting from a compatible two-surface seed. */
|
|
449
|
+
interface CrossSurfaceInteractionPath {
|
|
450
|
+
seedCandidateId: string;
|
|
451
|
+
terminalCandidateId: string;
|
|
452
|
+
terminalComponentIds: string[];
|
|
453
|
+
qualified: boolean;
|
|
454
|
+
steps: CrossSurfaceCompositionStep[];
|
|
455
|
+
}
|
|
456
|
+
interface CrossSurfaceInteractionAwareSelection {
|
|
457
|
+
/** Compatible pair that seeded the selected deterministic growth path. */
|
|
458
|
+
seedCandidateId: string;
|
|
459
|
+
/** Candidate reached by the winning path, even if the minimum size is not met. */
|
|
460
|
+
terminalCandidateId: string;
|
|
461
|
+
terminalComponentIds: string[];
|
|
462
|
+
/** null when no path produced a qualifying multi-component bundle. */
|
|
463
|
+
selectedCandidateId: string | null;
|
|
464
|
+
qualified: boolean;
|
|
465
|
+
/** Every compatible pair seed is retained so seed choice cannot hide an interaction. */
|
|
466
|
+
evaluatedPaths: CrossSurfaceInteractionPath[];
|
|
467
|
+
/** Convenience alias for the winning path's steps. */
|
|
468
|
+
steps: CrossSurfaceCompositionStep[];
|
|
469
|
+
}
|
|
470
|
+
interface CrossSurfaceSelections {
|
|
471
|
+
bestSingle: CrossSurfaceBestSingleSelection | null;
|
|
472
|
+
naiveStack: CrossSurfaceNaiveStackSelection | null;
|
|
473
|
+
interactionAware: CrossSurfaceInteractionAwareSelection | null;
|
|
474
|
+
}
|
|
475
|
+
interface CrossSurfaceInteractionReport<TRow extends CrossSurfaceTaskRow = CrossSurfaceTaskRow> {
|
|
476
|
+
taskIds: string[];
|
|
477
|
+
componentIds: string[];
|
|
478
|
+
candidateIds: string[];
|
|
479
|
+
costMetrics: string[];
|
|
480
|
+
/** Canonical candidate × task order; no input row is dropped. */
|
|
481
|
+
rows: TRow[];
|
|
482
|
+
missingAttempts: TRow[];
|
|
483
|
+
invalidAttempts: TRow[];
|
|
484
|
+
candidates: CrossSurfaceCandidateSummary[];
|
|
485
|
+
pairwise: CrossSurfacePairwiseEntry[];
|
|
486
|
+
selections: CrossSurfaceSelections;
|
|
487
|
+
}
|
|
488
|
+
//#endregion
|
|
489
|
+
//#region src/campaign/cross-surface-interaction.d.ts
|
|
490
|
+
/**
|
|
491
|
+
* Build the complete cross-surface evidence matrix and derive all three frozen
|
|
492
|
+
* candidates. The task/candidate/component orders are part of the input so
|
|
493
|
+
* neither insertion order nor an after-the-fact tie-break can change a result.
|
|
494
|
+
*/
|
|
495
|
+
declare function analyzeCrossSurfaceInteractions<TRow extends CrossSurfaceTaskRow>(input: AnalyzeCrossSurfaceInteractionsInput<TRow>): CrossSurfaceInteractionReport<TRow>;
|
|
496
|
+
//#endregion
|
|
497
|
+
//#region src/campaign/fixtures.d.ts
|
|
498
|
+
type EvalFixtureValidationMode = 'vitest' | 'none';
|
|
499
|
+
interface EvalFixtureFile {
|
|
500
|
+
path: string;
|
|
501
|
+
sha256: string;
|
|
502
|
+
bytes: number;
|
|
503
|
+
}
|
|
504
|
+
interface EvalFixture {
|
|
505
|
+
name: string;
|
|
506
|
+
path: string;
|
|
507
|
+
promptPath: string;
|
|
508
|
+
evalPath?: string;
|
|
509
|
+
packageJsonPath?: string;
|
|
510
|
+
prompt: string;
|
|
511
|
+
files: EvalFixtureFile[];
|
|
512
|
+
fingerprint: string;
|
|
513
|
+
}
|
|
514
|
+
interface EvalFixtureScenario extends Scenario {
|
|
515
|
+
kind: 'eval-fixture';
|
|
516
|
+
fixtureName: string;
|
|
517
|
+
fixturePath: string;
|
|
518
|
+
promptPath: string;
|
|
519
|
+
evalPath?: string;
|
|
520
|
+
packageJsonPath?: string;
|
|
521
|
+
prompt: string;
|
|
522
|
+
fingerprint: string;
|
|
523
|
+
}
|
|
524
|
+
interface EvalFixtureLoadOptions {
|
|
525
|
+
/** `vitest` requires EVAL.ts/EVAL.tsx and package.json type=module. `none` only requires PROMPT.md. */
|
|
526
|
+
validation?: EvalFixtureValidationMode;
|
|
527
|
+
/** Extra caller-owned knobs that affect fixture behavior, folded into the fingerprint. */
|
|
528
|
+
fingerprintConfig?: unknown;
|
|
529
|
+
}
|
|
530
|
+
interface LoadEvalFixtureScenariosOptions extends EvalFixtureLoadOptions {
|
|
531
|
+
names?: string[];
|
|
532
|
+
}
|
|
533
|
+
interface PlanEvalFixtureRunOptions<TArtifact = unknown> extends Pick<PlanCampaignRunOptions<EvalFixtureScenario, TArtifact>, 'dispatchRef' | 'judges' | 'seed' | 'reps' | 'resumable' | 'runDir'> {
|
|
534
|
+
evalsDir: string;
|
|
535
|
+
validation?: EvalFixtureValidationMode;
|
|
536
|
+
fingerprintConfig?: unknown;
|
|
537
|
+
names?: string[];
|
|
538
|
+
storage?: CampaignStorage;
|
|
539
|
+
}
|
|
540
|
+
type EvalFixtureRunPlan = CampaignRunPlan & {
|
|
541
|
+
fixtures: Array<Pick<EvalFixtureScenario, 'fixtureName' | 'fixturePath' | 'fingerprint'>>;
|
|
542
|
+
};
|
|
543
|
+
/** Walk `evalsDir` and return the relative name of every fixture directory (one containing an exact-case `PROMPT.md`). */
|
|
544
|
+
declare function discoverEvalFixtures(evalsDir: string): string[];
|
|
545
|
+
/**
|
|
546
|
+
* Load ONE fixture by name: reads `PROMPT.md` (plus `EVAL.ts`/`EVAL.tsx` and `package.json` under
|
|
547
|
+
* `vitest` validation) and content-fingerprints the full file set for cache identity.
|
|
548
|
+
*/
|
|
549
|
+
declare function loadEvalFixture(evalsDir: string, name: string, options?: EvalFixtureLoadOptions): EvalFixture;
|
|
550
|
+
/** Load fixtures (all discovered, or just `names`) as campaign `Scenario`s tagged `eval-fixture`. */
|
|
551
|
+
declare function loadEvalFixtureScenarios(evalsDir: string, options?: LoadEvalFixtureScenariosOptions): EvalFixtureScenario[];
|
|
552
|
+
/**
|
|
553
|
+
* Dry-run planner for a fixture campaign: loads the scenarios, delegates to `planCampaignRun`,
|
|
554
|
+
* and returns the plan plus each fixture's name/path/fingerprint.
|
|
555
|
+
*/
|
|
556
|
+
declare function planEvalFixtureRun<TArtifact = unknown>(options: PlanEvalFixtureRunOptions<TArtifact>): EvalFixtureRunPlan;
|
|
557
|
+
//#endregion
|
|
558
|
+
//#region src/campaign/gates/neutralization-gate.d.ts
|
|
559
|
+
interface NeutralizationGateOptions<TScenario extends Scenario = Scenario> {
|
|
560
|
+
scenarios: TScenario[];
|
|
561
|
+
/** Reject when the neutralized (content-blanked, footprint-matched) variant
|
|
562
|
+
* reproduces at least this fraction of the candidate's held-out lift. Default
|
|
563
|
+
* 0.5 — if blanking the content keeps half the lift, the content is decorative.
|
|
564
|
+
* Equality rejects: a neutralized lift == threshold·candidateLift is decorative. */
|
|
565
|
+
maxDecorativeFraction?: number;
|
|
566
|
+
}
|
|
567
|
+
/**
|
|
568
|
+
* Composable placebo gate: ships only when the candidate's held-out lift is NOT
|
|
569
|
+
* mostly reproduced by a footprint-matched neutralized variant.
|
|
570
|
+
*/
|
|
571
|
+
declare function neutralizationGate<TArtifact, TScenario extends Scenario>(options: NeutralizationGateOptions<TScenario>): Gate<TArtifact, TScenario>;
|
|
572
|
+
//#endregion
|
|
573
|
+
//#region src/pre-registration.d.ts
|
|
574
|
+
/**
|
|
575
|
+
* Pre-registered hypotheses — declare what you're testing BEFORE the
|
|
576
|
+
* run, check it AFTER. Prevents p-hacking, optional stopping, and the
|
|
577
|
+
* "we ran until it looked good" failure mode.
|
|
578
|
+
*
|
|
579
|
+
* Manifest is a plain JSON-friendly object. Sign it with a content hash
|
|
580
|
+
* + timestamp; the registered record becomes immutable. Post-run,
|
|
581
|
+
* evaluate the manifest against observed results — the library refuses
|
|
582
|
+
* to let you re-interpret a different metric as the declared one.
|
|
583
|
+
*/
|
|
584
|
+
interface HypothesisManifest {
|
|
585
|
+
id: string;
|
|
586
|
+
/** Human prose — goes into the audit trail. */
|
|
587
|
+
hypothesis: string;
|
|
588
|
+
/** Metric the hypothesis claims to move. */
|
|
589
|
+
metric: string;
|
|
590
|
+
/** 'increase' = candidate should score higher than baseline; 'decrease' = lower. */
|
|
591
|
+
direction: 'increase' | 'decrease';
|
|
592
|
+
/** Minimum effect size to count (same units as the metric). */
|
|
593
|
+
minEffect: number;
|
|
594
|
+
/** Alpha threshold. */
|
|
595
|
+
alpha: number;
|
|
596
|
+
/** Target statistical power at which sample size was pre-computed. */
|
|
597
|
+
power: number;
|
|
598
|
+
/** Declared N per arm before running. */
|
|
599
|
+
preRegisteredN: number;
|
|
600
|
+
/** ISO8601 timestamp the manifest was registered. */
|
|
601
|
+
registeredAt: string;
|
|
602
|
+
/** Optional identifiers to tie into the trace corpus. */
|
|
603
|
+
baselineLabel?: string;
|
|
604
|
+
candidateLabel?: string;
|
|
605
|
+
}
|
|
606
|
+
/**
|
|
607
|
+
* Identifier for the hashing scheme used to produce `contentHash`.
|
|
608
|
+
*
|
|
609
|
+
* `'sha256-content'` — sha256 hex over the canonicalized manifest with
|
|
610
|
+
* the `contentHash` and `algo` fields stripped. Held as a string union
|
|
611
|
+
* so future schemes can be added without breaking parsers; SignedManifest
|
|
612
|
+
* values without `algo` deserialize cleanly because the field is optional.
|
|
613
|
+
*/
|
|
614
|
+
type SignedManifestAlgo = 'sha256-content';
|
|
615
|
+
interface SignedManifest extends HypothesisManifest {
|
|
616
|
+
/** sha256 hex of canonicalized manifest (everything except contentHash and algo). */
|
|
617
|
+
contentHash: string;
|
|
618
|
+
/**
|
|
619
|
+
* Algorithm string describing how `contentHash` was produced.
|
|
620
|
+
*
|
|
621
|
+
* Optional on the type so serialized manifests without it still parse,
|
|
622
|
+
* but ALWAYS populated by {@link signManifest}. Consumers that want to
|
|
623
|
+
* enforce a known algorithm should reject manifests where this field
|
|
624
|
+
* is missing or unrecognized.
|
|
625
|
+
*/
|
|
626
|
+
algo?: SignedManifestAlgo;
|
|
627
|
+
}
|
|
628
|
+
interface HypothesisResult {
|
|
629
|
+
manifest: SignedManifest;
|
|
630
|
+
observedN: number;
|
|
631
|
+
observedEffect: number;
|
|
632
|
+
observedPValue: number;
|
|
633
|
+
/** True iff the observed effect hits the pre-declared direction with
|
|
634
|
+
* magnitude ≥ minEffect AND p < alpha. */
|
|
635
|
+
confirmed: boolean;
|
|
636
|
+
/** Enumerated reasons the hypothesis was rejected (each a machine-tag). */
|
|
637
|
+
rejectionReasons: Array<'wrong_direction' | 'effect_too_small' | 'not_significant' | 'undersampled'>;
|
|
638
|
+
notes?: string;
|
|
639
|
+
}
|
|
640
|
+
/**
|
|
641
|
+
* Deterministic JSON canonicalization — sort object keys recursively.
|
|
642
|
+
*
|
|
643
|
+
* Two semantically-equal objects produce byte-identical canonicalized output;
|
|
644
|
+
* this is what makes a content-hash stable across encoders, key insertion
|
|
645
|
+
* orders, and runtime versions. Exported for any consumer that needs the same
|
|
646
|
+
* canonicalization guarantee outside the manifest-signing path (e.g., signing
|
|
647
|
+
* an artifact bundle, hashing a dataset version, etc.).
|
|
648
|
+
*/
|
|
649
|
+
declare function canonicalize(v: unknown): unknown;
|
|
650
|
+
/**
|
|
651
|
+
* SHA-256 hex (full 64 chars) over the canonicalized JSON encoding of `obj`.
|
|
652
|
+
*
|
|
653
|
+
* The same primitive `signManifest` and `verifyManifest` are built on, exposed
|
|
654
|
+
* directly so consumers signing arbitrary structured content (artifact bundles,
|
|
655
|
+
* production packets, dataset manifests, etc.) don't have to re-derive
|
|
656
|
+
* canonicalize+sha256 from scratch.
|
|
657
|
+
*
|
|
658
|
+
* Stable across:
|
|
659
|
+
* - object key insertion order (canonicalization sorts keys recursively)
|
|
660
|
+
* - encoder choice (UTF-8 via TextEncoder, fixed)
|
|
661
|
+
* - runtime (uses the Web Crypto subtle digest, present in Node ≥18 and browsers)
|
|
662
|
+
*
|
|
663
|
+
* Named `hashJson` to disambiguate from `prompt-registry.ts`'s `hashContent`,
|
|
664
|
+
* which takes a string input and returns a truncated 12-char prompt id.
|
|
665
|
+
* Use `hashJson` when you mean "canonicalize then hash."
|
|
666
|
+
*
|
|
667
|
+
* @example
|
|
668
|
+
* const hash = await hashJson({ id: '1', kind: 'spec' })
|
|
669
|
+
* // 'a3f1...' (64 hex chars)
|
|
670
|
+
*/
|
|
671
|
+
declare function hashJson<T>(obj: T): Promise<string>;
|
|
672
|
+
/**
|
|
673
|
+
* Sign a manifest with a SHA-256 content hash.
|
|
674
|
+
*
|
|
675
|
+
* The hash covers the canonicalized manifest with the `contentHash`
|
|
676
|
+
* and `algo` fields stripped; this lets verifiers re-sign the rest and
|
|
677
|
+
* compare. Returned manifest always carries `algo: 'sha256-content'`
|
|
678
|
+
* so downstream consumers can identify the scheme; manifests without
|
|
679
|
+
* `algo` still verify because it is stripped before hashing on both sides.
|
|
680
|
+
*/
|
|
681
|
+
declare function signManifest(m: HypothesisManifest): Promise<SignedManifest>;
|
|
682
|
+
/**
|
|
683
|
+
* Verify that a signed manifest has not been tampered with.
|
|
684
|
+
*
|
|
685
|
+
* Strips `contentHash` and `algo` before re-signing so manifests without
|
|
686
|
+
* `algo` verify identically to ones that carry it.
|
|
687
|
+
*/
|
|
688
|
+
declare function verifyManifest(m: SignedManifest): Promise<boolean>;
|
|
689
|
+
/**
|
|
690
|
+
* Evaluate a pre-registered hypothesis against observed results.
|
|
691
|
+
* Mechanical — no re-interpretation permitted.
|
|
692
|
+
*/
|
|
693
|
+
declare function evaluateHypothesis(manifest: SignedManifest, observed: {
|
|
694
|
+
n: number;
|
|
695
|
+
effect: number;
|
|
696
|
+
pValue: number;
|
|
697
|
+
}): Promise<HypothesisResult>;
|
|
698
|
+
//#endregion
|
|
699
|
+
//#region src/campaign/gates/sequential.d.ts
|
|
700
|
+
type SequentialDecision = 'promote' | 'continue' | 'undecided-at-maxN';
|
|
701
|
+
interface SequentialObservation {
|
|
702
|
+
decision: SequentialDecision;
|
|
703
|
+
/** Current e-value (the betting wealth) against H0. */
|
|
704
|
+
eValue: number;
|
|
705
|
+
/** Paired deltas consumed so far. */
|
|
706
|
+
n: number;
|
|
707
|
+
/** Names the decision basis. For 'undecided-at-maxN' it states explicitly
|
|
708
|
+
* that exhausting the budget is NOT evidence of no effect. */
|
|
709
|
+
reason: string;
|
|
710
|
+
}
|
|
711
|
+
interface SequentialPairedGateOptions {
|
|
712
|
+
/** Type-I budget. With `preRegistration` bound this MUST match
|
|
713
|
+
* `manifest.alpha` (conflict throws). Default 0.05. */
|
|
714
|
+
alpha?: number;
|
|
715
|
+
/** Minimum paired deltas before a promote may fire. The stopping rule is
|
|
716
|
+
* "first n ≥ minN with e-value ≥ 1/alpha" — still a valid stopping time.
|
|
717
|
+
* Default 5. */
|
|
718
|
+
minN?: number;
|
|
719
|
+
/** Pre-registered observation budget. Required unless `preRegistration`
|
|
720
|
+
* supplies it via `preRegisteredN` (conflict throws). */
|
|
721
|
+
maxN?: number;
|
|
722
|
+
/** Bet truncation forwarded to `eProcess`. Default 0.5. */
|
|
723
|
+
maxBet?: number;
|
|
724
|
+
/** Bound on |delta| in the judge's native scale; deltas are mapped to
|
|
725
|
+
* x = (d/scale + 1)/2 ∈ [0,1]. A delta outside ±scale throws (use
|
|
726
|
+
* `detectScale` to pick 1 vs 100 BEFORE streaming). Default 1. */
|
|
727
|
+
scale?: number;
|
|
728
|
+
/** Seed for the data-independent shuffle of paired deltas in `decide(ctx)`
|
|
729
|
+
* (exchangeability guard). Default 1337. */
|
|
730
|
+
shuffleSeed?: number;
|
|
731
|
+
/** Bind the pre-registered hypothesis. Verified (content hash) at
|
|
732
|
+
* construction; alpha/maxN/direction/minEffect come FROM the manifest. */
|
|
733
|
+
preRegistration?: SignedManifest;
|
|
734
|
+
/** Override the gate name in reports. */
|
|
735
|
+
name?: string;
|
|
736
|
+
}
|
|
737
|
+
interface SequentialPairedGate<TArtifact = unknown, TScenario extends Scenario = Scenario> extends Gate<TArtifact, TScenario> {
|
|
738
|
+
/** Streaming entry point: feed one paired per-scenario delta
|
|
739
|
+
* (candidate − baseline, native scale). Each gate instance carries ONE
|
|
740
|
+
* observe-stream; `decide(ctx)` runs on its own fresh stream and never
|
|
741
|
+
* consumes or advances this one. 'promote' is sticky; observing past the
|
|
742
|
+
* pre-registered maxN throws (extending a finished stream after seeing
|
|
743
|
+
* the result reopens optional stopping — start a NEW pre-registered
|
|
744
|
+
* test). */
|
|
745
|
+
observe(delta: number): SequentialObservation;
|
|
746
|
+
/** Read-only snapshot of the observe-stream. */
|
|
747
|
+
state(): EProcessState & {
|
|
748
|
+
decision: SequentialDecision;
|
|
749
|
+
};
|
|
750
|
+
}
|
|
751
|
+
/**
|
|
752
|
+
* Anytime-valid sequential paired gate. Conforms to the existing `Gate`
|
|
753
|
+
* contract (`decide(ctx)` consumes candidate vs baseline judge scores via
|
|
754
|
+
* `pairHoldout` — same pairing granularity as the fixed-n gates: full cellId,
|
|
755
|
+
* never scenarioId) and adds a streaming `observe(delta)` entry for campaigns
|
|
756
|
+
* that score cells incrementally and want to stop mid-stream.
|
|
757
|
+
*
|
|
758
|
+
* Decision mapping onto the substrate's five-valued `GateDecision`:
|
|
759
|
+
* - 'promote' → 'ship'
|
|
760
|
+
* - 'continue' → 'need_more_work' (stream ended before maxN with
|
|
761
|
+
* the e-value undecided — more reps could decide)
|
|
762
|
+
* - 'undecided-at-maxN' → 'hold', with the reason stating it is NOT
|
|
763
|
+
* evidence of no effect (never a silent default)
|
|
764
|
+
*/
|
|
765
|
+
declare function sequentialPairedGate<TArtifact = unknown, TScenario extends Scenario = Scenario>(options: SequentialPairedGateOptions): SequentialPairedGate<TArtifact, TScenario>;
|
|
766
|
+
interface SequentialDecideOptions {
|
|
767
|
+
/** Type-I budget for the early-stop evidence. Default 0.05. */
|
|
768
|
+
alpha?: number;
|
|
769
|
+
/** Minimum paired deltas before a stop may fire. Default 5. */
|
|
770
|
+
minN?: number;
|
|
771
|
+
/** Bet truncation forwarded to `eProcess`. Default 0.5. */
|
|
772
|
+
maxBet?: number;
|
|
773
|
+
/** Bound on |per-scenario composite delta|. Default 1. */
|
|
774
|
+
scale?: number;
|
|
775
|
+
}
|
|
776
|
+
interface SequentialDecideFn {
|
|
777
|
+
(args: {
|
|
778
|
+
history: GenerationRecord[];
|
|
779
|
+
}): {
|
|
780
|
+
stop: boolean;
|
|
781
|
+
reason?: string;
|
|
782
|
+
};
|
|
783
|
+
/** Read-only snapshot of the accumulated e-process (observability + tests). */
|
|
784
|
+
state(): EProcessState;
|
|
785
|
+
}
|
|
786
|
+
/**
|
|
787
|
+
* `SurfaceProposer.decide` adapter — stops the optimization loop the moment
|
|
788
|
+
* the e-process decides the loop has produced a real improvement, instead of
|
|
789
|
+
* always running `maxGenerations`.
|
|
790
|
+
*
|
|
791
|
+
* Stream: for each generation g ≥ 1, the per-scenario composite deltas of
|
|
792
|
+
* generation g's top candidate vs the generation-0 top candidate (the
|
|
793
|
+
* incumbent the loop set out to beat), paired by scenarioId. H0: no proposed
|
|
794
|
+
* surface improves any scenario's expected composite over the incumbent —
|
|
795
|
+
* under it every delta has conditional mean ≤ 0 and the e-process is valid.
|
|
796
|
+
* Once wealth ≥ 1/alpha the loop stops and hands the winner to the promotion
|
|
797
|
+
* gate (which re-scores on HELD-OUT data — this adapter only spends the
|
|
798
|
+
* exploration budget, it never promotes).
|
|
799
|
+
*
|
|
800
|
+
* Honesty caveats: (1) the incumbent's scores are measured once and shared
|
|
801
|
+
* across all generations' deltas, so type-I control is exact only insofar as
|
|
802
|
+
* those scores approximate the incumbent's true per-scenario means (more reps
|
|
803
|
+
* → tighter); (2) an UNDECIDED process never stops the loop — absence of a
|
|
804
|
+
* crossing is NOT evidence of no effect, so the loop simply runs its normal
|
|
805
|
+
* course. Calling the adapter repeatedly with a growing history consumes each
|
|
806
|
+
* generation exactly once (re-feeding an already-seen record would double-count
|
|
807
|
+
* evidence).
|
|
808
|
+
*/
|
|
809
|
+
declare function sequentialDecide(options?: SequentialDecideOptions): SequentialDecideFn;
|
|
810
|
+
//#endregion
|
|
811
|
+
//#region src/campaign/gates/statistical-heldout.d.ts
|
|
812
|
+
interface PairedHoldout {
|
|
813
|
+
/** Baseline scalar per paired cell (same order as `after`/`cellIds`). */
|
|
814
|
+
before: number[];
|
|
815
|
+
/** Candidate scalar per paired cell. */
|
|
816
|
+
after: number[];
|
|
817
|
+
/** The full cellIds (`scenario:rep`) that paired, in order. */
|
|
818
|
+
cellIds: string[];
|
|
819
|
+
}
|
|
820
|
+
/**
|
|
821
|
+
* Pair candidate vs baseline holdout observations by FULL cellId. `select`
|
|
822
|
+
* pulls the scalar from a cell's judge reports (composite, or a named
|
|
823
|
+
* dimension); a cell contributes the mean of `select` across its judges. Cells
|
|
824
|
+
* whose scenario is not in `scenarioIds`, or where `select` is undefined for
|
|
825
|
+
* every judge on either side, are skipped on BOTH sides so the arrays stay
|
|
826
|
+
* paired. Throws when the two maps disagree on which holdout cells exist — a
|
|
827
|
+
* load-bearing invariant: the baseline + winner holdout campaigns run the same
|
|
828
|
+
* scenarios with the same seed base, so their cellIds MUST align; a mismatch
|
|
829
|
+
* means a silent pairing bug, not a soft fallback.
|
|
830
|
+
*/
|
|
831
|
+
declare function pairHoldout(candidate: Map<string, Record<string, JudgeScore>>, baseline: Map<string, Record<string, JudgeScore>>, scenarioIds: Set<string>, select: (s: JudgeScore) => number | undefined): PairedHoldout;
|
|
832
|
+
interface HeldoutSignificance {
|
|
833
|
+
paired: PairedHoldout;
|
|
834
|
+
/** The bootstrap the ship decision keys on — of the MEAN paired delta by
|
|
835
|
+
* default (see the tie note on `heldoutSignificance`). */
|
|
836
|
+
bootstrap: PairedBootstrapResult;
|
|
837
|
+
/** The MEDIAN paired-delta bootstrap, reported as a diagnostic. When many
|
|
838
|
+
* scenarios are tied (both sides solve them), the median is pinned near 0
|
|
839
|
+
* regardless of the mean lift — comparing the two exposes tie-domination. */
|
|
840
|
+
medianBootstrap: PairedBootstrapResult;
|
|
841
|
+
/** Fraction of paired observations that are exact ties (|delta| < 1e-9). A
|
|
842
|
+
* high tie fraction is WHY a median-based gate would have missed a real lift;
|
|
843
|
+
* it is the observability the tie fix adds. */
|
|
844
|
+
tieFraction: number;
|
|
845
|
+
/** n paired observations. */
|
|
846
|
+
n: number;
|
|
847
|
+
/** True iff n >= minProductiveRuns AND the CI lower bound clears the threshold. */
|
|
848
|
+
significant: boolean;
|
|
849
|
+
/** Set when n < minProductiveRuns — too little evidence to claim significance. */
|
|
850
|
+
fewRuns: boolean;
|
|
851
|
+
}
|
|
852
|
+
interface HeldoutSignificanceOptions {
|
|
853
|
+
deltaThreshold?: number;
|
|
854
|
+
minProductiveRuns?: number;
|
|
855
|
+
confidence?: number;
|
|
856
|
+
resamples?: number;
|
|
857
|
+
/** Fixed by default for a deterministic, reproducible gate verdict. */
|
|
858
|
+
seed?: number;
|
|
859
|
+
statistic?: 'mean' | 'median';
|
|
860
|
+
}
|
|
861
|
+
/** Significance of the held-out composite lift: ship only when the paired
|
|
862
|
+
* bootstrap CI lower bound on (candidate − baseline) exceeds `deltaThreshold`
|
|
863
|
+
* (default 0 ⇒ "confidently positive"). Below `minProductiveRuns` paired
|
|
864
|
+
* observations there is not enough evidence to claim significance → not
|
|
865
|
+
* significant (`fewRuns`). Interpret `deltaThreshold` in the judge's native
|
|
866
|
+
* composite scale. */
|
|
867
|
+
declare function heldoutSignificance(paired: PairedHoldout, opts?: HeldoutSignificanceOptions): HeldoutSignificance;
|
|
868
|
+
interface DimensionRegression {
|
|
869
|
+
dimension: string;
|
|
870
|
+
bootstrap: PairedBootstrapResult;
|
|
871
|
+
/** True iff the CI lower bound on (candidate − baseline) is below −tolerance:
|
|
872
|
+
* the candidate may have regressed this dimension by more than tolerance. */
|
|
873
|
+
regressed: boolean;
|
|
874
|
+
tolerance: number;
|
|
875
|
+
n: number;
|
|
876
|
+
}
|
|
877
|
+
/** Detect the native scale of a set of scores: 0-100 when any magnitude clears
|
|
878
|
+
* 1.5, else [0,1]. Used to auto-scale the regression tolerance so a default
|
|
879
|
+
* expressed for [0,1] is not silently a no-op on a 0-100 dimension. */
|
|
880
|
+
declare function detectScale(values: number[]): 1 | 100;
|
|
881
|
+
/** Per-critical-dimension regression guard. For each dimension, pair the
|
|
882
|
+
* candidate vs baseline values by full cellId and bootstrap the paired delta;
|
|
883
|
+
* a dimension is "regressed" when the CI lower bound < −tolerance (conservative
|
|
884
|
+
* — blocks if the credible worst case exceeds tolerance, which is the right
|
|
885
|
+
* posture for safety dimensions like `hallucination_free`). When `tolerance`
|
|
886
|
+
* is omitted it auto-scales: 0.05 on [0,1], 5 on 0-100. */
|
|
887
|
+
declare function dimensionRegressions(candidate: Map<string, Record<string, JudgeScore>>, baseline: Map<string, Record<string, JudgeScore>>, scenarioIds: Set<string>, criticalDimensions: string[], opts?: {
|
|
888
|
+
tolerance?: number;
|
|
889
|
+
confidence?: number;
|
|
890
|
+
resamples?: number;
|
|
891
|
+
seed?: number;
|
|
892
|
+
}): DimensionRegression[];
|
|
893
|
+
//#endregion
|
|
894
|
+
//#region src/campaign/grounded-reflection.d.ts
|
|
895
|
+
/**
|
|
896
|
+
* Evidence grounding for reflective optimizers (GEPA-style revise loops).
|
|
897
|
+
*
|
|
898
|
+
* Two failure modes recur when an LLM revises an artifact from raw rollout
|
|
899
|
+
* traces (first measured in agent-lab R358, where naive reflection REGRESSED
|
|
900
|
+
* the score 0.375 -> 0.125 before these helpers fixed it):
|
|
901
|
+
*
|
|
902
|
+
* 1. The environment often hides WHY a rollout failed - a tool call can
|
|
903
|
+
* succeed while an invisible downstream check fails - so the reviser
|
|
904
|
+
* cannot see the cause in the transcript. The only reliable signal is the
|
|
905
|
+
* field-level difference between what passing and failing rollouts did.
|
|
906
|
+
* `rolloutArgumentDiff` computes that difference deterministically so the
|
|
907
|
+
* reviser is handed the diff instead of being trusted to derive it.
|
|
908
|
+
*
|
|
909
|
+
* 2. Revisers invent plausible-but-wrong literal values ("use 'new'",
|
|
910
|
+
* "use 'sent'") that no passing rollout ever used, turning every rollout
|
|
911
|
+
* into a failure. `classifyUngroundedLiterals` mechanically detects them,
|
|
912
|
+
* separating HARMFUL literals (ones failing rollouts actually used -
|
|
913
|
+
* proven damage) from benign illustrations (e.g. a name example like
|
|
914
|
+
* 'Doe'), so callers can hard-reject the former and merely log the latter.
|
|
915
|
+
* Rejecting every ungrounded quoted word is too blunt: it killed a run
|
|
916
|
+
* over a surname illustration before the severity split existed.
|
|
917
|
+
*
|
|
918
|
+
* Pure data in, data out: no LLM calls, no filesystem, no domain knowledge.
|
|
919
|
+
*/
|
|
920
|
+
/** One tool/action call observed in a rollout: a name plus its arguments. */
|
|
921
|
+
interface RolloutCall {
|
|
922
|
+
readonly name: string;
|
|
923
|
+
readonly args: Readonly<Record<string, unknown>>;
|
|
924
|
+
}
|
|
925
|
+
/** A scored rollout: its calls plus the scalar outcome used to split pass/fail. */
|
|
926
|
+
interface ScoredRollout {
|
|
927
|
+
/** Caller-meaningful identifier (task id, cell id) used only for reporting. */
|
|
928
|
+
readonly id: string;
|
|
929
|
+
/** Scalar outcome in [0, 1]; `passThreshold` splits passing from failing. */
|
|
930
|
+
readonly score: number;
|
|
931
|
+
readonly calls: readonly RolloutCall[];
|
|
932
|
+
}
|
|
933
|
+
interface RolloutArgumentDiffOptions {
|
|
934
|
+
/** Rollouts with `score >= passThreshold` count as passing. Default 1. */
|
|
935
|
+
readonly passThreshold?: number;
|
|
936
|
+
/** Max distinct values listed per field per side in the rendered text. Default 4. */
|
|
937
|
+
readonly maxValuesPerField?: number;
|
|
938
|
+
}
|
|
939
|
+
interface RolloutArgumentDiff {
|
|
940
|
+
/** Human/LLM-readable per-field diff, one line per field. */
|
|
941
|
+
readonly text: string;
|
|
942
|
+
/** Lowercased stringified argument values seen in passing rollouts. */
|
|
943
|
+
readonly passingValues: ReadonlySet<string>;
|
|
944
|
+
/** Lowercased stringified argument values seen in failing rollouts. */
|
|
945
|
+
readonly failingValues: ReadonlySet<string>;
|
|
946
|
+
}
|
|
947
|
+
/**
|
|
948
|
+
* Deterministic per-field diff of call arguments between passing and failing
|
|
949
|
+
* rollouts. A field set by failing rollouts but left unset by passing ones is
|
|
950
|
+
* the classic poison-input signature; a field whose values differ across the
|
|
951
|
+
* split points at the correct value. Feed `text` to the reviser verbatim.
|
|
952
|
+
*/
|
|
953
|
+
declare function rolloutArgumentDiff(rollouts: readonly ScoredRollout[], opts?: RolloutArgumentDiffOptions): RolloutArgumentDiff;
|
|
954
|
+
interface UngroundedLiteralReport {
|
|
955
|
+
/** Quoted single-word literals in the text that no passing rollout used. */
|
|
956
|
+
readonly ungrounded: readonly string[];
|
|
957
|
+
/** The subset failing rollouts actually used - prescribing these is proven harmful. */
|
|
958
|
+
readonly harmful: readonly string[];
|
|
959
|
+
}
|
|
960
|
+
/**
|
|
961
|
+
* Scan revised artifact text for single-quoted single-word literals (the
|
|
962
|
+
* "use exactly 'new'" pattern) that appear in no passing rollout's argument
|
|
963
|
+
* values. Multi-word quotes pass (they are prose, not prescriptions).
|
|
964
|
+
* Callers should reject on `harmful` (with a bounded retry) and at most log
|
|
965
|
+
* `ungrounded` - see the module header for why the severities differ.
|
|
966
|
+
*/
|
|
967
|
+
declare function classifyUngroundedLiterals(text: string, diff: Pick<RolloutArgumentDiff, 'passingValues' | 'failingValues'>): UngroundedLiteralReport;
|
|
968
|
+
//#endregion
|
|
969
|
+
//#region src/campaign/labeled-store/fs-adapter.d.ts
|
|
970
|
+
interface FsLabeledScenarioStoreOptions {
|
|
971
|
+
/** Root directory for JSONL files. Created if missing. */
|
|
972
|
+
root: string;
|
|
973
|
+
/** Per-source rate limit. When set, writes exceeding the cap are rejected
|
|
974
|
+
* with a typed error. Default: no limit. */
|
|
975
|
+
maxWritesPerMinutePerBucket?: number;
|
|
976
|
+
/** Test seam — override `Date.now()` for deterministic tests. */
|
|
977
|
+
now?: () => number;
|
|
978
|
+
}
|
|
979
|
+
/** Typed rejection from a labeled-scenario store (bad provenance, rate limit, invalid sample args) — carries a stable string `code`. */
|
|
980
|
+
declare class LabeledScenarioStoreError extends Error {
|
|
981
|
+
readonly code: string;
|
|
982
|
+
constructor(code: string, message: string);
|
|
983
|
+
}
|
|
984
|
+
/**
|
|
985
|
+
* Filesystem `LabeledScenarioStore`: appends one JSONL file per source with provenance and
|
|
986
|
+
* rate-limit guards. For tests, local dev, and small workloads — high-throughput lands in Turso.
|
|
987
|
+
*/
|
|
988
|
+
declare class FsLabeledScenarioStore implements LabeledScenarioStore {
|
|
989
|
+
private readonly options;
|
|
990
|
+
private readonly now;
|
|
991
|
+
private readonly rateLimits;
|
|
992
|
+
constructor(options: FsLabeledScenarioStoreOptions);
|
|
993
|
+
observe(write: LabeledScenarioWrite): Promise<void>;
|
|
994
|
+
sample(args: LabeledScenarioSampleArgs): Promise<LabeledScenarioRecord[]>;
|
|
995
|
+
size(): Promise<{
|
|
996
|
+
train: number;
|
|
997
|
+
test: number;
|
|
998
|
+
bySource: Record<string, number>;
|
|
999
|
+
byTrust: Record<LabelTrust, number>;
|
|
1000
|
+
}>;
|
|
1001
|
+
private assertProvenance;
|
|
1002
|
+
private assertRateLimit;
|
|
1003
|
+
private toRecord;
|
|
1004
|
+
private pathForSource;
|
|
1005
|
+
}
|
|
1006
|
+
//#endregion
|
|
1007
|
+
//#region src/campaign/neutralize.d.ts
|
|
1008
|
+
/**
|
|
1009
|
+
* @module
|
|
1010
|
+
* Footprint-matched neutralization — the placebo control for content-vs-footprint
|
|
1011
|
+
* attribution in a promotion gate.
|
|
1012
|
+
*
|
|
1013
|
+
* A promoted surface can raise a held-out score two different ways:
|
|
1014
|
+
* 1. its CONTENT is informative (the thing we want to promote), or
|
|
1015
|
+
* 2. it merely added prompt/mount FOOTPRINT — more bytes, more lines, a longer
|
|
1016
|
+
* more authoritative-looking prompt — that the model spends attention on
|
|
1017
|
+
* regardless of what the bytes say.
|
|
1018
|
+
*
|
|
1019
|
+
* A held-out gate proves the candidate beat baseline; it cannot separate (1) from
|
|
1020
|
+
* (2). `neutralizeText` produces a variant that keeps the input's layout and
|
|
1021
|
+
* length while carrying ZERO information, so scoring it isolates the footprint
|
|
1022
|
+
* contribution (2). Feed the neutralized variant's scores to `neutralizationGate`:
|
|
1023
|
+
* any lift it still holds over baseline is decorative, and a candidate whose lift
|
|
1024
|
+
* survives neutralization is rejected however large its raw lift.
|
|
1025
|
+
*/
|
|
1026
|
+
/**
|
|
1027
|
+
* Blank every non-whitespace character to a 1-byte filler while preserving all
|
|
1028
|
+
* whitespace. Line count, indentation, and word/line lengths are unchanged — so
|
|
1029
|
+
* the neutralized variant has the same layout and (for ASCII) the same byte
|
|
1030
|
+
* footprint as the input, but no readable content. Whitespace is preserved
|
|
1031
|
+
* deliberately: collapsing it would change the token structure and stop the
|
|
1032
|
+
* variant from being a true footprint match.
|
|
1033
|
+
*/
|
|
1034
|
+
declare function neutralizeText(content: string): string;
|
|
1035
|
+
//#endregion
|
|
1036
|
+
//#region src/artifact-validator.d.ts
|
|
1037
|
+
/**
|
|
1038
|
+
* Artifact validators.
|
|
1039
|
+
*
|
|
1040
|
+
* Generic "score a produced artifact" primitive. Tax uses it for PDF form
|
|
1041
|
+
* correctness, research for sourced briefs, browser for task assertions, coding
|
|
1042
|
+
* for social posts. One interface, many validators; all plug into
|
|
1043
|
+
* `BenchmarkRunner` the same way.
|
|
1044
|
+
*
|
|
1045
|
+
* A validator receives an `Artifact` (file on disk, JSON blob, text, binary)
|
|
1046
|
+
* plus a `ValidationContext` (scenario id, the turns that produced it) and
|
|
1047
|
+
* returns a `ValidationResult` with pass/fail + 0..1 score + structured
|
|
1048
|
+
* issues.
|
|
1049
|
+
*/
|
|
1050
|
+
interface Artifact {
|
|
1051
|
+
/** Logical kind — validators type-guard on this */
|
|
1052
|
+
kind: 'file' | 'json' | 'text' | 'binary' | string;
|
|
1053
|
+
/** Filesystem-style path, optional */
|
|
1054
|
+
path?: string;
|
|
1055
|
+
/** String content for text/json/file kinds */
|
|
1056
|
+
content?: string;
|
|
1057
|
+
/** Binary content (if kind === 'binary') */
|
|
1058
|
+
bytes?: Uint8Array;
|
|
1059
|
+
/** Caller-supplied metadata (mimeType, sha256, size, etc.) */
|
|
1060
|
+
metadata?: Record<string, unknown>;
|
|
1061
|
+
}
|
|
1062
|
+
interface ValidationContext {
|
|
1063
|
+
scenarioId: string;
|
|
1064
|
+
turnIndex?: number;
|
|
1065
|
+
/** Prior artifacts for multi-artifact scenarios */
|
|
1066
|
+
priorArtifacts?: Artifact[];
|
|
1067
|
+
/** Free-form hints the validator uses for domain-specific checks */
|
|
1068
|
+
hints?: Record<string, unknown>;
|
|
1069
|
+
}
|
|
1070
|
+
interface ValidationIssue {
|
|
1071
|
+
severity: 'error' | 'warning' | 'info';
|
|
1072
|
+
message: string;
|
|
1073
|
+
/** Optional path into the artifact (e.g. JSON path or byte offset) */
|
|
1074
|
+
locus?: string;
|
|
1075
|
+
}
|
|
1076
|
+
interface ValidationResult {
|
|
1077
|
+
pass: boolean;
|
|
1078
|
+
/** 0–1 normalized score. Validators should be monotonic in pass-ness. */
|
|
1079
|
+
score: number;
|
|
1080
|
+
issues: ValidationIssue[];
|
|
1081
|
+
/** Diagnostic payload for reporters */
|
|
1082
|
+
evidence?: Record<string, unknown>;
|
|
1083
|
+
}
|
|
1084
|
+
interface ArtifactValidator {
|
|
1085
|
+
/** Stable identifier for the validator; appears in reports. */
|
|
1086
|
+
name: string;
|
|
1087
|
+
/** Optional description for human-facing reports. */
|
|
1088
|
+
description?: string;
|
|
1089
|
+
/** Called once per artifact; validators are expected to be pure + idempotent. */
|
|
1090
|
+
validate(artifact: Artifact, context: ValidationContext): Promise<ValidationResult>;
|
|
1091
|
+
}
|
|
1092
|
+
/**
|
|
1093
|
+
* Run every validator on the same artifact; aggregate pass as AND, score as
|
|
1094
|
+
* (weighted) mean, issues concatenated. Weights default to 1 each.
|
|
1095
|
+
*/
|
|
1096
|
+
declare function composeValidators(validators: ArtifactValidator[], options?: {
|
|
1097
|
+
name?: string;
|
|
1098
|
+
weights?: number[];
|
|
1099
|
+
}): ArtifactValidator;
|
|
1100
|
+
/** Pass if the artifact body matches a provided regex. */
|
|
1101
|
+
declare function regexMatch(name: string, pattern: RegExp): ArtifactValidator;
|
|
1102
|
+
/** Pass if JSON parses and every required key is present. */
|
|
1103
|
+
declare function jsonHasKeys(name: string, requiredPaths: string[]): ArtifactValidator;
|
|
1104
|
+
/** Pass if min ≤ byte length ≤ max. */
|
|
1105
|
+
declare function byteLengthRange(name: string, min: number, max: number): ArtifactValidator;
|
|
1106
|
+
/** Pass if the artifact contains every required substring (case-insensitive by default). */
|
|
1107
|
+
declare function containsAll(name: string, required: string[], options?: {
|
|
1108
|
+
caseSensitive?: boolean;
|
|
1109
|
+
}): ArtifactValidator;
|
|
1110
|
+
//#endregion
|
|
1111
|
+
//#region src/completion-verifier.d.ts
|
|
1112
|
+
/** What kind of produced state can satisfy a requirement structurally. */
|
|
1113
|
+
type SatisfiedBy = 'artifact' | 'proposal' | 'tool-call' | 'any';
|
|
1114
|
+
interface CompletionRequirement {
|
|
1115
|
+
/** Stable id from the task gold (e.g. a persona's `expected_requirements[].req_id`). */
|
|
1116
|
+
reqId: string;
|
|
1117
|
+
/** Human-readable description of the required deliverable. */
|
|
1118
|
+
title: string;
|
|
1119
|
+
/** Optional kind/category hint, matched against a produced item's kind. */
|
|
1120
|
+
category?: string;
|
|
1121
|
+
/** What produced state satisfies this requirement. Defaults to 'any'. */
|
|
1122
|
+
satisfiedBy?: SatisfiedBy;
|
|
1123
|
+
}
|
|
1124
|
+
interface TaskGold {
|
|
1125
|
+
taskId: string;
|
|
1126
|
+
requirements: CompletionRequirement[];
|
|
1127
|
+
}
|
|
1128
|
+
interface ProducedProposal {
|
|
1129
|
+
id: string;
|
|
1130
|
+
title: string;
|
|
1131
|
+
status: 'pending' | 'approved' | 'rejected';
|
|
1132
|
+
/** Optional persisted body — when present, enables a correctness check. */
|
|
1133
|
+
content?: string;
|
|
1134
|
+
}
|
|
1135
|
+
/** Everything observable about what a run actually produced. */
|
|
1136
|
+
interface ProducedState {
|
|
1137
|
+
/** Persisted vault artifacts. Reuses the shared `Artifact` shape. */
|
|
1138
|
+
artifacts: Artifact[];
|
|
1139
|
+
/** Proposals / filings the agent created. */
|
|
1140
|
+
proposals: ProducedProposal[];
|
|
1141
|
+
/** Names of tools the agent invoked. */
|
|
1142
|
+
toolCalls: string[];
|
|
1143
|
+
}
|
|
1144
|
+
interface RequirementCheck {
|
|
1145
|
+
reqId: string;
|
|
1146
|
+
title: string;
|
|
1147
|
+
/** A produced item of the right kind matched the requirement, non-empty. */
|
|
1148
|
+
structurallyPresent: boolean;
|
|
1149
|
+
/**
|
|
1150
|
+
* Whether the matched item actually fulfils the requirement. `null` when
|
|
1151
|
+
* not structurally present, when the matched item carries no content
|
|
1152
|
+
* to assess, or when the correctness check itself failed (`unmeasured`).
|
|
1153
|
+
*/
|
|
1154
|
+
correct: boolean | null;
|
|
1155
|
+
/** structurallyPresent && !unmeasured && correct !== false. */
|
|
1156
|
+
satisfied: boolean;
|
|
1157
|
+
/**
|
|
1158
|
+
* Set when the correctness check itself errored (LLM call failure or an
|
|
1159
|
+
* unparseable response after retry). The requirement's fulfilment is
|
|
1160
|
+
* UNKNOWN — `correct` stays null, `satisfied` is false, and
|
|
1161
|
+
* `completionVerdict` excludes the row from `completionRate`'s
|
|
1162
|
+
* denominator. Never folded into a zero: a synthetic zero is
|
|
1163
|
+
* indistinguishable from a real failure (see `JudgeParseError`).
|
|
1164
|
+
*/
|
|
1165
|
+
unmeasured?: true;
|
|
1166
|
+
/** Why the correctness check could not be measured (present iff `unmeasured`). */
|
|
1167
|
+
unmeasuredReason?: string;
|
|
1168
|
+
/** Human-readable evidence for the verdict. */
|
|
1169
|
+
evidence: string[];
|
|
1170
|
+
}
|
|
1171
|
+
/** Extends the substrate verdict spine: `valid` = `fullyComplete` and
|
|
1172
|
+
* `score` = `completionRate` — derived in `completionVerdict()`, the one
|
|
1173
|
+
* place those equalities hold by construction. */
|
|
1174
|
+
interface CompletionVerdict extends DefaultVerdict {
|
|
1175
|
+
taskId: string;
|
|
1176
|
+
requirements: RequirementCheck[];
|
|
1177
|
+
/** satisfied / MEASURABLE requirements (unmeasured rows leave the denominator). */
|
|
1178
|
+
completionRate: number;
|
|
1179
|
+
/** Every measurable requirement satisfied (false when anything is unmeasured). */
|
|
1180
|
+
fullyComplete: boolean;
|
|
1181
|
+
/** Requirements whose correctness check errored — reported, never scored as zero. */
|
|
1182
|
+
unmeasuredCount: number;
|
|
1183
|
+
}
|
|
1184
|
+
/**
|
|
1185
|
+
* Construct a `CompletionVerdict` from the per-requirement checks, deriving
|
|
1186
|
+
* `completionRate` / `fullyComplete` and the spine fields (`valid` =
|
|
1187
|
+
* `fullyComplete`, `score` = `completionRate`) in one place. Throws on zero
|
|
1188
|
+
* requirements — a verdict over nothing is a misconfiguration, mirroring
|
|
1189
|
+
* `verifyCompletion`'s gold-spec guard.
|
|
1190
|
+
*/
|
|
1191
|
+
declare function completionVerdict(input: {
|
|
1192
|
+
taskId: string;
|
|
1193
|
+
requirements: RequirementCheck[];
|
|
1194
|
+
}): CompletionVerdict;
|
|
1195
|
+
/**
|
|
1196
|
+
* Decides whether a produced item's content actually fulfils a requirement.
|
|
1197
|
+
* Injected so the structural verifier stays pure and unit-testable; the
|
|
1198
|
+
* production implementation is `createLlmCorrectnessChecker`.
|
|
1199
|
+
*/
|
|
1200
|
+
type CorrectnessChecker = (requirement: CompletionRequirement, content: string) => Promise<{
|
|
1201
|
+
correct: boolean;
|
|
1202
|
+
reason: string;
|
|
1203
|
+
}>;
|
|
1204
|
+
/**
|
|
1205
|
+
* Verify whether a run completed the task. `checkCorrectness` is injected —
|
|
1206
|
+
* `createLlmCorrectnessChecker` for production, a deterministic stub in tests.
|
|
1207
|
+
*
|
|
1208
|
+
* Throws on a gold spec with no requirements: an eval task that requires
|
|
1209
|
+
* nothing is a misconfiguration, not a vacuously-complete task.
|
|
1210
|
+
*/
|
|
1211
|
+
declare function verifyCompletion(gold: TaskGold, state: ProducedState, checkCorrectness: CorrectnessChecker): Promise<CompletionVerdict>;
|
|
1212
|
+
interface LlmCorrectnessCheckerOpts {
|
|
1213
|
+
model?: string;
|
|
1214
|
+
/** Optional ledger for direct use. */
|
|
1215
|
+
costLedger?: CostLedgerHandle;
|
|
1216
|
+
costPhase?: string;
|
|
1217
|
+
costTags?: Record<string, string>;
|
|
1218
|
+
signal?: AbortSignal;
|
|
1219
|
+
/** Max chars of artifact content sent to the checker. */
|
|
1220
|
+
maxContentChars?: number;
|
|
1221
|
+
/**
|
|
1222
|
+
* Checker LLM calls per requirement before giving up (parse failures and
|
|
1223
|
+
* call errors both consume attempts). The failure then surfaces as an
|
|
1224
|
+
* `unmeasured` requirement, never a zero.
|
|
1225
|
+
*/
|
|
1226
|
+
maxAttempts?: number;
|
|
1227
|
+
/**
|
|
1228
|
+
* Forensic capture of every checker request/response/error — without it a
|
|
1229
|
+
* checker failure is unauditable (the agent-turn raws never contain the
|
|
1230
|
+
* checker's own calls). Same sink contract as `LlmClient`.
|
|
1231
|
+
*/
|
|
1232
|
+
rawSink?: RawProviderSink;
|
|
1233
|
+
}
|
|
1234
|
+
/**
|
|
1235
|
+
* Parse the correctness checker's model response. Tolerates a response
|
|
1236
|
+
* truncated mid-JSON (max_tokens cap) by auto-closing the prefix — the
|
|
1237
|
+
* verdict boolean usually lands in the first few tokens, so a recovered
|
|
1238
|
+
* prefix with a boolean `correct` is a real measurement, not a guess.
|
|
1239
|
+
* Fails loud (JudgeParseError) when no boolean verdict is recoverable.
|
|
1240
|
+
*/
|
|
1241
|
+
declare function parseCorrectnessResponse(raw: string): {
|
|
1242
|
+
correct: boolean;
|
|
1243
|
+
reason: string;
|
|
1244
|
+
};
|
|
1245
|
+
/**
|
|
1246
|
+
* Production `CorrectnessChecker` — one LLM call per matched artifact,
|
|
1247
|
+
* deterministic (temperature 0), structured JSON out. Judges fulfilment
|
|
1248
|
+
* only: a plan, a gesture, or a description of what should be done does not
|
|
1249
|
+
* fulfil a requirement — the artifact must BE the deliverable.
|
|
1250
|
+
*/
|
|
1251
|
+
declare function createLlmCorrectnessChecker(chat: ChatClient, opts?: LlmCorrectnessCheckerOpts): CorrectnessChecker;
|
|
1252
|
+
/**
|
|
1253
|
+
* Deterministic `CorrectnessChecker` — the no-LLM counterpart to
|
|
1254
|
+
* `createLlmCorrectnessChecker`. A produced item fulfils a requirement when its
|
|
1255
|
+
* content is substantive (≥ `minContentLength` chars) AND recalls ≥ `minRecall`
|
|
1256
|
+
* of the requirement title's significant tokens. No network.
|
|
1257
|
+
*
|
|
1258
|
+
* Polarity-blind: token recall credits a negation that contains the
|
|
1259
|
+
* requirement's tokens ("I will NOT produce the comparison" recalls every token
|
|
1260
|
+
* of "produce the comparison"). The structural match stage is ALSO lexical, so
|
|
1261
|
+
* pairing the two collapses to a single gameable gate. Use this only as an
|
|
1262
|
+
* opt-in structural pre-filter or for tasks whose requirements have no polarity
|
|
1263
|
+
* to invert; for produced-state grading the correctness checker MUST be semantic
|
|
1264
|
+
* (`createLlmCorrectnessChecker`). See the anti-game fixtures in the test suite.
|
|
1265
|
+
*/
|
|
1266
|
+
declare function createTokenRecallChecker(opts?: {
|
|
1267
|
+
minRecall?: number;
|
|
1268
|
+
minContentLength?: number;
|
|
1269
|
+
}): CorrectnessChecker;
|
|
1270
|
+
//#endregion
|
|
1271
|
+
//#region src/produced-state.d.ts
|
|
1272
|
+
/** A tool the agent invoked. */
|
|
1273
|
+
interface ToolCallEventLike {
|
|
1274
|
+
type: 'tool_call';
|
|
1275
|
+
toolName: string;
|
|
1276
|
+
}
|
|
1277
|
+
/**
|
|
1278
|
+
* An artifact the agent produced. `content` is the enriched field — the
|
|
1279
|
+
* runtime's base `artifact` event carries only metadata; the completion
|
|
1280
|
+
* oracle needs the body to verify the deliverable, so the runtime emits it.
|
|
1281
|
+
*/
|
|
1282
|
+
interface ArtifactEventLike {
|
|
1283
|
+
type: 'artifact';
|
|
1284
|
+
artifactId: string;
|
|
1285
|
+
name?: string;
|
|
1286
|
+
mimeType?: string;
|
|
1287
|
+
uri?: string;
|
|
1288
|
+
content?: string;
|
|
1289
|
+
}
|
|
1290
|
+
/** A proposal / filing the agent created. */
|
|
1291
|
+
interface ProposalEventLike {
|
|
1292
|
+
type: 'proposal_created';
|
|
1293
|
+
proposalId: string;
|
|
1294
|
+
title: string;
|
|
1295
|
+
status?: 'pending' | 'approved' | 'rejected';
|
|
1296
|
+
content?: string;
|
|
1297
|
+
}
|
|
1298
|
+
/**
|
|
1299
|
+
* The subset of runtime stream events `extractProducedState` consumes.
|
|
1300
|
+
* agent-runtime's full `RuntimeStreamEvent` union satisfies this structurally;
|
|
1301
|
+
* the `{ type: string }` catch-all keeps the input permissive so callers can
|
|
1302
|
+
* pass the whole unfiltered telemetry stream — unrecognized events are skipped.
|
|
1303
|
+
*/
|
|
1304
|
+
type RuntimeEventLike = ToolCallEventLike | ArtifactEventLike | ProposalEventLike | {
|
|
1305
|
+
type: string;
|
|
1306
|
+
};
|
|
1307
|
+
/**
|
|
1308
|
+
* Normalize a run's runtime event stream into `ProducedState`.
|
|
1309
|
+
*
|
|
1310
|
+
* Pure and total — unrecognized event types are skipped. `toolCalls` is
|
|
1311
|
+
* deduplicated by name in first-seen order (completion cares about a tool's
|
|
1312
|
+
* presence, not its call count). An artifact with neither a name nor a uri
|
|
1313
|
+
* still yields an entry keyed by its `artifactId` so it is never silently
|
|
1314
|
+
* dropped; an artifact with no `content` yields empty content, which the
|
|
1315
|
+
* completion oracle's structural check then rejects on its own.
|
|
1316
|
+
*/
|
|
1317
|
+
declare function extractProducedState(events: readonly RuntimeEventLike[]): ProducedState;
|
|
1318
|
+
//#endregion
|
|
1319
|
+
//#region src/agent-profile.d.ts
|
|
1320
|
+
/**
|
|
1321
|
+
* The agentic coding harnesses an eval sweeps by default — the ones we care about
|
|
1322
|
+
* ranking. This is the SINGLE source of that list; consumers import it instead of
|
|
1323
|
+
* re-declaring their own (a re-declared list is how the fleet drifts). Pass an
|
|
1324
|
+
* explicit `harnesses` (e.g. `harnessTypeSchema.options` for literally every known
|
|
1325
|
+
* harness) to widen beyond these.
|
|
1326
|
+
*/
|
|
1327
|
+
declare const CODING_HARNESSES: readonly HarnessType[];
|
|
1328
|
+
interface ProfileAxisSpec {
|
|
1329
|
+
/** The domain profile to sweep. Its prompt/tools/skills are held fixed; only the
|
|
1330
|
+
* harness and model vary. `model.default` is the fallback model. */
|
|
1331
|
+
base: AgentProfile;
|
|
1332
|
+
/** Harnesses to cross. Default: {@link CODING_HARNESSES}. */
|
|
1333
|
+
harnesses?: readonly HarnessType[];
|
|
1334
|
+
/** Models to cross. Default: `[base.model.default]` — one model, i.e. today's
|
|
1335
|
+
* single-model behaviour, so omitting this never changes an existing run. */
|
|
1336
|
+
models?: readonly string[];
|
|
1337
|
+
/** Force every (harness, model) pair verbatim, even ones the harness can't run —
|
|
1338
|
+
* for deliberately testing failure modes. Default (false): SNAP instead — a
|
|
1339
|
+
* vendor-locked harness runs only the swept models in its family, or its native
|
|
1340
|
+
* default when it supports none, so no harness is dropped and none gets a
|
|
1341
|
+
* guaranteed-failing foreign-model cell. */
|
|
1342
|
+
keepIncompatible?: boolean;
|
|
1343
|
+
}
|
|
1344
|
+
/** Model sentinel for a vendor-locked harness that supports none of the swept models:
|
|
1345
|
+
* it carries no provider prefix, so `harnessSupportsModel` accepts it and the harness
|
|
1346
|
+
* resolves it to its own native default model at runtime (e.g. kimi-code → its Kimi
|
|
1347
|
+
* model). Lets `expandProfileAxes` snap-instead-of-drop without a per-harness flagship
|
|
1348
|
+
* table that would rot as router catalogs change. */
|
|
1349
|
+
declare const HARNESS_NATIVE_MODEL = "default";
|
|
1350
|
+
/**
|
|
1351
|
+
* Expand a base profile across the harness × model matrix into the `AgentProfile[]`
|
|
1352
|
+
* that `runProfileMatrix` / `selfImprove` score — the ONE place "which harnesses ×
|
|
1353
|
+
* which models do we evaluate" lives, so no product hand-rolls its own harness list
|
|
1354
|
+
* or column→profile mapping (the pattern that let those copies drift and silently
|
|
1355
|
+
* break the harness pivot).
|
|
1356
|
+
*
|
|
1357
|
+
* Each cell clones `base`, sets `model.default`, and stamps `metadata.harness` +
|
|
1358
|
+
* `metadata.harnessModel` (both hash-bearing, so every cell gets a distinct
|
|
1359
|
+
* `agentProfileId` row and results join back by harness/model via {@link harnessAxisOf}
|
|
1360
|
+
* with no hand-recomputed key). A vendor-locked harness snaps to its family's swept
|
|
1361
|
+
* models — or its native default ({@link HARNESS_NATIVE_MODEL}) when it supports none —
|
|
1362
|
+
* so every requested harness runs; `keepIncompatible` forces every pair verbatim.
|
|
1363
|
+
*
|
|
1364
|
+
* Omit `harnesses`/`models` to sweep the full default set — the "turn it on for
|
|
1365
|
+
* everything we care about" switch, identical in shape whether one harness or all.
|
|
1366
|
+
*/
|
|
1367
|
+
declare function expandProfileAxes(spec: ProfileAxisSpec): AgentProfile[];
|
|
1368
|
+
/**
|
|
1369
|
+
* Read the (harness, model) a matrix cell ran under, off a profile or a result row's
|
|
1370
|
+
* profile — the join-back for a `byHarness` pivot. Returns undefined when the profile
|
|
1371
|
+
* wasn't produced by {@link expandProfileAxes}. Callers group `result.byProfile` by
|
|
1372
|
+
* this instead of recomputing an id (recomputing the wrong key is what broke the pivot
|
|
1373
|
+
* in the hand-rolled copies).
|
|
1374
|
+
*/
|
|
1375
|
+
declare function harnessAxisOf(profile: Pick<AgentProfile, 'metadata'>): {
|
|
1376
|
+
harness: HarnessType;
|
|
1377
|
+
model: string;
|
|
1378
|
+
} | undefined;
|
|
1379
|
+
/**
|
|
1380
|
+
* Collision-resistant, path-safe, human-readable profile id for eval artifacts.
|
|
1381
|
+
* Scorecard joins still use `agentProfileHash`; this id is for run ids, matrix
|
|
1382
|
+
* keys, and directory names where two profiles must not collapse onto one row.
|
|
1383
|
+
* The suffix is the first 64 bits of the behaviour hash, enough for ordinary
|
|
1384
|
+
* eval matrices while keeping filenames readable.
|
|
1385
|
+
*/
|
|
1386
|
+
declare function agentProfileId(profile: AgentProfile): string;
|
|
1387
|
+
/**
|
|
1388
|
+
* Model snapshot used for `RunRecord.model`. Eval surfaces require a concrete
|
|
1389
|
+
* model id because run records reject bare/missing model aliases.
|
|
1390
|
+
*/
|
|
1391
|
+
declare function agentProfileModelId(profile: AgentProfile): string;
|
|
1392
|
+
/**
|
|
1393
|
+
* Deterministic behaviour identity for the canonical
|
|
1394
|
+
* `@tangle-network/agent-interface` AgentProfile.
|
|
1395
|
+
*
|
|
1396
|
+
* `name` and `description` are labels and do not affect the hash. Profile
|
|
1397
|
+
* `version`, prompt, model hints, tools, resources, hooks, modes, permissions,
|
|
1398
|
+
* and extensions do affect the hash. Resource array order is hash-bearing
|
|
1399
|
+
* because mount order can change agent behaviour. Undefined fields are treated
|
|
1400
|
+
* as absent; explicit `null` fields remain hash-bearing.
|
|
1401
|
+
*/
|
|
1402
|
+
declare function agentProfileHash(profile: AgentProfile): string;
|
|
1403
|
+
//#endregion
|
|
1404
|
+
//#region src/integrity/backend-integrity.d.ts
|
|
1405
|
+
interface BackendIntegrityReport {
|
|
1406
|
+
/** Total records inspected. */
|
|
1407
|
+
totalRecords: number;
|
|
1408
|
+
/** Records with input=0 AND output=0 (a stub fingerprint). */
|
|
1409
|
+
stubRecords: number;
|
|
1410
|
+
/** Records with nonzero token usage (real LLM activity). */
|
|
1411
|
+
realRecords: number;
|
|
1412
|
+
/** Records where output>0 but costUsd=0 (real LLM, broken cost ledger). */
|
|
1413
|
+
uncostedRecords: number;
|
|
1414
|
+
/** Sum of input tokens across all records. */
|
|
1415
|
+
totalInputTokens: number;
|
|
1416
|
+
/** Sum of output tokens across all records. */
|
|
1417
|
+
totalOutputTokens: number;
|
|
1418
|
+
/** Sum of costUsd across all records. */
|
|
1419
|
+
totalCostUsd: number;
|
|
1420
|
+
/** Worst-case integrity verdict. */
|
|
1421
|
+
verdict: 'real' | 'mixed' | 'stub';
|
|
1422
|
+
/** Human-readable diagnosis suitable for terminal output. */
|
|
1423
|
+
diagnosis: string;
|
|
1424
|
+
}
|
|
1425
|
+
/**
|
|
1426
|
+
* Error thrown when an integrity assertion fails. Caller can pattern-match
|
|
1427
|
+
* by `code === 'AGENT_EVAL_BACKEND_STUB'` to differentiate from other
|
|
1428
|
+
* errors.
|
|
1429
|
+
*/
|
|
1430
|
+
declare class BackendIntegrityError extends AgentEvalError {
|
|
1431
|
+
readonly report: BackendIntegrityReport;
|
|
1432
|
+
constructor(message: string, report: BackendIntegrityReport);
|
|
1433
|
+
}
|
|
1434
|
+
/**
|
|
1435
|
+
* Inspect a batch of RunRecords and return an integrity report. Pure
|
|
1436
|
+
* function — no I/O, no logging. The caller decides what to do with the
|
|
1437
|
+
* verdict (print warning, throw, gate CI, etc.).
|
|
1438
|
+
*/
|
|
1439
|
+
declare function summarizeBackendIntegrity(records: ReadonlyArray<RunRecord>): BackendIntegrityReport;
|
|
1440
|
+
/** Inspect settled agent calls from the canonical cost ledger. */
|
|
1441
|
+
declare function summarizeAgentReceiptIntegrity(receipts: ReadonlyArray<CostReceipt>): BackendIntegrityReport;
|
|
1442
|
+
/**
|
|
1443
|
+
* Throw BackendIntegrityError if the verdict is 'stub' — i.e. every record
|
|
1444
|
+
* shows zero LLM activity. Non-strict callers can pass `{ allowMixed: false }`
|
|
1445
|
+
* to also reject mixed verdicts (recommended for CI gates).
|
|
1446
|
+
*
|
|
1447
|
+
* Real backends pass through silently.
|
|
1448
|
+
*/
|
|
1449
|
+
declare function assertRealBackend(records: ReadonlyArray<RunRecord>, opts?: {
|
|
1450
|
+
allowMixed?: boolean;
|
|
1451
|
+
}): BackendIntegrityReport;
|
|
1452
|
+
/** Reject a cost ledger with no real agent call or a partial stub run. */
|
|
1453
|
+
declare function assertRealAgentReceipts(receipts: ReadonlyArray<CostReceipt>, opts?: {
|
|
1454
|
+
allowMixed?: boolean;
|
|
1455
|
+
}): BackendIntegrityReport;
|
|
1456
|
+
//#endregion
|
|
1457
|
+
//#region src/campaign/presets/run-profile-matrix.d.ts
|
|
1458
|
+
/** Thrown when the matrix is misconfigured (no profiles, a profile whose model
|
|
1459
|
+
* lacks a snapshot version, etc.). Distinct from `BackendIntegrityError`,
|
|
1460
|
+
* which signals a stub backend at run time. */
|
|
1461
|
+
declare class ProfileMatrixError extends AgentEvalError {
|
|
1462
|
+
constructor(message: string);
|
|
1463
|
+
}
|
|
1464
|
+
/** Dispatch for one cell: render `profile` against `scenario`, returning the
|
|
1465
|
+
* artifact the judges score. Run LLM work through `ctx.cost.runPaidCall` —
|
|
1466
|
+
* the integrity check depends on its receipt. */
|
|
1467
|
+
type ProfileDispatchFn<TScenario extends Scenario, TArtifact> = (profile: AgentProfile$1, scenario: TScenario, ctx: DispatchContext) => Promise<TArtifact>;
|
|
1468
|
+
interface RunProfileMatrixOptions<TScenario extends Scenario, TArtifact> {
|
|
1469
|
+
/** Axis 3 — the agent-under-test configurations. Each is one column. */
|
|
1470
|
+
profiles: AgentProfile$1[];
|
|
1471
|
+
/** Axis 1 — the persona/scenario corpus, run against every profile. */
|
|
1472
|
+
scenarios: TScenario[];
|
|
1473
|
+
/** Renders one (profile, scenario) cell. */
|
|
1474
|
+
dispatch: ProfileDispatchFn<TScenario, TArtifact>;
|
|
1475
|
+
/** The scoring axis. */
|
|
1476
|
+
judges?: JudgeConfig<TArtifact, TScenario>[];
|
|
1477
|
+
/** Where each profile's campaign writes artifacts/traces. One subdir per
|
|
1478
|
+
* profile. */
|
|
1479
|
+
runDir: string;
|
|
1480
|
+
/** Git SHA the harness ran from — stamped onto every RunRecord (mandatory
|
|
1481
|
+
* for paper-grade records). */
|
|
1482
|
+
commitSha: string;
|
|
1483
|
+
/** Logical experiment id shared across the whole matrix so the promotion
|
|
1484
|
+
* gate can pair profiles on matched scenarios. Default: a hash of the
|
|
1485
|
+
* profile + scenario ids. */
|
|
1486
|
+
experimentId?: string;
|
|
1487
|
+
/** Which split these runs belong to. Default `'search'`. */
|
|
1488
|
+
splitTag?: RunSplitTag;
|
|
1489
|
+
/** Replicates per (profile, scenario) cell for CI bands. Default 1. */
|
|
1490
|
+
reps?: number;
|
|
1491
|
+
/** Campaign seed (per profile). Default 42. */
|
|
1492
|
+
seed?: number;
|
|
1493
|
+
/**
|
|
1494
|
+
* Backend-integrity posture, enforced AFTER the matrix completes:
|
|
1495
|
+
* - `'assert'` (default) — throw `BackendIntegrityError` if the run was a
|
|
1496
|
+
* stub (and, with `allowMixed:false`, if it was mixed).
|
|
1497
|
+
* - `'warn'` — log the verdict but never throw.
|
|
1498
|
+
* - `'off'` — skip the guard entirely (only for offline/replay analysis).
|
|
1499
|
+
*/
|
|
1500
|
+
integrity?: 'assert' | 'warn' | 'off';
|
|
1501
|
+
/** Forwarded to `assertRealBackend`. Default true (tolerate partial 429
|
|
1502
|
+
* cascades); set false for strict CI gates. */
|
|
1503
|
+
allowMixed?: boolean;
|
|
1504
|
+
/** Max concurrent cells WITHIN each profile's campaign. Default 2.
|
|
1505
|
+
* Profiles run sequentially so the cost ceiling is honored deterministically. */
|
|
1506
|
+
maxConcurrency?: number;
|
|
1507
|
+
/** Cumulative USD cap per profile campaign. */
|
|
1508
|
+
costCeiling?: number;
|
|
1509
|
+
/** Capture flywheel — forwarded to each campaign. */
|
|
1510
|
+
labeledStore?: LabeledScenarioStore | 'off';
|
|
1511
|
+
captureSource?: LabeledScenarioSource;
|
|
1512
|
+
/** Storage backend. Default `fsCampaignStorage`. Pass
|
|
1513
|
+
* `inMemoryCampaignStorage()` for edge/CF-Worker/test runs. */
|
|
1514
|
+
storage?: CampaignStorage;
|
|
1515
|
+
/** Test seam — override the wall clock. */
|
|
1516
|
+
now?: () => Date;
|
|
1517
|
+
/** Optional persona key per scenario — drives the `byPersona` pivot. When
|
|
1518
|
+
* unset, `byPersona` is omitted. */
|
|
1519
|
+
personaOf?: (scenario: TScenario) => string;
|
|
1520
|
+
/** Validate every produced RunRecord with `validateRunRecord` (fail-loud).
|
|
1521
|
+
* Default true — catches bad model snapshots and non-finite judge dims at
|
|
1522
|
+
* the boundary instead of letting them poison downstream analysis. */
|
|
1523
|
+
validate?: boolean;
|
|
1524
|
+
/** Corpus-by-default: derive the trajectory text (`prompt` + `completion`)
|
|
1525
|
+
* for each cell from its artifact + scenario. When set, every produced
|
|
1526
|
+
* record carries `prompt`/`completion` (a `CorpusRecord`) so the run's
|
|
1527
|
+
* graded trajectories can be appended to the durable RL corpus with no
|
|
1528
|
+
* side-channel — `appendToCorpus(result.records, path)`. Fail-soft: a
|
|
1529
|
+
* throwing or undefined-returning extractor just omits the text. */
|
|
1530
|
+
corpusText?: (artifact: TArtifact, scenario: TScenario) => {
|
|
1531
|
+
prompt: string;
|
|
1532
|
+
completion: string;
|
|
1533
|
+
} | undefined;
|
|
1534
|
+
}
|
|
1535
|
+
interface ProfileSummary {
|
|
1536
|
+
profileId: string;
|
|
1537
|
+
profileHash: string;
|
|
1538
|
+
model: string;
|
|
1539
|
+
/** RunRecords produced for this profile (= scenarios × reps). */
|
|
1540
|
+
records: number;
|
|
1541
|
+
/** Mean across scored records, or null when the profile has no task labels. */
|
|
1542
|
+
meanComposite: number | null;
|
|
1543
|
+
totalCostUsd: number;
|
|
1544
|
+
/** Per-profile integrity verdict — surfaces a single profile that ran stub
|
|
1545
|
+
* even when the matrix as a whole looks real. */
|
|
1546
|
+
integrity: BackendIntegrityReport;
|
|
1547
|
+
}
|
|
1548
|
+
interface ScenarioRollup {
|
|
1549
|
+
meanComposite: number;
|
|
1550
|
+
n: number;
|
|
1551
|
+
}
|
|
1552
|
+
interface RunProfileMatrixResult<TArtifact, TScenario extends Scenario> {
|
|
1553
|
+
matrixId: string;
|
|
1554
|
+
experimentId: string;
|
|
1555
|
+
/** One RunRecord per (profile, scenario, rep) cell — the integrity-checked,
|
|
1556
|
+
* paper-grade output. Feed straight into `analyzeRuns`, `HeldOutGate`,
|
|
1557
|
+
* scorecards, the hosted wire format. */
|
|
1558
|
+
records: RunRecord[];
|
|
1559
|
+
byProfile: Record<string, ProfileSummary>;
|
|
1560
|
+
byScenario: Record<string, ScenarioRollup>;
|
|
1561
|
+
/** Present only when `personaOf` was supplied. */
|
|
1562
|
+
byPersona?: Record<string, ScenarioRollup>;
|
|
1563
|
+
/** Whole-matrix integrity report (the one `integrity:'assert'` enforces). */
|
|
1564
|
+
integrity: BackendIntegrityReport;
|
|
1565
|
+
/** The raw per-profile campaign results, keyed by profile id. */
|
|
1566
|
+
campaigns: Record<string, CampaignResult<TArtifact, TScenario>>;
|
|
1567
|
+
}
|
|
1568
|
+
/**
|
|
1569
|
+
* Profile × scenario matrix runner: fan N agent profiles across M scenarios, project each cell to a validated `RunRecord` with real token usage, and enforce the backend-integrity guard before returning.
|
|
1570
|
+
*/
|
|
1571
|
+
declare function runProfileMatrix<TScenario extends Scenario, TArtifact>(opts: RunProfileMatrixOptions<TScenario, TArtifact>): Promise<RunProfileMatrixResult<TArtifact, TScenario>>;
|
|
1572
|
+
//#endregion
|
|
1573
|
+
//#region src/campaign/presets/playback.d.ts
|
|
1574
|
+
/** One step of a user story — what the user does. The driver interprets
|
|
1575
|
+
* `payload` (a Playwright selector + action, or a sandbox chat turn). */
|
|
1576
|
+
interface PlaybackStep {
|
|
1577
|
+
/** Human-readable action, captured verbatim in the UX narrative. */
|
|
1578
|
+
action: string;
|
|
1579
|
+
/** Driver-specific payload (e.g. `{ selector, fill }` or `{ turn }`). */
|
|
1580
|
+
payload?: Record<string, unknown>;
|
|
1581
|
+
}
|
|
1582
|
+
/**
|
|
1583
|
+
* A user story = a runnable product journey plus the requirements that define
|
|
1584
|
+
* "this story works". Each requirement is one Jira ticket line. Extends
|
|
1585
|
+
* `Scenario` so a catalog drops straight into `runProfileMatrix({ scenarios })`.
|
|
1586
|
+
*/
|
|
1587
|
+
interface UserStory extends Scenario {
|
|
1588
|
+
/** Human-readable story title (the ticket headline). */
|
|
1589
|
+
title: string;
|
|
1590
|
+
/** Ordered steps the driver executes. */
|
|
1591
|
+
steps: PlaybackStep[];
|
|
1592
|
+
/** What must hold in the produced state for the story to pass. */
|
|
1593
|
+
requirements: CompletionRequirement[];
|
|
1594
|
+
}
|
|
1595
|
+
/** Dispatch context plus the profile under test (which cheap model, etc.). */
|
|
1596
|
+
interface PlaybackContext extends DispatchContext {
|
|
1597
|
+
profile: AgentProfile;
|
|
1598
|
+
}
|
|
1599
|
+
/**
|
|
1600
|
+
* Drives the real product through a story and returns the runtime event stream
|
|
1601
|
+
* `extractProducedState` consumes. Implemented by CONSUMERS —
|
|
1602
|
+
* `SandboxPlaybackDriver` (real API / sandbox workspace) and
|
|
1603
|
+
* `PlaywrightPlaybackDriver` (real UI) — because they depend on runtime /
|
|
1604
|
+
* browser infra the substrate must not import. The driver MUST report LLM
|
|
1605
|
+
* usage through `ctx.cost.runPaidCall` so the backend-integrity check sees real
|
|
1606
|
+
* tokens (a run that never reports tokens reads as a stub).
|
|
1607
|
+
*/
|
|
1608
|
+
interface PlaybackDriver<TStory extends UserStory = UserStory> {
|
|
1609
|
+
run(story: TStory, ctx: PlaybackContext): Promise<readonly RuntimeEventLike[]>;
|
|
1610
|
+
}
|
|
1611
|
+
/**
|
|
1612
|
+
* Adapt a `PlaybackDriver` into a `runProfileMatrix` dispatch. The artifact the
|
|
1613
|
+
* matrix scores is the `ProducedState` extracted from the driver's event
|
|
1614
|
+
* stream — grade it with `scoreUserStory` (or a judge wrapping it).
|
|
1615
|
+
*/
|
|
1616
|
+
declare function makePlaybackDispatch<TStory extends UserStory>(driver: PlaybackDriver<TStory>): ProfileDispatchFn<TStory, ProducedState>;
|
|
1617
|
+
/** A scored user story — the completion verdict plus its human title. */
|
|
1618
|
+
interface UserStoryVerdict extends CompletionVerdict {
|
|
1619
|
+
title: string;
|
|
1620
|
+
}
|
|
1621
|
+
/**
|
|
1622
|
+
* Score one story's produced state against its requirements. Thin wrapper over
|
|
1623
|
+
* `verifyCompletion` that builds the gold from the story and returns a
|
|
1624
|
+
* per-requirement PASS/FAIL verdict. `checkCorrectness` is injected — a
|
|
1625
|
+
* deterministic stub in tests, `createLlmCorrectnessChecker` in production.
|
|
1626
|
+
*/
|
|
1627
|
+
declare function scoreUserStory(story: UserStory, state: ProducedState, checkCorrectness: CorrectnessChecker): Promise<UserStoryVerdict>;
|
|
1628
|
+
/** One row of the launch scoreboard — story × requirement → PASS/FAIL. */
|
|
1629
|
+
interface ScoreboardRow {
|
|
1630
|
+
storyId: string;
|
|
1631
|
+
storyTitle: string;
|
|
1632
|
+
reqId: string;
|
|
1633
|
+
reqTitle: string;
|
|
1634
|
+
status: 'PASS' | 'FAIL';
|
|
1635
|
+
evidence: string[];
|
|
1636
|
+
}
|
|
1637
|
+
/**
|
|
1638
|
+
* Flatten story verdicts into the per-requirement scoreboard — the literal
|
|
1639
|
+
* Jira tick-off: one row per (story, requirement) with PASS/FAIL and the
|
|
1640
|
+
* evidence behind the verdict.
|
|
1641
|
+
*/
|
|
1642
|
+
declare function userStoryScoreboard(verdicts: readonly UserStoryVerdict[]): ScoreboardRow[];
|
|
1643
|
+
/** Launch-readiness headline counts rolled up from the per-requirement rows. */
|
|
1644
|
+
interface ScoreboardSummary {
|
|
1645
|
+
/** Distinct user stories on the board. */
|
|
1646
|
+
stories: number;
|
|
1647
|
+
/** Stories whose every requirement passed. */
|
|
1648
|
+
storiesFullyComplete: number;
|
|
1649
|
+
/** Total (story, requirement) rows. */
|
|
1650
|
+
requirements: number;
|
|
1651
|
+
/** Rows with status PASS. */
|
|
1652
|
+
passed: number;
|
|
1653
|
+
/** Rows with status FAIL. */
|
|
1654
|
+
failed: number;
|
|
1655
|
+
/** passed / requirements; 0 when there are no rows. */
|
|
1656
|
+
passRate: number;
|
|
1657
|
+
}
|
|
1658
|
+
/** Roll the per-requirement rows up into the launch headline counts. */
|
|
1659
|
+
declare function scoreboardSummary(rows: readonly ScoreboardRow[]): ScoreboardSummary;
|
|
1660
|
+
interface ScoreboardRenderOptions {
|
|
1661
|
+
/** Document H1. Defaults to a generic playback title. */
|
|
1662
|
+
title?: string;
|
|
1663
|
+
/** Key/value run metadata rendered under the headline (runId, backend, model, date). */
|
|
1664
|
+
meta?: Record<string, string>;
|
|
1665
|
+
/** Max chars of joined evidence shown per row. Default 160. */
|
|
1666
|
+
maxEvidenceChars?: number;
|
|
1667
|
+
}
|
|
1668
|
+
/**
|
|
1669
|
+
* Render the scoreboard as a launch-readiness Markdown document — the literal
|
|
1670
|
+
* "tick off every user story" artifact: a headline roll-up, the open tickets
|
|
1671
|
+
* (FAIL rows) up top as the launch blockers, then a per-story table of
|
|
1672
|
+
* requirement → PASS/FAIL with the evidence behind each verdict. Pure: same
|
|
1673
|
+
* rows in, same bytes out (no clock/random), so it is safe to snapshot.
|
|
1674
|
+
*/
|
|
1675
|
+
declare function renderScoreboardMarkdown(rows: readonly ScoreboardRow[], opts?: ScoreboardRenderOptions): string;
|
|
1676
|
+
//#endregion
|
|
1677
|
+
//#region src/campaign/run-dir.d.ts
|
|
1678
|
+
/** The shared, out-of-repo root for campaign/benchmark run bundles. Keeping run
|
|
1679
|
+
* outputs here means they never land in a repo working tree (no per-repo
|
|
1680
|
+
* gitignore, no clutter, no accidental commits). Layout:
|
|
1681
|
+
* ~/.tangle/traces/<repo>/runs/<runName>/
|
|
1682
|
+
* where <repo> disambiguates runs across repos in one place. */
|
|
1683
|
+
declare function tangleTracesRoot(): string;
|
|
1684
|
+
/** Resolve a campaign `runDir`. An absolute path is honored as-is (the caller
|
|
1685
|
+
* chose an explicit location). A bare name is placed under the shared home root
|
|
1686
|
+
* so bundles never pollute a repo working tree — the default the harness should
|
|
1687
|
+
* compute so callers pass a *name*, not a path. */
|
|
1688
|
+
declare function resolveRunDir(runDir: string, repo?: string): string;
|
|
1689
|
+
//#endregion
|
|
1690
|
+
//#region src/campaign/scenario-selection.d.ts
|
|
1691
|
+
/**
|
|
1692
|
+
* Discriminative scenario selection (research claim E2).
|
|
1693
|
+
*
|
|
1694
|
+
* The OR benchmark is SATURATING: run 7 measured ~75% tied holdout cells — most
|
|
1695
|
+
* problems are solved optimally by the baseline AND every candidate, so those
|
|
1696
|
+
* paired cells carry zero signal. A random/balanced holdout split spends its
|
|
1697
|
+
* budget on scenarios that cannot separate candidates.
|
|
1698
|
+
*
|
|
1699
|
+
* This picks the holdout by DISCRIMINATION power instead: a scenario every
|
|
1700
|
+
* candidate scores identically (variance ~0) carries no signal; one where the
|
|
1701
|
+
* scores spread carries the most. We drop fully saturated ties so each paired
|
|
1702
|
+
* holdout cell is spent on a scenario that can actually move a verdict.
|
|
1703
|
+
*/
|
|
1704
|
+
/** Per-scenario observation: the composite scores each candidate earned on it. */
|
|
1705
|
+
interface ScenarioSignal {
|
|
1706
|
+
scenarioId: string;
|
|
1707
|
+
/** Per-candidate composite scores observed for this scenario (>=1 values). */
|
|
1708
|
+
scores: number[];
|
|
1709
|
+
}
|
|
1710
|
+
interface DiscriminationScore {
|
|
1711
|
+
scenarioId: string;
|
|
1712
|
+
/** Higher = separates candidates more (spread of their scores). */
|
|
1713
|
+
discrimination: number;
|
|
1714
|
+
/** Higher = easier / more-saturated (mean of candidate scores). */
|
|
1715
|
+
meanScore: number;
|
|
1716
|
+
variance: number;
|
|
1717
|
+
/** variance ~0 AND meanScore at/above the ceiling ⇒ a saturated tie, no signal. */
|
|
1718
|
+
tied: boolean;
|
|
1719
|
+
}
|
|
1720
|
+
/**
|
|
1721
|
+
* Rank scenarios by how well they DISCRIMINATE candidates.
|
|
1722
|
+
*
|
|
1723
|
+
* `discrimination = variance` (spread of the candidate scores) — kept simple on
|
|
1724
|
+
* purpose; the headroom term (`saturationCeiling - meanScore`) only breaks ties
|
|
1725
|
+
* so that, among equally spread scenarios, the one with more room to improve
|
|
1726
|
+
* ranks first. Returned sorted by the deterministic order above.
|
|
1727
|
+
*/
|
|
1728
|
+
declare function scoreDiscrimination(signals: ScenarioSignal[], opts?: {
|
|
1729
|
+
saturationCeiling?: number;
|
|
1730
|
+
}): DiscriminationScore[];
|
|
1731
|
+
/**
|
|
1732
|
+
* Select the top-`k` most discriminative scenario ids for a holdout, EXCLUDING
|
|
1733
|
+
* fully saturated ties when enough non-tied scenarios exist (a tie in the
|
|
1734
|
+
* holdout wastes a paired cell).
|
|
1735
|
+
*
|
|
1736
|
+
* Prefers non-tied scenarios; if fewer than `k` non-tied exist, fills with the
|
|
1737
|
+
* least-saturated tied ones (tied scenarios are already ordered least-saturated
|
|
1738
|
+
* first by `meanScore` asc). Deterministic. Throws if `k < 1`. If
|
|
1739
|
+
* `signals.length <= k`, returns all ids in discrimination order.
|
|
1740
|
+
*/
|
|
1741
|
+
declare function selectDiscriminative(signals: ScenarioSignal[], k: number, opts?: {
|
|
1742
|
+
saturationCeiling?: number;
|
|
1743
|
+
}): string[];
|
|
1744
|
+
//#endregion
|
|
1745
|
+
//#region src/campaign/score-utils.d.ts
|
|
1746
|
+
/** Mean composite across cells with complete task-quality evidence.
|
|
1747
|
+
* Partial judge results remain on their cells but never enter this value.
|
|
1748
|
+
* A campaign with no complete score has no numeric mean and fails loudly. */
|
|
1749
|
+
declare function campaignMeanComposite<TArtifact, TScenario extends Scenario>(campaign: CampaignResult<TArtifact, TScenario>): number;
|
|
1750
|
+
/** Compare fixed-length lexicographic rank keys where each element is higher-is-better.
|
|
1751
|
+
* Returns a positive number when `a` ranks above `b`, negative when below, and
|
|
1752
|
+
* zero when equal. */
|
|
1753
|
+
declare function compareRankKeys(a: readonly number[], b: readonly number[]): number;
|
|
1754
|
+
interface CampaignBreakdown {
|
|
1755
|
+
/** Mean score per judge dimension across all cells. */
|
|
1756
|
+
dimensions: Record<string, number>;
|
|
1757
|
+
/** Per-scenario composite (mean over reps + judges) + the judge's free-form
|
|
1758
|
+
* `notes` for that scenario (the "why" a reflective proposer grounds on) +
|
|
1759
|
+
* an optional `emitted` excerpt of the candidate's raw output (the "what it
|
|
1760
|
+
* actually did" a reflective proposer grounds on). */
|
|
1761
|
+
scenarios: Array<{
|
|
1762
|
+
scenarioId: string;
|
|
1763
|
+
composite: number;
|
|
1764
|
+
notes?: string;
|
|
1765
|
+
emitted?: string;
|
|
1766
|
+
}>;
|
|
1767
|
+
}
|
|
1768
|
+
/** Per-candidate evidence a reflective/patch proposer grounds its next proposal
|
|
1769
|
+
* on: mean score per judge dimension + per-scenario composite. */
|
|
1770
|
+
declare function campaignBreakdown<TArtifact, TScenario extends Scenario>(campaign: CampaignResult<TArtifact, TScenario>): CampaignBreakdown;
|
|
1771
|
+
//#endregion
|
|
1772
|
+
//#region src/campaign/search-ledger-errors.d.ts
|
|
1773
|
+
declare class SearchLedgerError extends ValidationError {}
|
|
1774
|
+
declare class SearchLedgerIntegrityError extends SearchLedgerError {}
|
|
1775
|
+
declare class SearchLedgerConflictError extends SearchLedgerError {}
|
|
1776
|
+
//#endregion
|
|
1777
|
+
//#region src/campaign/search-ledger.d.ts
|
|
1778
|
+
declare const SEARCH_LEDGER_SCHEMA: 'tangle.search-ledger.v1';
|
|
1779
|
+
type SearchLedgerHash = LedgerHash;
|
|
1780
|
+
type SearchSurfaceKind = 'prompt' | 'tool-contract' | 'runtime-config' | 'memory' | 'knowledge' | 'agent-profile' | 'code' | 'deployment';
|
|
1781
|
+
/** Content-addressed artifact or receipt. Mutable paths are locators only; the
|
|
1782
|
+
* digest and byte length bind the exact bytes used by the search. */
|
|
1783
|
+
interface SearchArtifactRef {
|
|
1784
|
+
role: string;
|
|
1785
|
+
uri: string;
|
|
1786
|
+
sha256: SearchLedgerHash;
|
|
1787
|
+
byteLength: number;
|
|
1788
|
+
}
|
|
1789
|
+
/** Repository, dataset, or package source pinned to an immutable commit or
|
|
1790
|
+
* content digest. Branches, tags, and bare package versions are rejected. */
|
|
1791
|
+
interface SearchSourceRef {
|
|
1792
|
+
uri: string;
|
|
1793
|
+
revision: string;
|
|
1794
|
+
}
|
|
1795
|
+
interface SearchModelIdentity {
|
|
1796
|
+
provider: string;
|
|
1797
|
+
snapshot: string;
|
|
1798
|
+
}
|
|
1799
|
+
interface SearchCandidateSurface {
|
|
1800
|
+
surfaceId: string;
|
|
1801
|
+
kind: SearchSurfaceKind;
|
|
1802
|
+
artifact: SearchArtifactRef;
|
|
1803
|
+
}
|
|
1804
|
+
interface SearchCandidateLineage {
|
|
1805
|
+
/** Existing `LineageNode.id`; this ledger references rather than embeds it. */
|
|
1806
|
+
lineageNodeId: string;
|
|
1807
|
+
parentCandidateIds: string[];
|
|
1808
|
+
generation: number;
|
|
1809
|
+
proposer: string;
|
|
1810
|
+
proposerSource: SearchSourceRef;
|
|
1811
|
+
}
|
|
1812
|
+
type SearchOperationKind = 'candidate-generation' | 'analysis' | 'selection' | 'judge' | 'other';
|
|
1813
|
+
interface SearchPlannedTask {
|
|
1814
|
+
taskId: string;
|
|
1815
|
+
source: SearchSourceRef;
|
|
1816
|
+
benchmark: SearchSourceRef;
|
|
1817
|
+
/** Maximum transport attempts for this task and candidate. Only an explicit
|
|
1818
|
+
* passed/failed outcome satisfies the planned denominator. */
|
|
1819
|
+
maxAttempts: number;
|
|
1820
|
+
}
|
|
1821
|
+
interface SearchPlannedOperation {
|
|
1822
|
+
operationId: string;
|
|
1823
|
+
kind: SearchOperationKind;
|
|
1824
|
+
}
|
|
1825
|
+
interface SearchCandidateSlot {
|
|
1826
|
+
slotId: string;
|
|
1827
|
+
/** Planned candidate-generation call that must either produce this slot or
|
|
1828
|
+
* fail before the slot can be closed. Several slots may share one batched call. */
|
|
1829
|
+
generationOperationId: string;
|
|
1830
|
+
}
|
|
1831
|
+
interface SearchPlan {
|
|
1832
|
+
/** Stable slots and their proposer calls are frozen before search begins. */
|
|
1833
|
+
candidateSlots: SearchCandidateSlot[];
|
|
1834
|
+
/** Every task applies to every successfully registered candidate. */
|
|
1835
|
+
tasks: SearchPlannedTask[];
|
|
1836
|
+
/** Non-task spend slots: proposal, analysis, selection, extra judges, etc. */
|
|
1837
|
+
operations: SearchPlannedOperation[];
|
|
1838
|
+
}
|
|
1839
|
+
type SearchTokenAccounting = {
|
|
1840
|
+
status: 'known';
|
|
1841
|
+
inputTokens: number;
|
|
1842
|
+
outputTokens: number;
|
|
1843
|
+
cachedTokens: number;
|
|
1844
|
+
} | {
|
|
1845
|
+
status: 'unknown';
|
|
1846
|
+
reason: string;
|
|
1847
|
+
};
|
|
1848
|
+
type SearchCostAccounting = {
|
|
1849
|
+
status: 'known';
|
|
1850
|
+
usd: number;
|
|
1851
|
+
source: 'provider' | 'pricing-table' | 'free';
|
|
1852
|
+
} | {
|
|
1853
|
+
status: 'unknown';
|
|
1854
|
+
/** Known spend may still be a lower bound when one call was unpriced. */
|
|
1855
|
+
knownLowerBoundUsd: number;
|
|
1856
|
+
reason: string;
|
|
1857
|
+
};
|
|
1858
|
+
interface SearchAttemptAccounting {
|
|
1859
|
+
tokens: SearchTokenAccounting;
|
|
1860
|
+
cost: SearchCostAccounting;
|
|
1861
|
+
}
|
|
1862
|
+
interface SearchFailureReason {
|
|
1863
|
+
code: string;
|
|
1864
|
+
message: string;
|
|
1865
|
+
}
|
|
1866
|
+
type SearchTaskOutcome = {
|
|
1867
|
+
status: 'passed';
|
|
1868
|
+
score: number;
|
|
1869
|
+
metrics: Record<string, number>;
|
|
1870
|
+
} | {
|
|
1871
|
+
status: 'failed';
|
|
1872
|
+
score: number;
|
|
1873
|
+
metrics: Record<string, number>;
|
|
1874
|
+
failure: SearchFailureReason;
|
|
1875
|
+
} | {
|
|
1876
|
+
status: 'errored';
|
|
1877
|
+
metrics: Record<string, number>;
|
|
1878
|
+
error: SearchFailureReason & {
|
|
1879
|
+
retryable: boolean;
|
|
1880
|
+
};
|
|
1881
|
+
};
|
|
1882
|
+
type SearchSurfaceEffect = {
|
|
1883
|
+
status: 'measured';
|
|
1884
|
+
metric: string;
|
|
1885
|
+
baselineValue: number;
|
|
1886
|
+
candidateValue: number;
|
|
1887
|
+
delta: number;
|
|
1888
|
+
} | {
|
|
1889
|
+
status: 'not-measured';
|
|
1890
|
+
reason: string;
|
|
1891
|
+
};
|
|
1892
|
+
/** Per-attempt proof that a declared candidate surface was or was not active,
|
|
1893
|
+
* plus measured effect when the experiment supports attribution. */
|
|
1894
|
+
interface SearchSurfaceEvidence {
|
|
1895
|
+
surfaceId: string;
|
|
1896
|
+
fired: boolean;
|
|
1897
|
+
firingCount: number;
|
|
1898
|
+
effect: SearchSurfaceEffect;
|
|
1899
|
+
evidence: SearchArtifactRef[];
|
|
1900
|
+
}
|
|
1901
|
+
interface SearchLedgerEventBase {
|
|
1902
|
+
eventId: string;
|
|
1903
|
+
occurredAt: string;
|
|
1904
|
+
artifacts: SearchArtifactRef[];
|
|
1905
|
+
}
|
|
1906
|
+
interface SearchPlannedEvent extends SearchLedgerEventBase {
|
|
1907
|
+
kind: 'search-planned';
|
|
1908
|
+
plan: SearchPlan;
|
|
1909
|
+
}
|
|
1910
|
+
interface SearchCandidateRegisteredEvent extends SearchLedgerEventBase {
|
|
1911
|
+
kind: 'candidate-registered';
|
|
1912
|
+
slotId: string;
|
|
1913
|
+
generationOperationId: string;
|
|
1914
|
+
candidateId: string;
|
|
1915
|
+
lineage: SearchCandidateLineage;
|
|
1916
|
+
surfaces: SearchCandidateSurface[];
|
|
1917
|
+
}
|
|
1918
|
+
interface SearchCandidateSlotClosedEvent extends SearchLedgerEventBase {
|
|
1919
|
+
kind: 'candidate-slot-closed';
|
|
1920
|
+
slotId: string;
|
|
1921
|
+
generationOperationId: string;
|
|
1922
|
+
reason: SearchFailureReason;
|
|
1923
|
+
}
|
|
1924
|
+
interface SearchTaskAttemptedEvent extends SearchLedgerEventBase {
|
|
1925
|
+
kind: 'task-attempted';
|
|
1926
|
+
candidateId: string;
|
|
1927
|
+
runId: string;
|
|
1928
|
+
attemptIndex: number;
|
|
1929
|
+
task: {
|
|
1930
|
+
taskId: string;
|
|
1931
|
+
source: SearchSourceRef;
|
|
1932
|
+
};
|
|
1933
|
+
identity: {
|
|
1934
|
+
model: SearchModelIdentity;
|
|
1935
|
+
agent: SearchSourceRef;
|
|
1936
|
+
benchmark: SearchSourceRef;
|
|
1937
|
+
};
|
|
1938
|
+
outcome: SearchTaskOutcome;
|
|
1939
|
+
accounting: SearchAttemptAccounting;
|
|
1940
|
+
surfaceEvidence: SearchSurfaceEvidence[];
|
|
1941
|
+
}
|
|
1942
|
+
interface SearchOperationRecordedEvent extends SearchLedgerEventBase {
|
|
1943
|
+
kind: 'search-operation-recorded';
|
|
1944
|
+
operationId: string;
|
|
1945
|
+
operationKind: SearchOperationKind;
|
|
1946
|
+
execution: {
|
|
1947
|
+
kind: 'model';
|
|
1948
|
+
model: SearchModelIdentity;
|
|
1949
|
+
source: SearchSourceRef;
|
|
1950
|
+
} | {
|
|
1951
|
+
kind: 'deterministic';
|
|
1952
|
+
source: SearchSourceRef;
|
|
1953
|
+
};
|
|
1954
|
+
outcome: {
|
|
1955
|
+
status: 'completed';
|
|
1956
|
+
} | {
|
|
1957
|
+
status: 'partial';
|
|
1958
|
+
failure: SearchFailureReason;
|
|
1959
|
+
} | {
|
|
1960
|
+
status: 'failed';
|
|
1961
|
+
failure: SearchFailureReason;
|
|
1962
|
+
};
|
|
1963
|
+
accounting: SearchAttemptAccounting;
|
|
1964
|
+
}
|
|
1965
|
+
interface SearchCandidateDecidedEvent extends SearchLedgerEventBase {
|
|
1966
|
+
kind: 'candidate-decided';
|
|
1967
|
+
candidateId: string;
|
|
1968
|
+
decision: {
|
|
1969
|
+
status: 'selected';
|
|
1970
|
+
} | {
|
|
1971
|
+
status: 'rejected';
|
|
1972
|
+
reason: SearchFailureReason;
|
|
1973
|
+
};
|
|
1974
|
+
}
|
|
1975
|
+
interface SearchCompletedEvent extends SearchLedgerEventBase {
|
|
1976
|
+
kind: 'search-completed';
|
|
1977
|
+
result: {
|
|
1978
|
+
status: 'selected';
|
|
1979
|
+
candidateId: string;
|
|
1980
|
+
} | {
|
|
1981
|
+
status: 'all-rejected';
|
|
1982
|
+
reason: SearchFailureReason;
|
|
1983
|
+
};
|
|
1984
|
+
}
|
|
1985
|
+
type SearchLedgerEvent = SearchPlannedEvent | SearchCandidateRegisteredEvent | SearchCandidateSlotClosedEvent | SearchTaskAttemptedEvent | SearchOperationRecordedEvent | SearchCandidateDecidedEvent | SearchCompletedEvent;
|
|
1986
|
+
interface SearchLedgerEntry {
|
|
1987
|
+
schema: typeof SEARCH_LEDGER_SCHEMA;
|
|
1988
|
+
campaignId: string;
|
|
1989
|
+
sequence: number;
|
|
1990
|
+
previousHash: SearchLedgerHash | null;
|
|
1991
|
+
event: SearchLedgerEvent;
|
|
1992
|
+
entryHash: SearchLedgerHash;
|
|
1993
|
+
}
|
|
1994
|
+
type SearchAccountingAudit = {
|
|
1995
|
+
status: 'known';
|
|
1996
|
+
inputTokens: number;
|
|
1997
|
+
outputTokens: number;
|
|
1998
|
+
cachedTokens: number;
|
|
1999
|
+
costUsd: number;
|
|
2000
|
+
} | {
|
|
2001
|
+
status: 'partial';
|
|
2002
|
+
knownInputTokens: number;
|
|
2003
|
+
knownOutputTokens: number;
|
|
2004
|
+
knownCachedTokens: number;
|
|
2005
|
+
knownCostUsd: number;
|
|
2006
|
+
unknownTokenEventIds: string[];
|
|
2007
|
+
unknownCostEventIds: string[];
|
|
2008
|
+
};
|
|
2009
|
+
interface SearchLedgerAudit {
|
|
2010
|
+
campaignId: string;
|
|
2011
|
+
eventCount: number;
|
|
2012
|
+
candidateCount: number;
|
|
2013
|
+
closedCandidateSlotCount: number;
|
|
2014
|
+
attemptCount: number;
|
|
2015
|
+
operationCount: number;
|
|
2016
|
+
outcomes: {
|
|
2017
|
+
passed: number;
|
|
2018
|
+
failed: number;
|
|
2019
|
+
errored: number;
|
|
2020
|
+
};
|
|
2021
|
+
operationOutcomes: {
|
|
2022
|
+
completed: number;
|
|
2023
|
+
partial: number;
|
|
2024
|
+
failed: number;
|
|
2025
|
+
};
|
|
2026
|
+
decisions: {
|
|
2027
|
+
selected: number;
|
|
2028
|
+
rejected: number;
|
|
2029
|
+
pending: number;
|
|
2030
|
+
};
|
|
2031
|
+
expected: {
|
|
2032
|
+
candidateSlots: number;
|
|
2033
|
+
taskOutcomes: number;
|
|
2034
|
+
operations: number;
|
|
2035
|
+
missingCandidateSlots: string[];
|
|
2036
|
+
missingTaskOutcomes: string[];
|
|
2037
|
+
missingOperations: string[];
|
|
2038
|
+
};
|
|
2039
|
+
status: 'in-progress' | 'selected' | 'all-rejected';
|
|
2040
|
+
selectedCandidateId: string | null;
|
|
2041
|
+
accounting: SearchAccountingAudit;
|
|
2042
|
+
headHash: SearchLedgerHash | null;
|
|
2043
|
+
}
|
|
2044
|
+
interface SearchLedgerReplay {
|
|
2045
|
+
entries: SearchLedgerEntry[];
|
|
2046
|
+
plan: SearchPlannedEvent | null;
|
|
2047
|
+
candidates: SearchCandidateRegisteredEvent[];
|
|
2048
|
+
closedCandidateSlots: SearchCandidateSlotClosedEvent[];
|
|
2049
|
+
attempts: SearchTaskAttemptedEvent[];
|
|
2050
|
+
operations: SearchOperationRecordedEvent[];
|
|
2051
|
+
decisions: SearchCandidateDecidedEvent[];
|
|
2052
|
+
completion: SearchCompletedEvent | null;
|
|
2053
|
+
audit: SearchLedgerAudit;
|
|
2054
|
+
}
|
|
2055
|
+
interface SearchLedgerAppendResult {
|
|
2056
|
+
entry: SearchLedgerEntry;
|
|
2057
|
+
/** False when the exact event was already durably present. */
|
|
2058
|
+
appended: boolean;
|
|
2059
|
+
replay: SearchLedgerReplay;
|
|
2060
|
+
}
|
|
2061
|
+
/** Validate and return a canonical copy. Arrays whose order is not semantic are
|
|
2062
|
+
* sorted so retries from different processes produce byte-identical events. */
|
|
2063
|
+
declare function validateSearchLedgerEvent(input: unknown): SearchLedgerEvent;
|
|
2064
|
+
interface OpenSearchLedgerOptions {
|
|
2065
|
+
path: string;
|
|
2066
|
+
campaignId: string;
|
|
2067
|
+
}
|
|
2068
|
+
interface SearchLedger {
|
|
2069
|
+
readonly path: string;
|
|
2070
|
+
readonly campaignId: string;
|
|
2071
|
+
append(event: SearchLedgerEvent): Promise<SearchLedgerAppendResult>;
|
|
2072
|
+
replay(): Promise<SearchLedgerReplay>;
|
|
2073
|
+
}
|
|
2074
|
+
/** Open a durable filesystem search ledger. Construction performs no I/O; the
|
|
2075
|
+
* first `append` or `replay` validates the complete existing file. */
|
|
2076
|
+
declare function openSearchLedger(options: OpenSearchLedgerOptions): SearchLedger;
|
|
2077
|
+
declare class FileSearchLedger implements SearchLedger {
|
|
2078
|
+
readonly path: string;
|
|
2079
|
+
readonly campaignId: string;
|
|
2080
|
+
private readonly journal;
|
|
2081
|
+
constructor(path: string, campaignId: string);
|
|
2082
|
+
replay(): Promise<SearchLedgerReplay>;
|
|
2083
|
+
append(input: SearchLedgerEvent): Promise<SearchLedgerAppendResult>;
|
|
2084
|
+
}
|
|
2085
|
+
//#endregion
|
|
2086
|
+
//#region src/campaign/single-run-lock.d.ts
|
|
2087
|
+
/**
|
|
2088
|
+
* Single-run lock for evaluations that share one mutable environment.
|
|
2089
|
+
*
|
|
2090
|
+
* Two concurrent runs against a shared stateful gym silently corrupt each
|
|
2091
|
+
* other: each resets/mutates environment state mid-cell of the other, and
|
|
2092
|
+
* every score from both becomes garbage that LOOKS like worker variance
|
|
2093
|
+
* (agent-lab R357 burned hours on flip-flopping scores before tracing them
|
|
2094
|
+
* to exactly this). The fix is a pid lockfile: refuse to start while a live
|
|
2095
|
+
* holder exists, reclaim stale locks whose pid is gone, release only if the
|
|
2096
|
+
* lock is still ours.
|
|
2097
|
+
*
|
|
2098
|
+
* `alsoCheck` exists because independent runners can guard the same shared
|
|
2099
|
+
* resource with differently named lockfiles; a runner must respect all of
|
|
2100
|
+
* them even though it writes only its own.
|
|
2101
|
+
*/
|
|
2102
|
+
interface SingleRunLockOptions {
|
|
2103
|
+
/** Lockfile this runner writes (and checks). */
|
|
2104
|
+
readonly lockPath: string;
|
|
2105
|
+
/** Other runners' lockfiles guarding the same resource; checked, never written. */
|
|
2106
|
+
readonly alsoCheck?: readonly string[];
|
|
2107
|
+
/** Install a process 'exit' hook that releases the lock. Default true. */
|
|
2108
|
+
readonly releaseOnExit?: boolean;
|
|
2109
|
+
/** Owner pid recorded in the lockfile metadata. Default process.pid. */
|
|
2110
|
+
readonly pid?: number;
|
|
2111
|
+
}
|
|
2112
|
+
interface SingleRunLock {
|
|
2113
|
+
/** Remove the lockfile if this process still owns it. Idempotent. */
|
|
2114
|
+
release(): void;
|
|
2115
|
+
}
|
|
2116
|
+
/**
|
|
2117
|
+
* Acquire the lock or throw naming the live holder. A stale lock (holder pid
|
|
2118
|
+
* no longer running) is reclaimed by one contender. An interrupted reclaim
|
|
2119
|
+
* leaves a marker that fails closed instead of admitting overlapping runs.
|
|
2120
|
+
*/
|
|
2121
|
+
declare function acquireSingleRunLock(opts: SingleRunLockOptions): SingleRunLock;
|
|
2122
|
+
//#endregion
|
|
2123
|
+
//#region src/campaign/surface-identity.d.ts
|
|
2124
|
+
/** Validate the immutable identity shape; the owning executor verifies the Git objects and patch. */
|
|
2125
|
+
declare function assertCodeSurfaceIdentity(surface: unknown): asserts surface is CodeSurface;
|
|
2126
|
+
declare function assertComponentSurface(surface: unknown): asserts surface is ComponentSurface;
|
|
2127
|
+
declare function componentSurfaceIdentityMaterial(surface: ComponentSurface): string;
|
|
2128
|
+
/** Canonical, location-independent identity of a finalized code candidate.
|
|
2129
|
+
* Commit metadata is excluded: two commits with the same base, final tree,
|
|
2130
|
+
* and patch bytes are the same executable candidate. */
|
|
2131
|
+
declare function codeSurfaceIdentityMaterial(surface: CodeSurface): string;
|
|
2132
|
+
/** Full SHA-256 content identity for a prompt or finalized code surface. */
|
|
2133
|
+
declare function surfaceContentHash(surface: MutableSurface): `sha256:${string}`;
|
|
2134
|
+
/** Short loop key derived from the same content identity as provenance. */
|
|
2135
|
+
declare function surfaceHash(surface: MutableSurface): string;
|
|
2136
|
+
/** Canonical customer-visible description of the exact before/after surfaces. */
|
|
2137
|
+
declare function renderSurfaceDiff(winnerSurface: MutableSurface, baselineSurface: MutableSurface): string;
|
|
2138
|
+
//#endregion
|
|
2139
|
+
//#region src/campaign/transient-failure.d.ts
|
|
2140
|
+
/**
|
|
2141
|
+
* Transient-transport-failure classification for dispatch retry policies.
|
|
2142
|
+
*
|
|
2143
|
+
* When an eval cell dies, the harness must decide: retry (the infrastructure
|
|
2144
|
+
* hiccuped - a 502 storm, an admission-queue rejection, a dropped stream) or
|
|
2145
|
+
* score it (the agent genuinely failed). Getting this wrong corrupts results
|
|
2146
|
+
* in both directions: scoring transport hiccups as failures buries real
|
|
2147
|
+
* effects under noise (agent-lab R353 found 5/30 identical repeats were 502s
|
|
2148
|
+
* scored as task failures), while retrying genuine failures silently drops
|
|
2149
|
+
* the hard cells and inflates every arm.
|
|
2150
|
+
*
|
|
2151
|
+
* Full-duration timeouts are the deliberate knob: on saturated shared
|
|
2152
|
+
* infrastructure a timeout usually means the request never got a slot
|
|
2153
|
+
* (retry it), but on unthrottled infrastructure it means the agent flailed
|
|
2154
|
+
* on the task until the clock ran out (a real score-0). Both readings were
|
|
2155
|
+
* needed in practice within one week, so the classifier takes it as an
|
|
2156
|
+
* option instead of hardcoding either.
|
|
2157
|
+
*/
|
|
2158
|
+
interface TransientFailureOptions {
|
|
2159
|
+
/**
|
|
2160
|
+
* Treat full-duration timeouts ("timeout after 180000ms") as transient.
|
|
2161
|
+
* Enable on saturated shared infrastructure where queue starvation eats
|
|
2162
|
+
* the clock; leave off when the agent had the resources and simply failed.
|
|
2163
|
+
* Default false.
|
|
2164
|
+
*/
|
|
2165
|
+
readonly retryFullDurationTimeouts?: boolean;
|
|
2166
|
+
/** Additional caller-specific transient patterns. */
|
|
2167
|
+
readonly extraPatterns?: readonly RegExp[];
|
|
2168
|
+
}
|
|
2169
|
+
/**
|
|
2170
|
+
* True when the error text describes an infrastructure hiccup that should be
|
|
2171
|
+
* retried rather than scored. Empty/undefined input is not transient.
|
|
2172
|
+
*/
|
|
2173
|
+
declare function isTransientTransportFailure(message: string | null | undefined, opts?: TransientFailureOptions): boolean;
|
|
2174
|
+
//#endregion
|
|
2175
|
+
//#region src/campaign/worktree/index.d.ts
|
|
2176
|
+
type GitOutput = string | Uint8Array;
|
|
2177
|
+
type GitEnvironment = Readonly<Record<string, string>>;
|
|
2178
|
+
type GitRunner = (args: string[], cwd: string, env?: GitEnvironment) => GitOutput;
|
|
2179
|
+
interface Worktree {
|
|
2180
|
+
/** Absolute path to the checked-out worktree directory. */
|
|
2181
|
+
readonly path: string;
|
|
2182
|
+
/** The branch the worktree is on (becomes the PR branch on promotion). */
|
|
2183
|
+
readonly branch: string;
|
|
2184
|
+
/** The ref the worktree was forked from. */
|
|
2185
|
+
readonly baseRef: string;
|
|
2186
|
+
/** Exact commit `baseRef` resolved to before the worktree was created. */
|
|
2187
|
+
readonly baseCommit: string;
|
|
2188
|
+
/** Exact tree object for `baseCommit`. */
|
|
2189
|
+
readonly baseTree: string;
|
|
2190
|
+
}
|
|
2191
|
+
interface WorktreeAdapter {
|
|
2192
|
+
/** Create an isolated worktree on a fresh branch off `baseRef`. */
|
|
2193
|
+
create(opts: {
|
|
2194
|
+
baseRef: string;
|
|
2195
|
+
label: string;
|
|
2196
|
+
}): Promise<Worktree>;
|
|
2197
|
+
/** Commit pending changes, freeze the exact Git objects + binary patch, and
|
|
2198
|
+
* verify the worktree still matches that identity. */
|
|
2199
|
+
finalize(worktree: Worktree, summary: string): Promise<CodeSurface>;
|
|
2200
|
+
/** Idempotently remove the worktree and branch. Safe to retry after partial cleanup. */
|
|
2201
|
+
discard(worktree: Worktree): Promise<void>;
|
|
2202
|
+
}
|
|
2203
|
+
/** Typed failure from a `WorktreeAdapter` operation (create/finalize/discard) — wraps the underlying git error as `cause`. */
|
|
2204
|
+
declare class WorktreeAdapterError extends Error {
|
|
2205
|
+
readonly cause?: unknown;
|
|
2206
|
+
constructor(message: string, cause?: unknown);
|
|
2207
|
+
}
|
|
2208
|
+
interface GitWorktreeAdapterOptions {
|
|
2209
|
+
/** Repo root the worktrees fork from. */
|
|
2210
|
+
repoRoot: string;
|
|
2211
|
+
/** Directory worktrees are created under. Default: `<repoRoot>/.worktrees`. */
|
|
2212
|
+
worktreeDir?: string;
|
|
2213
|
+
/** Branch-name prefix. Default: `improve`. */
|
|
2214
|
+
branchPrefix?: string;
|
|
2215
|
+
/** Test seam — defaults to a real `git` runner. The return value must contain
|
|
2216
|
+
* stdout verbatim, and runners that execute Git must forward the optional
|
|
2217
|
+
* environment overrides used to isolate patch generation. */
|
|
2218
|
+
git?: GitRunner;
|
|
2219
|
+
}
|
|
2220
|
+
interface CodeSurfaceVerification {
|
|
2221
|
+
/** Verified worktree path. */
|
|
2222
|
+
path: string;
|
|
2223
|
+
/** Git's canonical root for the verified checkout. */
|
|
2224
|
+
repoRoot: string;
|
|
2225
|
+
/** Recomputed full content identity. */
|
|
2226
|
+
contentHash: `sha256:${string}`;
|
|
2227
|
+
/** Exact verified binary-patch bytes. Candidate-bundle builders encode this
|
|
2228
|
+
* directly instead of reproducing Git diff options. */
|
|
2229
|
+
patchBytes: Uint8Array;
|
|
2230
|
+
}
|
|
2231
|
+
/**
|
|
2232
|
+
* Git-backed `WorktreeAdapter`: creates isolated worktrees on fresh branches, commits agent changes, and discards losers.
|
|
2233
|
+
*/
|
|
2234
|
+
declare function gitWorktreeAdapter(opts: GitWorktreeAdapterOptions): WorktreeAdapter;
|
|
2235
|
+
/** Verify a finalized code surface against its current checkout. This rejects
|
|
2236
|
+
* dirty/ignored files, moved refs, missing Git objects, raw byte/mode
|
|
2237
|
+
* mismatches, external symlinks, and submodules. */
|
|
2238
|
+
declare function verifyCodeSurface(surface: CodeSurface, worktreeDir?: string): CodeSurfaceVerification;
|
|
2239
|
+
/** Resolve a code candidate for evaluation only after verifying its immutable
|
|
2240
|
+
* identity against the checkout at `worktreeRef`. */
|
|
2241
|
+
declare function resolveWorktreePath(surface: CodeSurface, worktreeDir?: string): string;
|
|
2242
|
+
//#endregion
|
|
2243
|
+
export { SearchTaskAttemptedEvent as $, sequentialDecide as $n, CrossSurfaceRankedSingle as $r, ArtifactEventLike as $t, SearchCandidateSlot as A, FsLabeledScenarioStoreOptions as An, CrossSurfaceBootstrapPolicy as Ar, ProfileDispatchFn as At, SearchLedgerHash as B, HeldoutSignificance as Bn, CrossSurfaceEligibility as Br, assertRealBackend as Bt, SEARCH_LEDGER_SCHEMA as C, byteLengthRange as Cn, planEvalFixtureRun as Cr, UserStory as Ct, SearchCandidateDecidedEvent as D, regexMatch as Dn, CrossSurfaceAdditionRejectionReason as Dr, scoreUserStory as Dt, SearchAttemptAccounting as E, jsonHasKeys as En, CrossSurfaceAdditionDecision as Er, renderScoreboardMarkdown as Et, SearchFailureReason as F, ScoredRollout as Fn, CrossSurfaceCandidateSummary as Fr, ScenarioRollup as Ft, SearchPlan as G, heldoutSignificance as Gn, CrossSurfaceInteractionPath as Gr, HARNESS_NATIVE_MODEL as Gt, SearchModelIdentity as H, PairedHoldout as Hn, CrossSurfaceIneligibilityReason as Hr, summarizeBackendIntegrity as Ht, SearchLedger as I, UngroundedLiteralReport as In, CrossSurfaceComponent as Ir, runProfileMatrix as It, SearchPlannedTask as J, SequentialDecideOptions as Jn, CrossSurfaceNaiveStackSelection as Jr, agentProfileHash as Jt, SearchPlannedEvent as K, pairHoldout as Kn, CrossSurfaceInteractionReport as Kr, HarnessType$1 as Kt, SearchLedgerAppendResult as L, classifyUngroundedLiterals as Ln, CrossSurfaceComponentEvidence as Lr, BackendIntegrityError as Lt, SearchCandidateSurface as M, RolloutArgumentDiff as Mn, CrossSurfaceCandidateComparison as Mr, ProfileSummary as Mt, SearchCompletedEvent as N, RolloutArgumentDiffOptions as Nn, CrossSurfaceCandidateEvidence as Nr, RunProfileMatrixOptions as Nt, SearchCandidateLineage as O, neutralizeText as On, CrossSurfaceAttemptCompleteness as Or, scoreboardSummary as Ot, SearchCostAccounting as P, RolloutCall as Pn, CrossSurfaceCandidateOutcome as Pr, RunProfileMatrixResult as Pt, SearchSurfaceKind as Q, SequentialPairedGateOptions as Qn, CrossSurfacePairwiseEntry as Qr, harnessAxisOf as Qt, SearchLedgerEntry as R, rolloutArgumentDiff as Rn, CrossSurfaceCompositionStep as Rr, BackendIntegrityReport as Rt, OpenSearchLedgerOptions as S, failureModeRecallJudge as Si, ValidationResult as Sn, loadEvalFixtureScenarios as Sr, ScoreboardSummary as St, SearchArtifactRef as T, containsAll as Tn, AnalyzeCrossSurfaceInteractionsInput as Tr, makePlaybackDispatch as Tt, SearchOperationKind as U, detectScale as Un, CrossSurfaceInteractionAwareSelection as Ur, AgentProfile$1 as Ut, SearchLedgerReplay as V, HeldoutSignificanceOptions as Vn, CrossSurfaceEvidenceBreakdown as Vr, summarizeAgentReceiptIntegrity as Vt, SearchOperationRecordedEvent as W, dimensionRegressions as Wn, CrossSurfaceInteractionEffect as Wr, CODING_HARNESSES as Wt, SearchSurfaceEffect as X, SequentialObservation as Xn, CrossSurfacePairEvidence as Xr, agentProfileModelId as Xt, SearchSourceRef as Y, SequentialDecision as Yn, CrossSurfacePairCompatibility as Yr, agentProfileId as Yt, SearchSurfaceEvidence as Z, SequentialPairedGate as Zn, CrossSurfacePairIncompatibilityReason as Zr, expandProfileAxes as Zt, surfaceHash as _, AnalystArtifact as _i, verifyCompletion as _n, EvalFixtureValidationMode as _r, PlaybackContext as _t, WorktreeAdapterError as a, MatchedPair as ai, CompletionVerdict as an, canonicalize as ar, SearchLedgerError as at, acquireSingleRunLock as b, FailureModeRecallJudgeOptions as bi, ValidationContext as bn, discoverEvalFixtures as br, ScoreboardRenderOptions as bt, verifyCodeSurface as c, PairArmsResult as ci, ProducedProposal as cn, signManifest as cr, campaignBreakdown as ct, assertCodeSurfaceIdentity as d, PairedArmsComparison as di, SatisfiedBy as dn, neutralizationGate as dr, DiscriminationScore as dt, CrossSurfaceRelativeCost as ei, ProposalEventLike as en, sequentialPairedGate as er, SearchTaskOutcome as et, assertComponentSurface as f, PairedCorrectness as fi, TaskGold as fn, EvalFixture as fr, ScenarioSignal as ft, surfaceContentHash as g, pairRunRecords as gi, parseCorrectnessResponse as gn, EvalFixtureScenario as gr, tangleTracesRoot as gt, renderSurfaceDiff as h, pairArms as hi, createTokenRecallChecker as hn, EvalFixtureRunPlan as hr, resolveRunDir as ht, WorktreeAdapter as i, ComparePairedArmsOptions as ii, CompletionRequirement as in, SignedManifestAlgo as ir, SearchLedgerConflictError as it, SearchCandidateSlotClosedEvent as j, LabeledScenarioStoreError as jn, CrossSurfaceCandidate as jr, ProfileMatrixError as jt, SearchCandidateRegisteredEvent as k, FsLabeledScenarioStore as kn, CrossSurfaceBestSingleSelection as kr, userStoryScoreboard as kt, TransientFailureOptions as l, PairRunRecordsResult as li, ProducedState as ln, verifyManifest as lr, campaignMeanComposite as lt, componentSurfaceIdentityMaterial as m, comparePairedArms as mi, createLlmCorrectnessChecker as mn, EvalFixtureLoadOptions as mr, selectDiscriminative as mt, GitWorktreeAdapterOptions as n, CrossSurfaceSelections as ni, ToolCallEventLike as nn, HypothesisResult as nr, openSearchLedger as nt, gitWorktreeAdapter as o, MatchedRunRecordPair as oi, CorrectnessChecker as on, evaluateHypothesis as or, SearchLedgerIntegrityError as ot, codeSurfaceIdentityMaterial as p, PairedMetricDelta as pi, completionVerdict as pn, EvalFixtureFile as pr, scoreDiscrimination as pt, SearchPlannedOperation as q, SequentialDecideFn as qn, CrossSurfaceInteractionTask as qr, ProfileAxisSpec as qt, Worktree as r, CrossSurfaceTaskRow as ri, extractProducedState as rn, SignedManifest as rr, validateSearchLedgerEvent as rt, resolveWorktreePath as s, PairArmsOptions as si, LlmCorrectnessCheckerOpts as sn, hashJson as sr, CampaignBreakdown as st, CodeSurfaceVerification as t, CrossSurfaceSelectionPolicy as ti, RuntimeEventLike as tn, HypothesisManifest as tr, SearchTokenAccounting as tt, isTransientTransportFailure as u, PairedArmRow as ui, RequirementCheck as un, NeutralizationGateOptions as ur, compareRankKeys as ut, SingleRunLock as v, AnalystScenario as vi, Artifact as vn, LoadEvalFixtureScenariosOptions as vr, PlaybackDriver as vt, SearchAccountingAudit as w, composeValidators as wn, analyzeCrossSurfaceInteractions as wr, UserStoryVerdict as wt, FileSearchLedger as x, buildAnalystSurfaceDispatch as xi, ValidationIssue as xn, loadEvalFixture as xr, ScoreboardRow as xt, SingleRunLockOptions as y, BuildAnalystSurfaceDispatchOptions as yi, ArtifactValidator as yn, PlanEvalFixtureRunOptions as yr, PlaybackStep as yt, SearchLedgerEvent as z, DimensionRegression as zn, CrossSurfaceDistribution as zr, assertRealAgentReceipts as zt };
|
|
2244
|
+
//# sourceMappingURL=index-DE5fb3EC.d.ts.map
|