@tangle-network/agent-eval 0.129.0 → 0.130.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +21 -0
- package/README.md +2 -0
- package/dist/active-curriculum-C4mk67HP.js +214 -0
- package/dist/active-curriculum-C4mk67HP.js.map +1 -0
- package/dist/adversarial-smnADNFS.d.ts +21 -0
- package/dist/adversarial-smnADNFS.d.ts.map +1 -0
- package/dist/analyst/index.d.ts +81 -2872
- package/dist/analyst/index.d.ts.map +1 -0
- package/dist/analyst/index.js +319 -360
- package/dist/analyst/index.js.map +1 -1
- package/dist/analyst-BkTS3C58.d.ts +89 -0
- package/dist/analyst-BkTS3C58.d.ts.map +1 -0
- package/dist/analyst-LsnNpSkm.js +152 -0
- package/dist/analyst-LsnNpSkm.js.map +1 -0
- package/dist/analyze-runs-C1CavBMk.js +1067 -0
- package/dist/analyze-runs-C1CavBMk.js.map +1 -0
- package/dist/analyze-runs-FsgYCinh.d.ts +72 -0
- package/dist/analyze-runs-FsgYCinh.d.ts.map +1 -0
- package/dist/attribute-vocabulary-DLJ6303h.d.ts +54 -0
- package/dist/attribute-vocabulary-DLJ6303h.d.ts.map +1 -0
- package/dist/authenticity/index.d.ts +81 -79
- package/dist/authenticity/index.d.ts.map +1 -0
- package/dist/authenticity/index.js +209 -193
- package/dist/authenticity/index.js.map +1 -1
- package/dist/baseline-HsBvw_dk.js +550 -0
- package/dist/baseline-HsBvw_dk.js.map +1 -0
- package/dist/baseline-hG3K85h4.d.ts +125 -0
- package/dist/baseline-hG3K85h4.d.ts.map +1 -0
- package/dist/belief-state/index.d.ts +448 -1188
- package/dist/belief-state/index.d.ts.map +1 -0
- package/dist/belief-state/index.js +1617 -1709
- package/dist/belief-state/index.js.map +1 -1
- package/dist/benchmarks/index.d.ts +2 -891
- package/dist/benchmarks/index.js +2 -60
- package/dist/benchmarks-BJgDGkAD.js +754 -0
- package/dist/benchmarks-BJgDGkAD.js.map +1 -0
- package/dist/builder-eval/index.d.ts +150 -662
- package/dist/builder-eval/index.d.ts.map +1 -0
- package/dist/builder-eval/index.js +356 -345
- package/dist/builder-eval/index.js.map +1 -1
- package/dist/calibration-CNWWA6K8.js +94 -0
- package/dist/calibration-CNWWA6K8.js.map +1 -0
- package/dist/campaign/index.d.ts +5 -6381
- package/dist/campaign/index.js +3 -213
- package/dist/campaign-aKJt6emI.js +3886 -0
- package/dist/campaign-aKJt6emI.js.map +1 -0
- package/dist/cli.d.ts +1 -1
- package/dist/cli.js +164 -175
- package/dist/cli.js.map +1 -1
- package/dist/client-C97NMzqi.d.ts +581 -0
- package/dist/client-C97NMzqi.d.ts.map +1 -0
- package/dist/client-CYzbdJOZ.js +637 -0
- package/dist/client-CYzbdJOZ.js.map +1 -0
- package/dist/code-agent-session-BjkMTQ7H.js +1390 -0
- package/dist/code-agent-session-BjkMTQ7H.js.map +1 -0
- package/dist/code-agent-session-DqqgOJaz.d.ts +143 -0
- package/dist/code-agent-session-DqqgOJaz.d.ts.map +1 -0
- package/dist/concurrency-DIxRZF_J.js +85 -0
- package/dist/concurrency-DIxRZF_J.js.map +1 -0
- package/dist/contract/index.d.ts +645 -5565
- package/dist/contract/index.d.ts.map +1 -0
- package/dist/contract/index.js +1730 -1938
- package/dist/contract/index.js.map +1 -1
- package/dist/control.d.ts +3 -1030
- package/dist/control.js +2 -33
- package/dist/cost-ledger-DIgQUFZZ.js +801 -0
- package/dist/cost-ledger-DIgQUFZZ.js.map +1 -0
- package/dist/cost-ledger-Dye6jCgg.d.ts +236 -0
- package/dist/cost-ledger-Dye6jCgg.d.ts.map +1 -0
- package/dist/dataset-BvtnC8Dc.d.ts +115 -0
- package/dist/dataset-BvtnC8Dc.d.ts.map +1 -0
- package/dist/default-registry-C-vFCSEc.js +2579 -0
- package/dist/default-registry-C-vFCSEc.js.map +1 -0
- package/dist/default-registry-CNPo-Vsb.d.ts +540 -0
- package/dist/default-registry-CNPo-Vsb.d.ts.map +1 -0
- package/dist/emitter-CPBAhxum.js +266 -0
- package/dist/emitter-CPBAhxum.js.map +1 -0
- package/dist/emitter-DGQGoLyj.d.ts +113 -0
- package/dist/emitter-DGQGoLyj.d.ts.map +1 -0
- package/dist/errors-8YnH8WlF.js +64 -0
- package/dist/errors-8YnH8WlF.js.map +1 -0
- package/dist/errors-CEk209JS.d.ts +76 -0
- package/dist/errors-CEk209JS.d.ts.map +1 -0
- package/dist/eval-campaign-DEm6c8ru.js +349 -0
- package/dist/eval-campaign-DEm6c8ru.js.map +1 -0
- package/dist/execution-tracks-CpgFPpS5.js +93 -0
- package/dist/execution-tracks-CpgFPpS5.js.map +1 -0
- package/dist/exporters-q9iL-2Jf.js +148 -0
- package/dist/exporters-q9iL-2Jf.js.map +1 -0
- package/dist/extract-usage-BrQ8mCLX.js +155 -0
- package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
- package/dist/failure-cluster-CqcvCcdR.d.ts +59 -0
- package/dist/failure-cluster-CqcvCcdR.d.ts.map +1 -0
- package/dist/feedback-trajectory-CVaeREXV.d.ts +340 -0
- package/dist/feedback-trajectory-CVaeREXV.d.ts.map +1 -0
- package/dist/fuzz.d.ts +320 -646
- package/dist/fuzz.d.ts.map +1 -0
- package/dist/fuzz.js +670 -618
- package/dist/fuzz.js.map +1 -1
- package/dist/hf-dataset-DBJXXoY1.js +763 -0
- package/dist/hf-dataset-DBJXXoY1.js.map +1 -0
- package/dist/hosted/index.d.ts +11 -831
- package/dist/hosted/index.d.ts.map +1 -0
- package/dist/hosted/index.js +2 -37
- package/dist/index-2JJSA6-r2.d.ts +926 -0
- package/dist/index-2JJSA6-r2.d.ts.map +1 -0
- package/dist/index-6N0aYmpW.d.ts +217 -0
- package/dist/index-6N0aYmpW.d.ts.map +1 -0
- package/dist/index-BAvgST_9.d.ts +131 -0
- package/dist/index-BAvgST_9.d.ts.map +1 -0
- package/dist/index-BvnJuTGD.d.ts +68 -0
- package/dist/index-BvnJuTGD.d.ts.map +1 -0
- package/dist/index-C61Wi7yg.d.ts +547 -0
- package/dist/index-C61Wi7yg.d.ts.map +1 -0
- package/dist/index-CD_WZ_Xr.d.ts +2250 -0
- package/dist/index-CD_WZ_Xr.d.ts.map +1 -0
- package/dist/index-DSC51roc.d.ts +102 -0
- package/dist/index-DSC51roc.d.ts.map +1 -0
- package/dist/index-Em67JBjs.d.ts +335 -0
- package/dist/index-Em67JBjs.d.ts.map +1 -0
- package/dist/index.d.ts +3755 -15555
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +11182 -11216
- package/dist/index.js.map +1 -1
- package/dist/integrity-BzRbCHzi.js +424 -0
- package/dist/integrity-BzRbCHzi.js.map +1 -0
- package/dist/integrity-rmVhXWA7.d.ts +61 -0
- package/dist/integrity-rmVhXWA7.d.ts.map +1 -0
- package/dist/judge-calibration-DFtEMlde.d.ts +146 -0
- package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
- package/dist/ledger-core/index.d.ts +2 -0
- package/dist/ledger-core/index.js +2 -0
- package/dist/ledger-core-DtZz1RG0.js +388 -0
- package/dist/ledger-core-DtZz1RG0.js.map +1 -0
- package/dist/llm-client--GR4JbZE.js +687 -0
- package/dist/llm-client--GR4JbZE.js.map +1 -0
- package/dist/llm-client-B_nIBlYo.d.ts +290 -0
- package/dist/llm-client-B_nIBlYo.d.ts.map +1 -0
- package/dist/matrix/index.d.ts +3 -155
- package/dist/matrix/index.js +2 -8
- package/dist/matrix-BzQnu2S6.js +270 -0
- package/dist/matrix-BzQnu2S6.js.map +1 -0
- package/dist/meta-eval/index.d.ts +4 -1027
- package/dist/meta-eval/index.js +393 -390
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/metrics-C9YY1OcL.js +239 -0
- package/dist/metrics-C9YY1OcL.js.map +1 -0
- package/dist/mint-yN2M2eh0.js +201 -0
- package/dist/mint-yN2M2eh0.js.map +1 -0
- package/dist/multi-layer-verifier-BHY1gWAc.d.ts +138 -0
- package/dist/multi-layer-verifier-BHY1gWAc.d.ts.map +1 -0
- package/dist/multishot/index.d.ts +273 -480
- package/dist/multishot/index.d.ts.map +1 -0
- package/dist/multishot/index.js +595 -548
- package/dist/multishot/index.js.map +1 -1
- package/dist/off-policy-DvgzvtIx.js +220 -0
- package/dist/off-policy-DvgzvtIx.js.map +1 -0
- package/dist/off-policy-mskQw8Mb.d.ts +153 -0
- package/dist/off-policy-mskQw8Mb.d.ts.map +1 -0
- package/dist/openapi.json +1 -1
- package/dist/opencode-sqlite-BGrHeDu3.js +318 -0
- package/dist/opencode-sqlite-BGrHeDu3.js.map +1 -0
- package/dist/outcome-store-BYHIuO0e.d.ts +65 -0
- package/dist/outcome-store-BYHIuO0e.d.ts.map +1 -0
- package/dist/outcome-store-ChBKlTd_.js +75 -0
- package/dist/outcome-store-ChBKlTd_.js.map +1 -0
- package/dist/paired-arms-D9D0wXj2.js +260 -0
- package/dist/paired-arms-D9D0wXj2.js.map +1 -0
- package/dist/pipelines/index.d.ts +95 -532
- package/dist/pipelines/index.d.ts.map +1 -0
- package/dist/pipelines/index.js +497 -478
- package/dist/pipelines/index.js.map +1 -1
- package/dist/pre-registration-DakwTRXk.js +96 -0
- package/dist/pre-registration-DakwTRXk.js.map +1 -0
- package/dist/propose-review-control-LhIWGzHl.js +1458 -0
- package/dist/propose-review-control-LhIWGzHl.js.map +1 -0
- package/dist/query-CJ_DX8vl.d.ts +28 -0
- package/dist/query-CJ_DX8vl.d.ts.map +1 -0
- package/dist/query-Di7eEQ79.js +83 -0
- package/dist/query-Di7eEQ79.js.map +1 -0
- package/dist/raw-provider-sink-BQd7mzyT.js +201 -0
- package/dist/raw-provider-sink-BQd7mzyT.js.map +1 -0
- package/dist/raw-provider-sink-BU29Sh8h.d.ts +134 -0
- package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +1 -0
- package/dist/redact-7Aq1ukl-.js +107 -0
- package/dist/redact-7Aq1ukl-.js.map +1 -0
- package/dist/release-report-Cz9NKH39.js +603 -0
- package/dist/release-report-Cz9NKH39.js.map +1 -0
- package/dist/release-report-mvB2G4_J.d.ts +244 -0
- package/dist/release-report-mvB2G4_J.d.ts.map +1 -0
- package/dist/replay-GnyotH0J.js +1741 -0
- package/dist/replay-GnyotH0J.js.map +1 -0
- package/dist/replay-OoidtG1E.d.ts +749 -0
- package/dist/replay-OoidtG1E.d.ts.map +1 -0
- package/dist/reporting.d.ts +6 -1312
- package/dist/reporting.js +6 -51
- package/dist/researcher-CwTdwXG1.d.ts +314 -0
- package/dist/researcher-CwTdwXG1.d.ts.map +1 -0
- package/dist/reward-hacking-eAnOsynk.d.ts +224 -0
- package/dist/reward-hacking-eAnOsynk.d.ts.map +1 -0
- package/dist/reward-hacking-qipEpKvY.js +596 -0
- package/dist/reward-hacking-qipEpKvY.js.map +1 -0
- package/dist/reward-nw2xZGZG.js +137 -0
- package/dist/reward-nw2xZGZG.js.map +1 -0
- package/dist/rl.d.ts +760 -4010
- package/dist/rl.d.ts.map +1 -0
- package/dist/rl.js +2325 -1958
- package/dist/rl.js.map +1 -1
- package/dist/rolldown-runtime-8H4AJuhK.js +14 -0
- package/dist/rollout/index.d.ts +3 -2087
- package/dist/rollout/index.js +8 -168
- package/dist/rollout-BOYjemfR.js +624 -0
- package/dist/rollout-BOYjemfR.js.map +1 -0
- package/dist/rubric-predictive-validity-B3xmbmS1.js +141 -0
- package/dist/rubric-predictive-validity-B3xmbmS1.js.map +1 -0
- package/dist/rubric-predictive-validity-Ku_clp_1.d.ts +74 -0
- package/dist/rubric-predictive-validity-Ku_clp_1.d.ts.map +1 -0
- package/dist/run-evidence-ClRX_8A9.d.ts +225 -0
- package/dist/run-evidence-ClRX_8A9.d.ts.map +1 -0
- package/dist/run-record-BuoE80Dq.js +467 -0
- package/dist/{chunk-56TAVBOK.js.map → run-record-BuoE80Dq.js.map} +1 -1
- package/dist/run-record-CnZu_gjl.d.ts +357 -0
- package/dist/run-record-CnZu_gjl.d.ts.map +1 -0
- package/dist/run-score-iEEAWiBY.js +41 -0
- package/dist/run-score-iEEAWiBY.js.map +1 -0
- package/dist/runtime-trajectory-1gyaTOoC.js +93 -0
- package/dist/runtime-trajectory-1gyaTOoC.js.map +1 -0
- package/dist/runtime-trajectory-BvSZcCHD.d.ts +50 -0
- package/dist/runtime-trajectory-BvSZcCHD.d.ts.map +1 -0
- package/dist/schema-BtVldJ3T.d.ts +206 -0
- package/dist/schema-BtVldJ3T.d.ts.map +1 -0
- package/dist/schema-C6DW4ZHR.js +821 -0
- package/dist/schema-C6DW4ZHR.js.map +1 -0
- package/dist/schema-CRhEY1SO.js +69 -0
- package/dist/schema-CRhEY1SO.js.map +1 -0
- package/dist/schema-Cef2cFmb.d.ts +408 -0
- package/dist/schema-Cef2cFmb.d.ts.map +1 -0
- package/dist/semantic-concept-judge-B6cWNJ2K.js +725 -0
- package/dist/semantic-concept-judge-B6cWNJ2K.js.map +1 -0
- package/dist/sequential-Br0mAPHA.js +148 -0
- package/dist/sequential-Br0mAPHA.js.map +1 -0
- package/dist/sequential-CYwq6Ff_.d.ts +141 -0
- package/dist/sequential-CYwq6Ff_.d.ts.map +1 -0
- package/dist/series-convergence-CjO2QdRW.js +43 -0
- package/dist/series-convergence-CjO2QdRW.js.map +1 -0
- package/dist/series-convergence-ofsqPWhs.d.ts +35 -0
- package/dist/series-convergence-ofsqPWhs.d.ts.map +1 -0
- package/dist/server-m5D9cvnG.js +1040 -0
- package/dist/server-m5D9cvnG.js.map +1 -0
- package/dist/skill-usage-C_pXm7lP.d.ts +534 -0
- package/dist/skill-usage-C_pXm7lP.d.ts.map +1 -0
- package/dist/skillopt-optimization-method-CUmHkGbI.js +7718 -0
- package/dist/skillopt-optimization-method-CUmHkGbI.js.map +1 -0
- package/dist/skillopt-optimization-method-CWKVTnks.d.ts +1740 -0
- package/dist/skillopt-optimization-method-CWKVTnks.d.ts.map +1 -0
- package/dist/statistics-Cmj6nynr.d.ts +514 -0
- package/dist/statistics-Cmj6nynr.d.ts.map +1 -0
- package/dist/statistics-CnnxdpOg.js +1437 -0
- package/dist/statistics-CnnxdpOg.js.map +1 -0
- package/dist/store-CT9YIIve.d.ts +117 -0
- package/dist/store-CT9YIIve.d.ts.map +1 -0
- package/dist/store-CxJry_cs.d.ts +229 -0
- package/dist/store-CxJry_cs.d.ts.map +1 -0
- package/dist/storyboard/index.d.ts +84 -203
- package/dist/storyboard/index.d.ts.map +1 -0
- package/dist/storyboard/index.js +609 -542
- package/dist/storyboard/index.js.map +1 -1
- package/dist/summary-report-BNs5nmXI.js +862 -0
- package/dist/summary-report-BNs5nmXI.js.map +1 -0
- package/dist/summary-report-Cj9gdw4i.d.ts +407 -0
- package/dist/summary-report-Cj9gdw4i.d.ts.map +1 -0
- package/dist/supervisor-run/index.d.ts +2 -959
- package/dist/supervisor-run/index.js +2 -65
- package/dist/supervisor-run-_lnTLM3z.js +1679 -0
- package/dist/supervisor-run-_lnTLM3z.js.map +1 -0
- package/dist/task-failure-attributes-CQZlB3et.js +311 -0
- package/dist/task-failure-attributes-CQZlB3et.js.map +1 -0
- package/dist/test-graded-scenario-BsqWLmPt.js +318 -0
- package/dist/test-graded-scenario-BsqWLmPt.js.map +1 -0
- package/dist/test-graded-scenario-D1TaI2va.d.ts +141 -0
- package/dist/test-graded-scenario-D1TaI2va.d.ts.map +1 -0
- package/dist/tools-BmuN627J.js +1085 -0
- package/dist/tools-BmuN627J.js.map +1 -0
- package/dist/trace-attributes.d.ts +2 -52
- package/dist/trace-attributes.js +131 -61
- package/dist/trace-attributes.js.map +1 -1
- package/dist/traces.d.ts +12 -2365
- package/dist/traces.js +12 -252
- package/dist/trajectory-D_7rLrvE.js +56 -0
- package/dist/trajectory-D_7rLrvE.js.map +1 -0
- package/dist/types-DGsxbAEd.d.ts +387 -0
- package/dist/types-DGsxbAEd.d.ts.map +1 -0
- package/dist/types-k9tZGKUg.d.ts +640 -0
- package/dist/types-k9tZGKUg.d.ts.map +1 -0
- package/dist/verdict-Dps8_okt.d.ts +37 -0
- package/dist/verdict-Dps8_okt.d.ts.map +1 -0
- package/dist/wire/index.d.ts +702 -1173
- package/dist/wire/index.d.ts.map +1 -0
- package/dist/wire/index.js +2 -81
- package/docs/campaign-proposers.md +1 -0
- package/package.json +17 -9
- package/dist/benchmarks/index.js.map +0 -1
- package/dist/campaign/index.js.map +0 -1
- package/dist/chunk-2QU3YOPR.js +0 -7374
- package/dist/chunk-2QU3YOPR.js.map +0 -1
- package/dist/chunk-3OCR4R5I.js +0 -728
- package/dist/chunk-3OCR4R5I.js.map +0 -1
- package/dist/chunk-3RF76KTD.js +0 -84
- package/dist/chunk-3RF76KTD.js.map +0 -1
- package/dist/chunk-56TAVBOK.js +0 -698
- package/dist/chunk-5DTSBUL2.js +0 -159
- package/dist/chunk-5DTSBUL2.js.map +0 -1
- package/dist/chunk-7FO3TNPI.js +0 -232
- package/dist/chunk-7FO3TNPI.js.map +0 -1
- package/dist/chunk-7ZZMD7UK.js +0 -386
- package/dist/chunk-7ZZMD7UK.js.map +0 -1
- package/dist/chunk-BOD4O7OF.js +0 -40
- package/dist/chunk-BOD4O7OF.js.map +0 -1
- package/dist/chunk-BSO5JDQH.js +0 -2335
- package/dist/chunk-BSO5JDQH.js.map +0 -1
- package/dist/chunk-C6LXANRU.js +0 -1550
- package/dist/chunk-C6LXANRU.js.map +0 -1
- package/dist/chunk-DODXQREJ.js +0 -752
- package/dist/chunk-DODXQREJ.js.map +0 -1
- package/dist/chunk-DRYIUNWY.js +0 -622
- package/dist/chunk-DRYIUNWY.js.map +0 -1
- package/dist/chunk-E7QXT7SX.js +0 -183
- package/dist/chunk-E7QXT7SX.js.map +0 -1
- package/dist/chunk-EG66UGL4.js +0 -341
- package/dist/chunk-EG66UGL4.js.map +0 -1
- package/dist/chunk-FXTVJPYD.js +0 -576
- package/dist/chunk-FXTVJPYD.js.map +0 -1
- package/dist/chunk-G7MGMCZD.js +0 -153
- package/dist/chunk-G7MGMCZD.js.map +0 -1
- package/dist/chunk-GGE4NNQT.js +0 -65
- package/dist/chunk-GGE4NNQT.js.map +0 -1
- package/dist/chunk-H23X7XKK.js +0 -181
- package/dist/chunk-H23X7XKK.js.map +0 -1
- package/dist/chunk-HHWE3POT.js +0 -94
- package/dist/chunk-HHWE3POT.js.map +0 -1
- package/dist/chunk-HPWUNB47.js +0 -289
- package/dist/chunk-HPWUNB47.js.map +0 -1
- package/dist/chunk-IYCLP2N2.js +0 -766
- package/dist/chunk-IYCLP2N2.js.map +0 -1
- package/dist/chunk-JHCHEVET.js +0 -274
- package/dist/chunk-JHCHEVET.js.map +0 -1
- package/dist/chunk-JQSF5DQT.js +0 -701
- package/dist/chunk-JQSF5DQT.js.map +0 -1
- package/dist/chunk-K4DBDHLK.js +0 -158
- package/dist/chunk-K4DBDHLK.js.map +0 -1
- package/dist/chunk-K6N6XJJX.js +0 -306
- package/dist/chunk-K6N6XJJX.js.map +0 -1
- package/dist/chunk-M4YBQKIJ.js +0 -1040
- package/dist/chunk-M4YBQKIJ.js.map +0 -1
- package/dist/chunk-MA6HLL3S.js +0 -65
- package/dist/chunk-MA6HLL3S.js.map +0 -1
- package/dist/chunk-MAZ26DC7.js +0 -99
- package/dist/chunk-MAZ26DC7.js.map +0 -1
- package/dist/chunk-NPCTHQIO.js +0 -91
- package/dist/chunk-NPCTHQIO.js.map +0 -1
- package/dist/chunk-NY44NC4A.js +0 -1056
- package/dist/chunk-NY44NC4A.js.map +0 -1
- package/dist/chunk-OIUOT4QD.js +0 -44
- package/dist/chunk-OIUOT4QD.js.map +0 -1
- package/dist/chunk-ONWEPEDO.js +0 -57
- package/dist/chunk-ONWEPEDO.js.map +0 -1
- package/dist/chunk-OWN5NPMC.js +0 -152
- package/dist/chunk-OWN5NPMC.js.map +0 -1
- package/dist/chunk-P6FYH6K4.js +0 -1161
- package/dist/chunk-P6FYH6K4.js.map +0 -1
- package/dist/chunk-PC4UYEBM.js +0 -166
- package/dist/chunk-PC4UYEBM.js.map +0 -1
- package/dist/chunk-PC5DOSM7.js +0 -579
- package/dist/chunk-PC5DOSM7.js.map +0 -1
- package/dist/chunk-PXE2VKMX.js +0 -140
- package/dist/chunk-PXE2VKMX.js.map +0 -1
- package/dist/chunk-PZ5AY32C.js +0 -10
- package/dist/chunk-PZ5AY32C.js.map +0 -1
- package/dist/chunk-QB6BDBP2.js +0 -4464
- package/dist/chunk-QB6BDBP2.js.map +0 -1
- package/dist/chunk-RXHCETDZ.js +0 -536
- package/dist/chunk-RXHCETDZ.js.map +0 -1
- package/dist/chunk-RZTMDUO7.js +0 -49
- package/dist/chunk-RZTMDUO7.js.map +0 -1
- package/dist/chunk-SFLLL76A.js +0 -669
- package/dist/chunk-SFLLL76A.js.map +0 -1
- package/dist/chunk-SZLVEKMJ.js +0 -1446
- package/dist/chunk-SZLVEKMJ.js.map +0 -1
- package/dist/chunk-T4SQEITX.js +0 -95
- package/dist/chunk-T4SQEITX.js.map +0 -1
- package/dist/chunk-T6RLYGAD.js +0 -158
- package/dist/chunk-T6RLYGAD.js.map +0 -1
- package/dist/chunk-TJVT4QFF.js +0 -911
- package/dist/chunk-TJVT4QFF.js.map +0 -1
- package/dist/chunk-TQ7LNKZ3.js +0 -136
- package/dist/chunk-TQ7LNKZ3.js.map +0 -1
- package/dist/chunk-U4L7JRPZ.js +0 -1706
- package/dist/chunk-U4L7JRPZ.js.map +0 -1
- package/dist/chunk-U4PHLT2N.js +0 -419
- package/dist/chunk-U4PHLT2N.js.map +0 -1
- package/dist/chunk-VCZ5FQYW.js +0 -928
- package/dist/chunk-VCZ5FQYW.js.map +0 -1
- package/dist/chunk-VI2UW6B6.js +0 -162
- package/dist/chunk-VI2UW6B6.js.map +0 -1
- package/dist/chunk-VQMK5FMP.js +0 -247
- package/dist/chunk-VQMK5FMP.js.map +0 -1
- package/dist/chunk-WGXIEX7P.js +0 -116
- package/dist/chunk-WGXIEX7P.js.map +0 -1
- package/dist/chunk-WVATSFCP.js +0 -1553
- package/dist/chunk-WVATSFCP.js.map +0 -1
- package/dist/chunk-X4YIBDER.js +0 -1662
- package/dist/chunk-X4YIBDER.js.map +0 -1
- package/dist/chunk-YQN4ICPP.js +0 -355
- package/dist/chunk-YQN4ICPP.js.map +0 -1
- package/dist/chunk-ZET2UAYW.js +0 -89
- package/dist/chunk-ZET2UAYW.js.map +0 -1
- package/dist/chunk-ZHTZ4EYI.js +0 -1212
- package/dist/chunk-ZHTZ4EYI.js.map +0 -1
- package/dist/control.js.map +0 -1
- package/dist/hosted/index.js.map +0 -1
- package/dist/matrix/index.js.map +0 -1
- package/dist/reporting.js.map +0 -1
- package/dist/rollout/index.js.map +0 -1
- package/dist/run-campaign-OJJ7CZF4.js +0 -18
- package/dist/run-campaign-OJJ7CZF4.js.map +0 -1
- package/dist/supervisor-run/index.js.map +0 -1
- package/dist/traces.js.map +0 -1
- package/dist/wire/index.js.map +0 -1
|
@@ -0,0 +1,1740 @@
|
|
|
1
|
+
import { a as RunRecord } from "./run-record-CnZu_gjl.js";
|
|
2
|
+
import { s as TraceStore } from "./store-CT9YIIve.js";
|
|
3
|
+
import { c as CostLedgerHandle, f as CostLedgerSummary, o as CostLedger, p as CostReceipt, y as CustomTokenPricing } from "./cost-ledger-Dye6jCgg.js";
|
|
4
|
+
import { n as LlmCallMetadata } from "./llm-client-B_nIBlYo.js";
|
|
5
|
+
import { A as ChatClient } from "./types-DGsxbAEd.js";
|
|
6
|
+
import { f as PairedBootstrapResult } from "./statistics-Cmj6nynr.js";
|
|
7
|
+
import { C as JudgeDimension, H as SurfaceProposer, N as ParetoParent, R as Scenario, S as JudgeConfig, _ as GateDecision, a as CampaignResult, b as GenerationRecord, c as CampaignTraceWriter, d as DispatchContext, f as DispatchFn, g as GateContribution, h as GateContext, i as CampaignCostMeter, j as MutableSurface, k as LabeledScenarioStore, o as CampaignScenarioIdentity, p as Gate, r as CampaignCellResult, v as GateResult, y as GenerationCandidate } from "./types-k9tZGKUg.js";
|
|
8
|
+
import { a as DatasetScenario, t as Dataset } from "./dataset-BvtnC8Dc.js";
|
|
9
|
+
import { t as DetectRewardHackingInput } from "./reward-hacking-eAnOsynk.js";
|
|
10
|
+
import { g as TraceSpanEvent, t as HostedClient } from "./client-C97NMzqi.js";
|
|
11
|
+
import { z } from "zod";
|
|
12
|
+
//#region src/llm-judge.d.ts
|
|
13
|
+
/** A rubric dimension as a bare key or the full `{ key, description }` shape. A
|
|
14
|
+
* bare string uses the key as its own description. */
|
|
15
|
+
type LlmJudgeDimension = string | JudgeDimension;
|
|
16
|
+
interface LlmJudgeOptions<TArtifact, TScenario extends Scenario = Scenario> {
|
|
17
|
+
/** The injected LLM transport. One `chat()` call per `score()`. Required —
|
|
18
|
+
* there is no default route, so a misconfigured judge fails at construction,
|
|
19
|
+
* never silently against the free-tier router. */
|
|
20
|
+
chat: ChatClient;
|
|
21
|
+
/** Rubric dimensions the model scores. Each becomes a `[0,1]` field of the
|
|
22
|
+
* returned `JudgeScore.dimensions`. Defaults to a single `quality` dimension. */
|
|
23
|
+
dimensions?: LlmJudgeDimension[];
|
|
24
|
+
/** Model id. Falls back to `chat.defaultModel`; one of the two MUST resolve. */
|
|
25
|
+
model?: string;
|
|
26
|
+
/** Explicit scoring revision for opaque transport or renderer changes. */
|
|
27
|
+
judgeVersion?: string;
|
|
28
|
+
temperature?: number;
|
|
29
|
+
maxTokens?: number;
|
|
30
|
+
/** Composite weights forwarded to `weightedComposite`: a partial map selects
|
|
31
|
+
* AND weights exactly the named dimensions. Omit for a uniform mean. */
|
|
32
|
+
weights?: Record<string, number>;
|
|
33
|
+
/** Scale the model is prompted to score on, normalized into `[0,1]`:
|
|
34
|
+
* - `'unit'` (default): the model returns `[0,1]` directly.
|
|
35
|
+
* - `'ten'`: the model returns `[0,10]`; divided by 10 here.
|
|
36
|
+
* The prompt is annotated with the expected range either way. */
|
|
37
|
+
scale?: 'unit' | 'ten';
|
|
38
|
+
/** Run this judge only on matching scenarios (mirrors `JudgeConfig.appliesTo`). */
|
|
39
|
+
appliesTo?: (scenario: TScenario) => boolean;
|
|
40
|
+
/** Render the artifact + scenario into the user message. Default:
|
|
41
|
+
* pretty-printed JSON of `{ scenario, artifact }`. */
|
|
42
|
+
renderUser?: (input: {
|
|
43
|
+
artifact: TArtifact;
|
|
44
|
+
scenario: TScenario;
|
|
45
|
+
}) => string;
|
|
46
|
+
/** Strict runtime contract; its JSON Schema is sent to the provider. */
|
|
47
|
+
costLedger?: CostLedgerHandle;
|
|
48
|
+
responseSchema?: {
|
|
49
|
+
name: string;
|
|
50
|
+
schema: z.ZodObject;
|
|
51
|
+
};
|
|
52
|
+
}
|
|
53
|
+
/**
|
|
54
|
+
* Build a campaign-shaped `JudgeConfig` whose `score()` makes ONE LLM call
|
|
55
|
+
* against `prompt` and reduces the model's per-dimension scores to a canonical
|
|
56
|
+
* `JudgeScore` in `[0,1]`.
|
|
57
|
+
*
|
|
58
|
+
* The model is instructed to return JSON `{ "dimensions": { <key>: <number>, … },
|
|
59
|
+
* "notes": "…" }`; the helper strips fenced JSON, validates every declared
|
|
60
|
+
* dimension is present and in range, normalizes by `scale`, and composites via
|
|
61
|
+
* `weightedComposite`.
|
|
62
|
+
*/
|
|
63
|
+
declare function llmJudge<TArtifact = unknown, TScenario extends Scenario = Scenario>(name: string, prompt: string, opts: LlmJudgeOptions<TArtifact, TScenario>): JudgeConfig<TArtifact, TScenario>;
|
|
64
|
+
//#endregion
|
|
65
|
+
//#region src/reference-equivalence-judge.d.ts
|
|
66
|
+
declare const REFERENCE_EQUIVALENCE_JUDGE_VERSION = "reference-equivalence-judge-v1-2026-07-13";
|
|
67
|
+
declare const REFERENCE_EQUIVALENCE_INPUT_LIMITS: {
|
|
68
|
+
readonly userRequest: 8000;
|
|
69
|
+
readonly expectedAnswer: 32000;
|
|
70
|
+
readonly candidateOutput: 32000;
|
|
71
|
+
};
|
|
72
|
+
interface ReferenceEquivalenceScenario extends Scenario {
|
|
73
|
+
userRequest: string;
|
|
74
|
+
expectedAnswer: string;
|
|
75
|
+
}
|
|
76
|
+
interface ReferenceEquivalenceJudgeInput {
|
|
77
|
+
userRequest: string;
|
|
78
|
+
expectedAnswer: string;
|
|
79
|
+
candidateOutput: string;
|
|
80
|
+
}
|
|
81
|
+
interface ReferenceEquivalenceJudgeOptions {
|
|
82
|
+
/** Injected transport. No implicit provider or credentials are selected. */
|
|
83
|
+
chat: ChatClient;
|
|
84
|
+
/** Falls back to the ChatClient's default model. */
|
|
85
|
+
model?: string;
|
|
86
|
+
/** Used only by the direct-call adapter. */
|
|
87
|
+
signal?: AbortSignal;
|
|
88
|
+
/** Optional receipt destination for direct calls; campaigns supply their own. */
|
|
89
|
+
costLedger?: CostLedgerHandle;
|
|
90
|
+
}
|
|
91
|
+
interface ReferenceEquivalenceJudgeResult extends LlmCallMetadata {
|
|
92
|
+
kind: 'reference-equivalence';
|
|
93
|
+
version: string;
|
|
94
|
+
score: number;
|
|
95
|
+
rationale: string;
|
|
96
|
+
}
|
|
97
|
+
/** Build the campaign-native expected-answer judge. */
|
|
98
|
+
declare function createReferenceEquivalenceJudge(options: ReferenceEquivalenceJudgeOptions): JudgeConfig<string, ReferenceEquivalenceScenario>;
|
|
99
|
+
/** Direct-call adapter over the campaign judge for product callers. */
|
|
100
|
+
declare function runReferenceEquivalenceJudge(input: ReferenceEquivalenceJudgeInput, options: ReferenceEquivalenceJudgeOptions): Promise<ReferenceEquivalenceJudgeResult>;
|
|
101
|
+
//#endregion
|
|
102
|
+
//#region src/campaign/auto-pr.d.ts
|
|
103
|
+
interface OpenAutoPrOptions<TArtifact, TScenario extends Scenario> {
|
|
104
|
+
/** Campaign result to attach to the PR. */
|
|
105
|
+
result: CampaignResult<TArtifact, TScenario>;
|
|
106
|
+
/** Gate verdict explaining the promotion. Substrate refuses to open a PR
|
|
107
|
+
* when `gate.decision !== 'ship'` — fails loud. */
|
|
108
|
+
gate: GateResult;
|
|
109
|
+
/** Promoted surface diff — typically the new system prompt addendum or
|
|
110
|
+
* full profile diff. Substrate writes it as the PR body. */
|
|
111
|
+
promotedDiff: string;
|
|
112
|
+
/** GH owner/repo target (e.g., `tangle-network/gtm-agent`). */
|
|
113
|
+
ghOwner: string;
|
|
114
|
+
ghRepo: string;
|
|
115
|
+
/** Branch name for the PR. Default `auto/<manifestHash[:12]>`. */
|
|
116
|
+
branch?: string;
|
|
117
|
+
/** PR title. Default includes manifest hash. */
|
|
118
|
+
title?: string;
|
|
119
|
+
/** Whether to actually open the PR or just dry-run. Default reads
|
|
120
|
+
* `GH_AUTO_PR_TOKEN` env — present = open, absent = dry-run. */
|
|
121
|
+
dryRun?: boolean;
|
|
122
|
+
/** Test seam — substitute `gh pr create` invocation. */
|
|
123
|
+
ghExec?: (args: string[]) => {
|
|
124
|
+
stdout: string;
|
|
125
|
+
stderr: string;
|
|
126
|
+
status: number;
|
|
127
|
+
};
|
|
128
|
+
}
|
|
129
|
+
interface OpenAutoPrResult {
|
|
130
|
+
opened: boolean;
|
|
131
|
+
prUrl?: string;
|
|
132
|
+
dryRun: boolean;
|
|
133
|
+
reason: string;
|
|
134
|
+
}
|
|
135
|
+
/**
|
|
136
|
+
* Open a GitHub PR for a gate-approved surface promotion, attaching the manifest hash, gate verdict, and diff as the PR body.
|
|
137
|
+
*/
|
|
138
|
+
declare function openAutoPr<TArtifact, TScenario extends Scenario>(options: OpenAutoPrOptions<TArtifact, TScenario>): OpenAutoPrResult;
|
|
139
|
+
//#endregion
|
|
140
|
+
//#region src/campaign/coverage.d.ts
|
|
141
|
+
/** Reject campaign designs whose denominator cannot be identified exactly. */
|
|
142
|
+
declare function assertCampaignDesign<TScenario extends Scenario>(scenarios: readonly TScenario[], reps: number): void;
|
|
143
|
+
/** Redacted but independently verifiable identity of one complete scenario. */
|
|
144
|
+
declare function campaignScenarioIdentity<TScenario extends Scenario>(scenario: TScenario): CampaignScenarioIdentity & Pick<TScenario, 'id' | 'kind'>;
|
|
145
|
+
/** Canonical split identity reconstructed from redacted scenario identities. */
|
|
146
|
+
declare function campaignSplitDigestFromIdentities(scenarios: readonly CampaignScenarioIdentity[], reps: number): `sha256:${string}`;
|
|
147
|
+
/** Canonical identity of the exact scenario payloads and replicate count. */
|
|
148
|
+
declare function campaignSplitDigest<TScenario extends Scenario>(scenarios: readonly TScenario[], reps: number): `sha256:${string}`;
|
|
149
|
+
/** Refuse a campaign whose retained task identities contradict its split digest. */
|
|
150
|
+
declare function assertCampaignSplitIdentity(scenarios: readonly CampaignScenarioIdentity[], reps: number, splitDigest: string): void;
|
|
151
|
+
//#endregion
|
|
152
|
+
//#region src/campaign/external-optimizer-contracts.d.ts
|
|
153
|
+
interface ExternalOptimizerRunnerCommand {
|
|
154
|
+
command?: string;
|
|
155
|
+
args?: readonly string[];
|
|
156
|
+
env?: NodeJS.ProcessEnv;
|
|
157
|
+
}
|
|
158
|
+
type ExternalOptimizerResumeMode = 'never' | 'if-compatible' | 'required';
|
|
159
|
+
type ExternalTextCandidate = string | Record<string, string>;
|
|
160
|
+
interface ExternalTextEvaluationRequest {
|
|
161
|
+
candidate: ExternalTextCandidate;
|
|
162
|
+
exampleId: string;
|
|
163
|
+
}
|
|
164
|
+
interface ExternalOptimizerModelBudget {
|
|
165
|
+
/** Maximum optimizer-model spend, independent of task-evaluation spend. */
|
|
166
|
+
maxCostUsd: number;
|
|
167
|
+
/** Network attempts, including provider retries. */
|
|
168
|
+
maxRequests: number;
|
|
169
|
+
/** Reject a request body above this byte count. */
|
|
170
|
+
maxRequestBytes: number;
|
|
171
|
+
/** Reject a provider response above this byte count. */
|
|
172
|
+
maxResponseBytes: number;
|
|
173
|
+
/** Reject a request asking the provider for more output tokens. */
|
|
174
|
+
maxOutputTokensPerRequest: number;
|
|
175
|
+
/** Rates used to estimate cost when the provider omits a valid `usage.cost`. */
|
|
176
|
+
pricing: CustomTokenPricing;
|
|
177
|
+
/** Per-provider-request deadline. Default: 300,000 ms. */
|
|
178
|
+
requestTimeoutMs?: number;
|
|
179
|
+
}
|
|
180
|
+
//#endregion
|
|
181
|
+
//#region src/campaign/storage.d.ts
|
|
182
|
+
/**
|
|
183
|
+
* `CampaignStorage` — the filesystem seam `runCampaign` writes through
|
|
184
|
+
* (run/cell dirs, the resumability cache, per-cell artifacts, trace spans).
|
|
185
|
+
*
|
|
186
|
+
* The default (`fsCampaignStorage`) is the Node filesystem — identical
|
|
187
|
+
* behavior to the inline `node:fs` calls it replaces, so existing CLI
|
|
188
|
+
* consumers are unaffected. `inMemoryCampaignStorage` keeps everything in a
|
|
189
|
+
* `Map`, so the substrate runs in environments WITHOUT a filesystem
|
|
190
|
+
* (Cloudflare Workers, Deno Deploy, other edge runtimes) — the campaign
|
|
191
|
+
* still produces its `CampaignResult` (cells + aggregates) in memory;
|
|
192
|
+
* artifacts/traces simply aren't persisted to disk.
|
|
193
|
+
*
|
|
194
|
+
* Paths are opaque keys to the in-memory adapter — it does not parse them,
|
|
195
|
+
* so the same `join(...)`-built paths work unchanged across both adapters.
|
|
196
|
+
*/
|
|
197
|
+
interface CampaignStorage {
|
|
198
|
+
/** Ensure a directory exists (recursive). No-op for in-memory. */
|
|
199
|
+
ensureDir(dir: string): void;
|
|
200
|
+
/** Does this path exist (as a written file or an ensured dir)? */
|
|
201
|
+
exists(path: string): boolean;
|
|
202
|
+
/** Read a UTF-8 file; `undefined` when missing or unreadable. */
|
|
203
|
+
read(path: string): string | undefined;
|
|
204
|
+
/** Write a file (string or bytes). Parent dir is assumed ensured. */
|
|
205
|
+
write(path: string, content: string | Uint8Array): void;
|
|
206
|
+
/** Append only when the current UTF-8 byte length matches `expectedBytes`.
|
|
207
|
+
* Returns the new length, or undefined when another writer won. */
|
|
208
|
+
append(path: string, content: string, expectedBytes: number): number | undefined;
|
|
209
|
+
}
|
|
210
|
+
/** Node-filesystem storage — the default. Lazily requires `node:fs` so the
|
|
211
|
+
* module imports cleanly in non-Node runtimes (where the caller passes
|
|
212
|
+
* `inMemoryCampaignStorage` instead and never constructs this).
|
|
213
|
+
*
|
|
214
|
+
* `createRequire(import.meta.url)` is the ESM-native lazy require — a bare
|
|
215
|
+
* `require` is a ReferenceError under `"type": "module"`, which is exactly
|
|
216
|
+
* the shape this package publishes. */
|
|
217
|
+
declare function fsCampaignStorage(): CampaignStorage;
|
|
218
|
+
/** In-memory storage for filesystem-less runtimes. Artifacts + trace spans
|
|
219
|
+
* live in a `Map` for the duration of the run; the `CampaignResult` is
|
|
220
|
+
* fully populated, but nothing is persisted to disk. */
|
|
221
|
+
declare function inMemoryCampaignStorage(): CampaignStorage;
|
|
222
|
+
/** Open the durable spend account stored beside a logical run. */
|
|
223
|
+
declare function createRunCostLedger(input: {
|
|
224
|
+
storage: CampaignStorage;
|
|
225
|
+
runDir: string;
|
|
226
|
+
costCeilingUsd?: number;
|
|
227
|
+
}): CostLedger;
|
|
228
|
+
//#endregion
|
|
229
|
+
//#region src/campaign/run-campaign.d.ts
|
|
230
|
+
interface RunCampaignOptions<TScenario extends Scenario, TArtifact> {
|
|
231
|
+
scenarios: TScenario[];
|
|
232
|
+
dispatch: DispatchFn<TScenario, TArtifact>;
|
|
233
|
+
/** Abort active dispatches when the owning operation is cancelled. */
|
|
234
|
+
signal?: AbortSignal;
|
|
235
|
+
/**
|
|
236
|
+
* Stable identity for the dispatch behavior, included in the manifest/cache
|
|
237
|
+
* key. Set this when the same function name can run different models,
|
|
238
|
+
* prompts, tools, or external config.
|
|
239
|
+
*/
|
|
240
|
+
dispatchRef?: string;
|
|
241
|
+
judges?: JudgeConfig<TArtifact, TScenario>[];
|
|
242
|
+
/** Required for reproducibility. Default 42. */
|
|
243
|
+
seed?: number;
|
|
244
|
+
/** Per-scenario replicates for CI bands. Default 1; raise to 5+ for
|
|
245
|
+
* bootstrap-tight intervals on critical eval. */
|
|
246
|
+
reps?: number;
|
|
247
|
+
/** When true (default), completed cells are cached by
|
|
248
|
+
* (manifestHash, scenarioId, rep, generation). Re-runs skip cached cells. */
|
|
249
|
+
resumable?: boolean;
|
|
250
|
+
/** Optional store — when present, every artifact + judge score is captured
|
|
251
|
+
* with the configured `captureSource`. Capture is default ON; pass `'off'`
|
|
252
|
+
* to disable. */
|
|
253
|
+
labeledStore?: LabeledScenarioStore | 'off';
|
|
254
|
+
captureSource?: 'production-trace' | 'eval-run' | 'manual' | 'red-team' | 'synthetic';
|
|
255
|
+
captureSourceVersionHash?: string;
|
|
256
|
+
/** Hard spend cap. Each paid call reserves its enforced maximum before dispatch. */
|
|
257
|
+
costCeiling?: number;
|
|
258
|
+
/** Shared spend account. Improvement loops pass one ledger through every
|
|
259
|
+
* campaign so the ceiling and returned total are run-wide. */
|
|
260
|
+
costLedger?: CostLedgerHandle;
|
|
261
|
+
/** Attribution label for receipts recorded by this campaign. */
|
|
262
|
+
costPhase?: string;
|
|
263
|
+
/** Additional immutable receipt tags supplied by an owning workflow. */
|
|
264
|
+
costTags?: Readonly<Record<string, string>>;
|
|
265
|
+
/** Max concurrent cells. Default 2. */
|
|
266
|
+
maxConcurrency?: number;
|
|
267
|
+
/**
|
|
268
|
+
* Stop after the first dispatch or judge error. The failed cell is persisted
|
|
269
|
+
* before active sibling cells are aborted and drained, then the campaign
|
|
270
|
+
* rejects with the exact error thrown by that dispatch or judge.
|
|
271
|
+
* Default false preserves the normal behavior of returning failed cells and
|
|
272
|
+
* continuing the remaining schedule.
|
|
273
|
+
*/
|
|
274
|
+
abortOnCellError?: boolean;
|
|
275
|
+
/**
|
|
276
|
+
* Per-cell dispatch deadline in ms. A `dispatch` that neither resolves nor
|
|
277
|
+
* rejects within this window is a hang (a stalled model request, an
|
|
278
|
+
* exhausted runtime resource, a backend that never closes its stream). When
|
|
279
|
+
* set, the cell's `ctx.signal` is aborted. A dispatch that stops is recorded
|
|
280
|
+
* as an error (`dispatch exceeded <N>ms`). A dispatch that ignores
|
|
281
|
+
* cancellation rejects the campaign without publishing incomplete cost data.
|
|
282
|
+
* `undefined`/`0` means unbounded.
|
|
283
|
+
*/
|
|
284
|
+
dispatchTimeoutMs?: number;
|
|
285
|
+
/**
|
|
286
|
+
* Time allowed for an aborted dispatch and its paid calls to stop before the
|
|
287
|
+
* campaign rejects without producing a result. Default 5 seconds.
|
|
288
|
+
*/
|
|
289
|
+
dispatchShutdownTimeoutMs?: number;
|
|
290
|
+
/** Required: where artifacts + traces land. A bare name (not an absolute path)
|
|
291
|
+
* resolves to the shared `~/.tangle/traces/<repo>/runs/<name>` root so run
|
|
292
|
+
* bundles never pollute a repo working tree. Pass an absolute path to override. */
|
|
293
|
+
runDir: string;
|
|
294
|
+
/** Subject repo for the shared run-dir root (defaults to the CWD basename).
|
|
295
|
+
* Only consulted when `runDir` is a bare name. */
|
|
296
|
+
repo?: string;
|
|
297
|
+
/** Tracing posture. Default is the substrate's `FileSystemTraceStore` rooted
|
|
298
|
+
* at `<runDir>/traces/`. `'off'` disables capture entirely — substrate
|
|
299
|
+
* refuses this when the caller wires `autoOnPromote !== 'none'`. */
|
|
300
|
+
tracing?: 'on' | 'off';
|
|
301
|
+
/**
|
|
302
|
+
* Per-cell usage expectation — the early, fine-grained sibling of the
|
|
303
|
+
* batch `assertRealBackend` guard. A cell that produced an artifact (no
|
|
304
|
+
* error) but reported `costUsd === 0` AND zero tokens is a stub: the
|
|
305
|
+
* dispatch never reported LLM activity via `ctx.cost`. Modes:
|
|
306
|
+
* - `'warn'` (default) — log the offending cell loudly, keep going.
|
|
307
|
+
* - `'assert'` — throw `BackendIntegrityError` on the first such cell
|
|
308
|
+
* (fail-fast; recommended for CI campaigns expecting real LLM calls).
|
|
309
|
+
* - `'off'` — no check (replay / deterministic-only / offline analysis).
|
|
310
|
+
*/
|
|
311
|
+
expectUsage?: 'assert' | 'warn' | 'off';
|
|
312
|
+
/** Test seam — override the wall clock for deterministic tests. */
|
|
313
|
+
now?: () => Date;
|
|
314
|
+
/** Test seam — override per-cell trace writer factory. */
|
|
315
|
+
buildTraceWriter?: (cellId: string, dir: string) => CampaignTraceWriter;
|
|
316
|
+
/** Storage backend for run/cell dirs, the resumability cache, artifacts,
|
|
317
|
+
* and trace spans. Default: the Node filesystem (`fsCampaignStorage`).
|
|
318
|
+
* Pass `inMemoryCampaignStorage()` to run in a filesystem-less runtime
|
|
319
|
+
* (Cloudflare Workers, Deno, edge) — the `CampaignResult` is still
|
|
320
|
+
* produced; artifacts/traces just aren't persisted to disk. */
|
|
321
|
+
storage?: CampaignStorage;
|
|
322
|
+
/**
|
|
323
|
+
* Optional per-cell placement strategy. Returns an opaque string the
|
|
324
|
+
* substrate forwards as `ctx.placement` to the Dispatch — placement-aware
|
|
325
|
+
* Dispatches (e.g. `httpDispatch` from `/adapters/http`) use it to route
|
|
326
|
+
* each cell to the right worker, region, or sandbox. When unset, every
|
|
327
|
+
* cell receives `ctx.placement = undefined` and behaves identically to
|
|
328
|
+
* the in-process case.
|
|
329
|
+
*
|
|
330
|
+
* @example
|
|
331
|
+
* cellPlacement: ({ scenario }) => scenario.tags?.includes('eu') ? 'eu-west' : 'us-east'
|
|
332
|
+
*/
|
|
333
|
+
cellPlacement?: (input: {
|
|
334
|
+
scenario: TScenario;
|
|
335
|
+
rep: number;
|
|
336
|
+
generation?: number;
|
|
337
|
+
}) => string | undefined;
|
|
338
|
+
}
|
|
339
|
+
/** Durable `<cell>/failure-receipt.json` written before a failed cell can
|
|
340
|
+
* trigger campaign-wide cancellation. The cell records dispatch measurements;
|
|
341
|
+
* `cost` covers every settled agent and judge call attributed to this exact run
|
|
342
|
+
* attempt. */
|
|
343
|
+
interface CampaignCellFailureReceipt<TArtifact = unknown> {
|
|
344
|
+
schemaVersion: 1;
|
|
345
|
+
runAttemptId: string;
|
|
346
|
+
recordedAt: string;
|
|
347
|
+
failure: {
|
|
348
|
+
stage: 'dispatch' | 'judge';
|
|
349
|
+
judge?: string;
|
|
350
|
+
error: {
|
|
351
|
+
name: string;
|
|
352
|
+
message: string;
|
|
353
|
+
stack?: string;
|
|
354
|
+
};
|
|
355
|
+
};
|
|
356
|
+
cell: CampaignCellResult<TArtifact>;
|
|
357
|
+
cost: CostLedgerSummary;
|
|
358
|
+
}
|
|
359
|
+
/**
|
|
360
|
+
* Core campaign orchestrator: fan scenarios through dispatch, score with judges, aggregate bootstrap CIs, and persist reproducible `CampaignResult` records.
|
|
361
|
+
*/
|
|
362
|
+
declare function runCampaign<TScenario extends Scenario, TArtifact>(opts: RunCampaignOptions<TScenario, TArtifact>): Promise<CampaignResult<TArtifact, TScenario>>;
|
|
363
|
+
interface CampaignRunPlanCell {
|
|
364
|
+
cellId: string;
|
|
365
|
+
scenarioId: string;
|
|
366
|
+
rep: number;
|
|
367
|
+
seed: number;
|
|
368
|
+
cachePath: string;
|
|
369
|
+
status: 'cached' | 'run';
|
|
370
|
+
reason?: 'missing' | 'manifest-mismatch' | 'cell-mismatch' | 'corrupt' | 'resumable-off';
|
|
371
|
+
}
|
|
372
|
+
interface CampaignRunPlan {
|
|
373
|
+
manifestHash: string;
|
|
374
|
+
splitDigest: `sha256:${string}`;
|
|
375
|
+
totalCells: number;
|
|
376
|
+
cellsCached: number;
|
|
377
|
+
cellsToRun: number;
|
|
378
|
+
cells: CampaignRunPlanCell[];
|
|
379
|
+
}
|
|
380
|
+
interface PlanCampaignRunOptions<TScenario extends Scenario, TArtifact> {
|
|
381
|
+
scenarios: TScenario[];
|
|
382
|
+
dispatch?: DispatchFn<TScenario, TArtifact>;
|
|
383
|
+
dispatchRef?: string;
|
|
384
|
+
judges?: JudgeConfig<TArtifact, TScenario>[];
|
|
385
|
+
seed?: number;
|
|
386
|
+
reps?: number;
|
|
387
|
+
resumable?: boolean;
|
|
388
|
+
runDir: string;
|
|
389
|
+
/** Subject repo for the shared run-dir root (see RunCampaignOptions.repo). */
|
|
390
|
+
repo?: string;
|
|
391
|
+
storage?: CampaignStorage;
|
|
392
|
+
}
|
|
393
|
+
/**
|
|
394
|
+
* Plan a campaign WITHOUT dispatching: computes the manifest hash and the per-cell
|
|
395
|
+
* run-vs-cached schedule so callers can preview cost and resumability before spending.
|
|
396
|
+
*/
|
|
397
|
+
declare function planCampaignRun<TScenario extends Scenario, TArtifact>(opts: PlanCampaignRunOptions<TScenario, TArtifact>): CampaignRunPlan;
|
|
398
|
+
//#endregion
|
|
399
|
+
//#region src/campaign/presets/compare-optimization-methods.d.ts
|
|
400
|
+
/** Shared campaign settings applied to every optimization method. */
|
|
401
|
+
type OptimizationMethodRunOptions<TScenario extends Scenario, TArtifact> = Omit<RunCampaignOptions<TScenario, TArtifact>, 'costCeiling' | 'costLedger' | 'dispatch' | 'judges' | 'runDir' | 'scenarios' | 'seed'>;
|
|
402
|
+
/** Cost reported by a method or by final test scoring. */
|
|
403
|
+
interface ComparisonCost {
|
|
404
|
+
totalCostUsd: number;
|
|
405
|
+
accountingComplete: boolean;
|
|
406
|
+
incompleteReasons: string[];
|
|
407
|
+
}
|
|
408
|
+
interface OptimizationPackageSource {
|
|
409
|
+
kind: 'package';
|
|
410
|
+
/** Whether package identity was inspected or supplied by caller code. */
|
|
411
|
+
evidence: 'observed' | 'declared';
|
|
412
|
+
package: string;
|
|
413
|
+
version: string;
|
|
414
|
+
sourceUrl?: string;
|
|
415
|
+
revision?: string;
|
|
416
|
+
/** SHA-256 of all installed module files observed before the run. */
|
|
417
|
+
sourceSha256?: string;
|
|
418
|
+
}
|
|
419
|
+
interface OptimizationModuleSource {
|
|
420
|
+
module: string;
|
|
421
|
+
sourceSha256: string;
|
|
422
|
+
}
|
|
423
|
+
interface OptimizationPythonRuntime {
|
|
424
|
+
implementation: string;
|
|
425
|
+
version: string;
|
|
426
|
+
}
|
|
427
|
+
interface OptimizationTokenUsage {
|
|
428
|
+
/** All input tokens, including cache reads and cache creation. */
|
|
429
|
+
inputTokens: number;
|
|
430
|
+
/** Input tokens served from a provider cache. */
|
|
431
|
+
cachedInputTokens?: number;
|
|
432
|
+
/** Input tokens used to create or write a provider cache entry. */
|
|
433
|
+
cacheWriteInputTokens?: number;
|
|
434
|
+
outputTokens: number;
|
|
435
|
+
/** Reasoning tokens included in `outputTokens`. */
|
|
436
|
+
reasoningTokens?: number;
|
|
437
|
+
totalTokens: number;
|
|
438
|
+
calls: number;
|
|
439
|
+
}
|
|
440
|
+
interface OptimizationMethodProvenance {
|
|
441
|
+
/** External optimizer package. */
|
|
442
|
+
source: OptimizationPackageSource;
|
|
443
|
+
/** Python bridge package that invoked the optimizer. */
|
|
444
|
+
bridge?: OptimizationPackageSource;
|
|
445
|
+
/** Custom engine modules imported by the optimizer. */
|
|
446
|
+
modules?: OptimizationModuleSource[];
|
|
447
|
+
/** Python implementation used by the bridge process. */
|
|
448
|
+
python?: OptimizationPythonRuntime;
|
|
449
|
+
/** Exact model identifier configured for optimizer-owned model calls. */
|
|
450
|
+
optimizerModel?: string;
|
|
451
|
+
runId: string;
|
|
452
|
+
/** Content identity shared by compatible resumptions. */
|
|
453
|
+
compatibleRunId?: string;
|
|
454
|
+
resumed: boolean;
|
|
455
|
+
evaluationCount: number;
|
|
456
|
+
artifactDir: string;
|
|
457
|
+
tokenUsage?: OptimizationTokenUsage;
|
|
458
|
+
}
|
|
459
|
+
/** Shared inputs for one optimization method. Final test data is absent. */
|
|
460
|
+
interface OptimizationMethodInput<TScenario extends Scenario, TArtifact> {
|
|
461
|
+
/** Surface every method starts from. */
|
|
462
|
+
readonly baselineSurface: MutableSurface;
|
|
463
|
+
/** Evidence used to author or fit candidates. */
|
|
464
|
+
readonly trainScenarios: readonly TScenario[];
|
|
465
|
+
/** Data used for candidate acceptance, early stopping, and model selection. */
|
|
466
|
+
readonly selectionScenarios: readonly TScenario[];
|
|
467
|
+
/** Runs one scenario with a candidate surface. */
|
|
468
|
+
readonly dispatchWithSurface: (surface: MutableSurface, scenario: TScenario, ctx: DispatchContext) => Promise<TArtifact>;
|
|
469
|
+
/** Scores artifacts produced by `dispatchWithSurface`. */
|
|
470
|
+
readonly judges: readonly JudgeConfig<TArtifact, TScenario>[];
|
|
471
|
+
/** Method-specific artifacts are written below this directory. */
|
|
472
|
+
readonly runDir: string;
|
|
473
|
+
readonly seed: number;
|
|
474
|
+
/** Shared defaults for every method. A method may override them explicitly. */
|
|
475
|
+
readonly runOptions: Readonly<OptimizationMethodRunOptions<TScenario, TArtifact>>;
|
|
476
|
+
/** Durable spend account shared by every method and final scoring. */
|
|
477
|
+
readonly costLedger: CostLedgerHandle;
|
|
478
|
+
}
|
|
479
|
+
interface OptimizationMethodResult {
|
|
480
|
+
/** Surface selected without using the final test partition. */
|
|
481
|
+
winnerSurface: MutableSurface;
|
|
482
|
+
/** Optimization spend. Excludes final test scoring. */
|
|
483
|
+
cost: ComparisonCost;
|
|
484
|
+
/** Optimization duration. Excludes final test scoring. */
|
|
485
|
+
durationMs?: number;
|
|
486
|
+
/** Exact external implementation and run identity, when the method uses one. */
|
|
487
|
+
provenance?: OptimizationMethodProvenance;
|
|
488
|
+
}
|
|
489
|
+
/** A complete optimization method, including candidate generation and selection. */
|
|
490
|
+
interface OptimizationMethod<TScenario extends Scenario = Scenario, TArtifact = unknown> {
|
|
491
|
+
/** Unique, trimmed display name. Its normalized form must also be unique. */
|
|
492
|
+
name: string;
|
|
493
|
+
optimize: (input: OptimizationMethodInput<TScenario, TArtifact>) => Promise<OptimizationMethodResult>;
|
|
494
|
+
}
|
|
495
|
+
interface OptimizationMethodScore {
|
|
496
|
+
name: string;
|
|
497
|
+
/** Mean final-test composite of the baseline (identical across methods). */
|
|
498
|
+
baselineComposite: number;
|
|
499
|
+
/** Mean final-test composite of this method's selected surface. */
|
|
500
|
+
winnerComposite: number;
|
|
501
|
+
/** Mean per-scenario final-test lift (winner minus baseline). */
|
|
502
|
+
lift: number;
|
|
503
|
+
/** Simultaneous paired-bootstrap interval for per-scenario lift.
|
|
504
|
+
* `low > 0` excludes zero after adjustment for all reported contrasts. */
|
|
505
|
+
liftCi: {
|
|
506
|
+
low: number;
|
|
507
|
+
high: number;
|
|
508
|
+
};
|
|
509
|
+
/** Optimization spend reported by the method. Excludes final test scoring. */
|
|
510
|
+
optimizationCost: ComparisonCost;
|
|
511
|
+
/** Optimization duration reported by the method. Excludes final test scoring. */
|
|
512
|
+
durationMs?: number;
|
|
513
|
+
/** Exact external implementation and run identity, when reported by the method. */
|
|
514
|
+
provenance?: OptimizationMethodProvenance;
|
|
515
|
+
/** Paired final-test values used to compute lift and its interval. */
|
|
516
|
+
scenarioScores: Array<{
|
|
517
|
+
scenarioId: string;
|
|
518
|
+
baselineComposite: number;
|
|
519
|
+
winnerComposite: number;
|
|
520
|
+
lift: number;
|
|
521
|
+
}>;
|
|
522
|
+
winnerSurface: MutableSurface;
|
|
523
|
+
/** 1-based, by descending lift. */
|
|
524
|
+
rank: number;
|
|
525
|
+
}
|
|
526
|
+
interface OptimizationMethodPairwise {
|
|
527
|
+
/** Higher-ranked method. */
|
|
528
|
+
a: string;
|
|
529
|
+
b: string;
|
|
530
|
+
/** Mean per-scenario untouched-test delta (a − b). */
|
|
531
|
+
deltaMean: number;
|
|
532
|
+
low: number;
|
|
533
|
+
high: number;
|
|
534
|
+
/** `a` if the CI clears 0, `b` if it is entirely negative, else `'tie'`. */
|
|
535
|
+
favored: string;
|
|
536
|
+
}
|
|
537
|
+
interface OptimizationMethodComparison {
|
|
538
|
+
/** Sorted by descending lift; `rank` set accordingly. */
|
|
539
|
+
scores: OptimizationMethodScore[];
|
|
540
|
+
best: OptimizationMethodScore;
|
|
541
|
+
/** Best vs each other method, using simultaneous paired-bootstrap intervals. */
|
|
542
|
+
pairwise: OptimizationMethodPairwise[];
|
|
543
|
+
testScenarioIds: string[];
|
|
544
|
+
/** Sum of the costs reported by every optimization method. */
|
|
545
|
+
optimizationCost: ComparisonCost;
|
|
546
|
+
/** Baseline and distinct winner scoring on the final test partition. */
|
|
547
|
+
testCost: ComparisonCost;
|
|
548
|
+
/** Optimization plus final test scoring. */
|
|
549
|
+
totalCost: ComparisonCost;
|
|
550
|
+
/** Caller-requested simultaneous coverage across all reported contrasts. */
|
|
551
|
+
confidence: number;
|
|
552
|
+
/** Bonferroni-adjusted confidence used for each bootstrap interval. */
|
|
553
|
+
intervalConfidence: number;
|
|
554
|
+
/** Method-vs-baseline plus all possible method-vs-method contrasts. */
|
|
555
|
+
comparisonCount: number;
|
|
556
|
+
/** Deterministic bootstrap and campaign seed. */
|
|
557
|
+
seed: number;
|
|
558
|
+
/** Bootstrap draws used for each interval. */
|
|
559
|
+
resamples: number;
|
|
560
|
+
/** Agent runs averaged within each test scenario before resampling scenarios. */
|
|
561
|
+
reps: number;
|
|
562
|
+
}
|
|
563
|
+
interface CompareOptimizationMethodsOptions<TScenario extends Scenario, TArtifact> extends Omit<RunCampaignOptions<TScenario, TArtifact>, 'dispatch' | 'judges' | 'scenarios'> {
|
|
564
|
+
methods: OptimizationMethod<TScenario, TArtifact>[];
|
|
565
|
+
baselineSurface: MutableSurface;
|
|
566
|
+
/** Evidence used by every optimizer to author or fit candidates. */
|
|
567
|
+
trainScenarios: TScenario[];
|
|
568
|
+
/** Candidate acceptance, early-stopping, and optimizer-selection data. */
|
|
569
|
+
selectionScenarios: TScenario[];
|
|
570
|
+
/** Untouched final comparison data. Never passed to an optimization method. */
|
|
571
|
+
testScenarios: TScenario[];
|
|
572
|
+
/** Scores a surface on a scenario. The methods and final test share this function. */
|
|
573
|
+
dispatchWithSurface: (surface: MutableSurface, scenario: TScenario, ctx: DispatchContext) => Promise<TArtifact>;
|
|
574
|
+
judges: JudgeConfig<TArtifact, TScenario>[];
|
|
575
|
+
/** Bootstrap resamples for the lift intervals. Default is at least 2000 and
|
|
576
|
+
* rises when the requested simultaneous confidence needs finer tails. */
|
|
577
|
+
resamples?: number;
|
|
578
|
+
/** Shared defaults for each method's train and selection campaigns. */
|
|
579
|
+
optimizationRunOptions?: OptimizationMethodRunOptions<TScenario, TArtifact>;
|
|
580
|
+
/** Number of optimization methods to run concurrently. Default 1. */
|
|
581
|
+
optimizationConcurrency?: number;
|
|
582
|
+
/** Simultaneous confidence across method-vs-baseline and method-vs-method contrasts.
|
|
583
|
+
* Each bootstrap interval is Bonferroni-adjusted. Default 0.95. */
|
|
584
|
+
confidence?: number;
|
|
585
|
+
/** Shared spend limit across every method's optimizer and evaluation calls plus final scoring. */
|
|
586
|
+
costCeiling?: number;
|
|
587
|
+
}
|
|
588
|
+
/**
|
|
589
|
+
* Compare complete optimization methods on disjoint train, selection, and final test data.
|
|
590
|
+
*/
|
|
591
|
+
declare function compareOptimizationMethods<TScenario extends Scenario, TArtifact>(opts: CompareOptimizationMethodsOptions<TScenario, TArtifact>): Promise<OptimizationMethodComparison>;
|
|
592
|
+
/** Keep the cost fields a custom optimization method must report. */
|
|
593
|
+
declare function costFromLedgerSummary(summary: CostLedgerSummary): ComparisonCost;
|
|
594
|
+
/** Preserve every optimizer token class while keeping total input and output explicit. */
|
|
595
|
+
declare function optimizationTokenUsageFromSummary(summary: CostLedgerSummary, receipts: readonly CostReceipt[]): OptimizationTokenUsage | undefined;
|
|
596
|
+
//#endregion
|
|
597
|
+
//#region src/campaign/external-text-evaluation.d.ts
|
|
598
|
+
interface ExternalOptimizationExample {
|
|
599
|
+
id: string;
|
|
600
|
+
data: unknown;
|
|
601
|
+
}
|
|
602
|
+
interface ExternalTextEvaluationResponse {
|
|
603
|
+
score: number;
|
|
604
|
+
info: {
|
|
605
|
+
scenarioId: string;
|
|
606
|
+
dimensions: Record<string, number>;
|
|
607
|
+
notes?: string;
|
|
608
|
+
artifact?: unknown;
|
|
609
|
+
};
|
|
610
|
+
}
|
|
611
|
+
//#endregion
|
|
612
|
+
//#region src/campaign/external-text-optimization-contract.d.ts
|
|
613
|
+
interface ExternalTextOptimizerContext {
|
|
614
|
+
readonly runId: string;
|
|
615
|
+
readonly name: string;
|
|
616
|
+
readonly objective: string;
|
|
617
|
+
readonly evaluationId: string;
|
|
618
|
+
readonly background?: string;
|
|
619
|
+
readonly seedCandidate: ExternalTextCandidate;
|
|
620
|
+
readonly trainSet: readonly ExternalOptimizationExample[];
|
|
621
|
+
readonly selectionSet: readonly ExternalOptimizationExample[];
|
|
622
|
+
readonly maxEvaluations: number;
|
|
623
|
+
readonly seed: number;
|
|
624
|
+
/** Stable directory for optimizer checkpoints from compatible attempts. */
|
|
625
|
+
readonly stateDir: string;
|
|
626
|
+
readonly restoreRequested: boolean;
|
|
627
|
+
readonly artifactDir: string;
|
|
628
|
+
readonly signal: AbortSignal;
|
|
629
|
+
/** Record every optimizer-owned paid call through this attributed account. */
|
|
630
|
+
readonly cost: CampaignCostMeter;
|
|
631
|
+
readonly evaluate: (request: ExternalTextEvaluationRequest) => Promise<ExternalTextEvaluationResponse>;
|
|
632
|
+
}
|
|
633
|
+
interface ExternalTextOptimizerResult {
|
|
634
|
+
bestCandidate: ExternalTextCandidate;
|
|
635
|
+
resumed: boolean;
|
|
636
|
+
costAccounting: {
|
|
637
|
+
kind: 'metered';
|
|
638
|
+
} | {
|
|
639
|
+
kind: 'no-paid-work';
|
|
640
|
+
} | {
|
|
641
|
+
kind: 'external';
|
|
642
|
+
reason: string;
|
|
643
|
+
};
|
|
644
|
+
}
|
|
645
|
+
/**
|
|
646
|
+
* Configuration for adapting another text optimizer.
|
|
647
|
+
*
|
|
648
|
+
* `run` owns search. Agent Eval owns split isolation, bounded candidate
|
|
649
|
+
* evaluation, exact cost collection, provenance, and final comparison.
|
|
650
|
+
*/
|
|
651
|
+
interface ExternalTextOptimizationMethodConfig<TScenario extends Scenario, TArtifact = unknown> {
|
|
652
|
+
name: string;
|
|
653
|
+
source: Omit<OptimizationPackageSource, 'evidence'>;
|
|
654
|
+
objective: string;
|
|
655
|
+
evaluationId: string;
|
|
656
|
+
background?: string;
|
|
657
|
+
maxEvaluations: number;
|
|
658
|
+
/** Hard limit for calls made through `context.cost`. Use 0 for no paid work. */
|
|
659
|
+
maxOptimizerCostUsd: number;
|
|
660
|
+
/** Abort `context.signal` after this duration. Default: 30 minutes. */
|
|
661
|
+
timeoutMs?: number;
|
|
662
|
+
/** Default: `never`. Compatible runs reuse one state directory. */
|
|
663
|
+
resume?: ExternalOptimizerResumeMode;
|
|
664
|
+
maxCandidateChars?: number;
|
|
665
|
+
maxEvidenceChars?: number;
|
|
666
|
+
describeScenario?: (scenario: TScenario) => unknown;
|
|
667
|
+
describeArtifact?: (artifact: TArtifact, scenario: TScenario) => unknown;
|
|
668
|
+
run: (context: ExternalTextOptimizerContext) => Promise<ExternalTextOptimizerResult>;
|
|
669
|
+
}
|
|
670
|
+
//#endregion
|
|
671
|
+
//#region src/campaign/external-text-optimization.d.ts
|
|
672
|
+
/**
|
|
673
|
+
* Adapt a third-party text optimizer without reimplementing its search.
|
|
674
|
+
*
|
|
675
|
+
* The callback never receives final test cases. Calls to `evaluate` are
|
|
676
|
+
* counted before execution and stop at `maxEvaluations`.
|
|
677
|
+
*/
|
|
678
|
+
declare function externalTextOptimizationMethod<TScenario extends Scenario, TArtifact>(config: ExternalTextOptimizationMethodConfig<TScenario, TArtifact>): OptimizationMethod<TScenario, TArtifact>;
|
|
679
|
+
//#endregion
|
|
680
|
+
//#region src/campaign/gates/compose.d.ts
|
|
681
|
+
/** Compose gates — all must `ship` for the composite to `ship`. First
|
|
682
|
+
* non-ship verdict short-circuits the composite verdict, but ALL gates run
|
|
683
|
+
* (so the result records every gate's reason — useful for diagnostics). */
|
|
684
|
+
declare function composeGate<TArtifact = unknown, TScenario extends Scenario = Scenario>(...gates: Array<Gate<TArtifact, TScenario>>): Gate<TArtifact, TScenario>;
|
|
685
|
+
//#endregion
|
|
686
|
+
//#region src/canary.d.ts
|
|
687
|
+
type CanaryKind = 'silent_judge_fallback' | 'judge_calibration_drift' | 'distribution_shift';
|
|
688
|
+
type CanarySeverity = 'info' | 'warn' | 'error';
|
|
689
|
+
interface CanaryAlert {
|
|
690
|
+
kind: CanaryKind;
|
|
691
|
+
severity: CanarySeverity;
|
|
692
|
+
message: string;
|
|
693
|
+
/** Numbers that informed the decision — drop straight into a
|
|
694
|
+
* dashboard / paper figure. */
|
|
695
|
+
evidence: Record<string, unknown>;
|
|
696
|
+
}
|
|
697
|
+
interface CanaryReport {
|
|
698
|
+
alerts: CanaryAlert[];
|
|
699
|
+
/** Per-kind summary count. */
|
|
700
|
+
counts: Record<CanaryKind, number>;
|
|
701
|
+
/** Whether each enabled detector had enough observations to run. */
|
|
702
|
+
evaluations: CanaryEvaluation[];
|
|
703
|
+
}
|
|
704
|
+
interface CanaryEvaluation {
|
|
705
|
+
kind: CanaryKind;
|
|
706
|
+
status: 'evaluated' | 'not_evaluated';
|
|
707
|
+
observations: number;
|
|
708
|
+
reason?: string;
|
|
709
|
+
}
|
|
710
|
+
interface CanaryOptions {
|
|
711
|
+
/**
|
|
712
|
+
* Silent-fallback detection.
|
|
713
|
+
* - `constant`: confidence value treated as the fallback signal.
|
|
714
|
+
* Default 0.30 (matches the soft-fail default in
|
|
715
|
+
* `propose-review.ts`).
|
|
716
|
+
* - `consecutiveThreshold`: trip the alert after this many
|
|
717
|
+
* consecutive runs at `constant` (or `fallback === true`).
|
|
718
|
+
* Default 3.
|
|
719
|
+
*/
|
|
720
|
+
silentFallback?: {
|
|
721
|
+
constant?: number;
|
|
722
|
+
consecutiveThreshold?: number;
|
|
723
|
+
/** Floating-point tolerance when comparing against `constant`. */
|
|
724
|
+
epsilon?: number;
|
|
725
|
+
};
|
|
726
|
+
/**
|
|
727
|
+
* Calibration-drift detection.
|
|
728
|
+
* - `historyWindow`: number of past runs (oldest-first) treated as
|
|
729
|
+
* the historical baseline. Default 50.
|
|
730
|
+
* - `recentWindow`: number of recent runs (newest-first) compared
|
|
731
|
+
* against history. Default 20.
|
|
732
|
+
* - `ksAlpha`: alpha for the KS statistic vs critical value.
|
|
733
|
+
* Default 0.05.
|
|
734
|
+
* - `minRecent`: minimum recent runs required to even attempt the
|
|
735
|
+
* check. Default 10.
|
|
736
|
+
*/
|
|
737
|
+
calibrationDrift?: {
|
|
738
|
+
historyWindow?: number;
|
|
739
|
+
recentWindow?: number;
|
|
740
|
+
ksAlpha?: number;
|
|
741
|
+
minRecent?: number;
|
|
742
|
+
};
|
|
743
|
+
/**
|
|
744
|
+
* Distribution-shift detection.
|
|
745
|
+
* - `category`: function that maps a run to a categorical bucket.
|
|
746
|
+
* Required to enable this canary; if omitted the chi-square check
|
|
747
|
+
* is skipped entirely.
|
|
748
|
+
* - `chiSquareAlpha`: alpha. Default 0.05.
|
|
749
|
+
* - `historyWindow`, `recentWindow`, `minRecent`: like above.
|
|
750
|
+
*/
|
|
751
|
+
distributionShift?: {
|
|
752
|
+
category: (run: RunRecord) => string | null;
|
|
753
|
+
chiSquareAlpha?: number;
|
|
754
|
+
historyWindow?: number;
|
|
755
|
+
recentWindow?: number;
|
|
756
|
+
minRecent?: number;
|
|
757
|
+
};
|
|
758
|
+
}
|
|
759
|
+
/**
|
|
760
|
+
* Run all configured canaries against a chronological run list.
|
|
761
|
+
* Runs MUST be sorted oldest-to-newest by the caller — the order of
|
|
762
|
+
* the input is used to define "recent" vs "historical" windows.
|
|
763
|
+
*/
|
|
764
|
+
declare function runCanaries(runs: RunRecord[], opts?: CanaryOptions): CanaryReport;
|
|
765
|
+
//#endregion
|
|
766
|
+
//#region src/red-team.d.ts
|
|
767
|
+
type RedTeamCategory = 'prompt_injection_direct' | 'prompt_injection_indirect' | 'jailbreak_persona' | 'jailbreak_dan' | 'pii_leak' | 'permission_escalation' | 'data_exfil' | 'policy_override';
|
|
768
|
+
interface RedTeamPayload {
|
|
769
|
+
category: RedTeamCategory;
|
|
770
|
+
/** The hostile input the agent sees. */
|
|
771
|
+
input: string;
|
|
772
|
+
/**
|
|
773
|
+
* What a safe agent should do: "refuse" means explicit refusal,
|
|
774
|
+
* "ignore" means process benignly without executing the attack.
|
|
775
|
+
*/
|
|
776
|
+
expected: 'refuse' | 'ignore';
|
|
777
|
+
/** Substring(s) that MUST NOT appear in output. */
|
|
778
|
+
forbiddenStrings?: string[];
|
|
779
|
+
/** Tool names the agent MUST NOT call for this scenario. */
|
|
780
|
+
forbiddenTools?: string[];
|
|
781
|
+
}
|
|
782
|
+
interface RedTeamCase extends DatasetScenario {
|
|
783
|
+
payload: RedTeamPayload;
|
|
784
|
+
}
|
|
785
|
+
interface RedTeamFinding {
|
|
786
|
+
scenarioId: string;
|
|
787
|
+
category: RedTeamCategory;
|
|
788
|
+
passed: boolean;
|
|
789
|
+
reason: string;
|
|
790
|
+
evidence?: string;
|
|
791
|
+
}
|
|
792
|
+
interface RedTeamReport {
|
|
793
|
+
findings: RedTeamFinding[];
|
|
794
|
+
passRateByCategory: Record<RedTeamCategory, number>;
|
|
795
|
+
overallPassRate: number;
|
|
796
|
+
}
|
|
797
|
+
/** Ship a canonical, small corpus. Consumers extend via `extendCorpus`. */
|
|
798
|
+
declare const DEFAULT_RED_TEAM_CORPUS: RedTeamCase[];
|
|
799
|
+
declare function redTeamDataset(extraCases?: RedTeamCase[]): Dataset;
|
|
800
|
+
/**
|
|
801
|
+
* Score a single agent response against the red-team case.
|
|
802
|
+
* `toolCalls` should be the names of tools the agent invoked during the run.
|
|
803
|
+
*/
|
|
804
|
+
declare function scoreRedTeamOutput(output: string, toolCalls: string[], rtCase: RedTeamCase): RedTeamFinding;
|
|
805
|
+
/** Aggregate red-team findings into per-category pass rates. */
|
|
806
|
+
declare function redTeamReport(findings: RedTeamFinding[]): RedTeamReport;
|
|
807
|
+
/**
|
|
808
|
+
* Extract the tool-call names from a corpus run — convenience for the
|
|
809
|
+
* common pipeline (run the scenario → score the run).
|
|
810
|
+
*/
|
|
811
|
+
declare function toolNamesForRun(store: TraceStore, runId: string): Promise<string[]>;
|
|
812
|
+
//#endregion
|
|
813
|
+
//#region src/campaign/gates/default-production-gate.d.ts
|
|
814
|
+
type DefaultProductionGateCheck = 'dimension-regression' | 'budget' | 'red-team' | 'reward-hacking' | 'canary';
|
|
815
|
+
type DefaultProductionRewardHackingOptions = Omit<DetectRewardHackingInput, 'runs' | 'truthOf'> & {
|
|
816
|
+
truthOf: NonNullable<DetectRewardHackingInput['truthOf']>;
|
|
817
|
+
};
|
|
818
|
+
interface DefaultProductionGateOptions {
|
|
819
|
+
/** Required: scenarios held out from training; substrate compares
|
|
820
|
+
* candidate-on-holdout vs baseline-on-holdout. */
|
|
821
|
+
holdoutScenarios: Scenario[];
|
|
822
|
+
/** Minimum held-out lift the **paired-bootstrap CI lower bound** must clear
|
|
823
|
+
* to ship — NOT a point estimate. Default 0 ⇒ "confidently positive at the
|
|
824
|
+
* confidence level". Interpreted in the judge's native composite scale (set
|
|
825
|
+
* e.g. 2 for a 0-100 rubric to require a ≥2-point significant gain). */
|
|
826
|
+
deltaThreshold?: number;
|
|
827
|
+
/** Confidence level for the held-out + dimension bootstraps. Default 0.95. */
|
|
828
|
+
confidence?: number;
|
|
829
|
+
/** Bootstrap resamples. Default 2000. */
|
|
830
|
+
bootstrapResamples?: number;
|
|
831
|
+
/** Fixed bootstrap seed for a deterministic verdict. Default 1337. */
|
|
832
|
+
bootstrapSeed?: number;
|
|
833
|
+
/** Minimum paired holdout observations (scenarios × reps) before a
|
|
834
|
+
* significance claim is allowed; below it the gate HOLDS with `few_runs`
|
|
835
|
+
* rather than reading a degenerate CI. Default 3. */
|
|
836
|
+
minProductiveRuns?: number;
|
|
837
|
+
/** Ship statistic for the held-out significance test. Default `'mean'`
|
|
838
|
+
* (tie-robust — see `heldoutSignificance`). Pass `'median'` for
|
|
839
|
+
* outlier-robustness at the cost of tie-blindness. */
|
|
840
|
+
heldoutStatistic?: 'mean' | 'median';
|
|
841
|
+
/** Critical judge dimensions that must NOT significantly regress even when
|
|
842
|
+
* the net composite rises (anti-Goodhart). The gate HOLDS if any listed
|
|
843
|
+
* dimension's paired-delta CI lower bound < −`regressionTolerance`. E.g.
|
|
844
|
+
* `['hallucination_free']` for a legal agent. */
|
|
845
|
+
criticalDimensions?: string[];
|
|
846
|
+
/** Tolerance for the per-dimension regression guard, in the dimension's
|
|
847
|
+
* native scale. When omitted it auto-scales off observed magnitudes:
|
|
848
|
+
* 0.05 on [0,1], 5 on 0-100. */
|
|
849
|
+
regressionTolerance?: number;
|
|
850
|
+
/** Total $ budget for the complete improvement run. Requires
|
|
851
|
+
* `GateContext.costLedger`; missing or incomplete accounting holds. */
|
|
852
|
+
budgetUsd?: number;
|
|
853
|
+
/** Static artifact-screening cases. Only `expected: 'ignore'` cases without
|
|
854
|
+
* tool assertions are valid because this check does not dispatch case inputs
|
|
855
|
+
* or observe tool calls. */
|
|
856
|
+
redTeamBattery?: RedTeamCase[];
|
|
857
|
+
/** Shared run history, oldest first. Supplying history does not enable either
|
|
858
|
+
* monitoring check; configure `rewardHacking` and/or `canary` explicitly. */
|
|
859
|
+
recentRuns?: RunRecord[];
|
|
860
|
+
/** Enable reward-hacking monitoring with a caller-owned independent truth channel. */
|
|
861
|
+
rewardHacking?: DefaultProductionRewardHackingOptions;
|
|
862
|
+
/** Enable canary monitoring. Pass `{}` to use the canary defaults. */
|
|
863
|
+
canary?: CanaryOptions;
|
|
864
|
+
/** Optional checks that must be evaluated even when their normal input is
|
|
865
|
+
* absent. Configuring a check's input also makes that check required.
|
|
866
|
+
* Missing evidence always records `not_evaluated`; required unevaluated
|
|
867
|
+
* checks hold the release decision. Held-out significance is always required. */
|
|
868
|
+
requiredChecks?: DefaultProductionGateCheck[];
|
|
869
|
+
}
|
|
870
|
+
/**
|
|
871
|
+
* Opinionated production gate composing held-out significance, red-team, reward-hacking, and canary checks into a single `Gate.decide` decision.
|
|
872
|
+
*/
|
|
873
|
+
declare function defaultProductionGate<TArtifact, TScenario extends Scenario>(options: DefaultProductionGateOptions): Gate<TArtifact, TScenario>;
|
|
874
|
+
//#endregion
|
|
875
|
+
//#region src/campaign/gates/heldout-gate.d.ts
|
|
876
|
+
interface HeldOutGateOptions<TScenario extends Scenario = Scenario> {
|
|
877
|
+
scenarios: TScenario[];
|
|
878
|
+
/** Effect-size threshold the CI lower bound must clear, in the judge's native
|
|
879
|
+
* scale. Default 0.5. Equality holds; CI.low must be greater than this value. */
|
|
880
|
+
deltaThreshold?: number;
|
|
881
|
+
/** Bootstrap CI confidence. Default 0.95. */
|
|
882
|
+
confidence?: number;
|
|
883
|
+
/** Minimum paired holdout observations to claim significance. Default 3. */
|
|
884
|
+
minProductiveRuns?: number;
|
|
885
|
+
/** Bootstrap resamples. Default 2000. */
|
|
886
|
+
resamples?: number;
|
|
887
|
+
/** Fixed bootstrap seed for deterministic verdicts. Default 1337. */
|
|
888
|
+
bootstrapSeed?: number;
|
|
889
|
+
}
|
|
890
|
+
/**
|
|
891
|
+
* Composable held-out gate: ships only when the PAIRED bootstrap CI lower bound
|
|
892
|
+
* of the candidate-minus-baseline composite delta clears `deltaThreshold`.
|
|
893
|
+
*/
|
|
894
|
+
declare function heldOutGate<TArtifact, TScenario extends Scenario>(options: HeldOutGateOptions<TScenario>): Gate<TArtifact, TScenario>;
|
|
895
|
+
//#endregion
|
|
896
|
+
//#region src/campaign/gates/power-preflight.d.ts
|
|
897
|
+
/**
|
|
898
|
+
* Power preflight — "can this budget detect the effect you are hunting?"
|
|
899
|
+
*
|
|
900
|
+
* The failure it prevents (measured, twice): a live prompt-improvement campaign ran
|
|
901
|
+
* 333 sandbox cells over 5.6 hours and produced a +0.08 holdout lift the ship gate
|
|
902
|
+
* (paired bootstrap, CI.low > 0.05) could not distinguish from zero — because at
|
|
903
|
+
* that holdout size and worker variance the MINIMUM DETECTABLE lift was larger than
|
|
904
|
+
* any effect a prompt change plausibly produces. The budget was spent learning what
|
|
905
|
+
* a 30-second calculation on the baseline cells already knew. No eval framework we
|
|
906
|
+
* know of surfaces this; every underpowered improvement run everywhere ends in an
|
|
907
|
+
* uninformative "hold".
|
|
908
|
+
*
|
|
909
|
+
* Model: the ship rule is `CI.low(paired Δ) > deltaThreshold`. Approximating the
|
|
910
|
+
* bootstrap CI as normal, `CI.low ≈ effect − z·sd_Δ/√n`, so the smallest shippable
|
|
911
|
+
* true effect is `MDE = deltaThreshold + z·sd_Δ/√n`. The paired-delta SD is unknown
|
|
912
|
+
* before the candidate exists; we bound it by the zero-correlation case
|
|
913
|
+
* `sd_Δ ≤ √2·sd_baseline` — a CONSERVATIVE (upper) MDE, which is the correct
|
|
914
|
+
* direction for a warning. Pairing is per cell (`scenario:rep`), so reps multiply n.
|
|
915
|
+
*
|
|
916
|
+
* Standalone by design: feed it any baseline composites (a `gate:'none'` run, a
|
|
917
|
+
* live-proof table) BEFORE budgeting the real search; `selfImprove` also attaches
|
|
918
|
+
* it to every result and warns when the run was structurally unable to ship.
|
|
919
|
+
*/
|
|
920
|
+
interface PowerPreflightOptions {
|
|
921
|
+
/** Per-cell baseline composites on the HOLDOUT scenarios (one per scenario:rep cell). */
|
|
922
|
+
baselineComposites: number[];
|
|
923
|
+
/** Paired observations the budgeted comparison will produce
|
|
924
|
+
* (holdout scenarios × reps). Defaults to `baselineComposites.length`. */
|
|
925
|
+
pairedN?: number;
|
|
926
|
+
/** The ship gate's effect-size threshold. Default 0.05 (defaultProductionGate). */
|
|
927
|
+
deltaThreshold?: number;
|
|
928
|
+
/** CI confidence the gate uses. Default 0.95. */
|
|
929
|
+
confidence?: number;
|
|
930
|
+
/** True when the holdout is scored by the SAME judge/scorer family as the gate
|
|
931
|
+
* (selfImprove's default composition — one judge scores everything). Under a
|
|
932
|
+
* shared channel, raising paired n reduces only the IDIOSYNCRATIC noise share;
|
|
933
|
+
* systematic judge bias is untouched, so the MDE here is a lower bound and the
|
|
934
|
+
* only full debiaser is an independent second scoring channel
|
|
935
|
+
* (recursive-self-improvement S1c, closed form in EXP-023 P0). Default false. */
|
|
936
|
+
sharedScorerChannel?: boolean;
|
|
937
|
+
}
|
|
938
|
+
interface PowerPreflight {
|
|
939
|
+
/** Paired observations the comparison will have. */
|
|
940
|
+
n: number;
|
|
941
|
+
/** Baseline per-cell composite standard deviation (the variance the effect must beat). */
|
|
942
|
+
sd: number;
|
|
943
|
+
/** Minimum detectable lift: the smallest TRUE effect the gate could ship at this budget. */
|
|
944
|
+
mde: number;
|
|
945
|
+
/** Baseline holdout composite mean. */
|
|
946
|
+
baselineMean: number;
|
|
947
|
+
/** Headroom to a perfect 1.0 composite (the largest achievable lift on a [0,1] judge). */
|
|
948
|
+
headroom: number;
|
|
949
|
+
/** True when even the largest achievable effect (headroom) is below the MDE —
|
|
950
|
+
* the run is structurally unable to ship regardless of proposal quality.
|
|
951
|
+
* Only asserted for [0,1]-scaled judges (see `scaleAssumed`). */
|
|
952
|
+
underpowered: boolean;
|
|
953
|
+
/** True when composites look [0,1]-scaled; headroom/underpowered are only
|
|
954
|
+
* meaningful under that convention (0-100 judges get mde/sd/n but no verdict). */
|
|
955
|
+
scaleAssumed: boolean;
|
|
956
|
+
deltaThreshold: number;
|
|
957
|
+
confidence: number;
|
|
958
|
+
/** Set when the holdout shares the gate's scoring channel: more cells cannot
|
|
959
|
+
* buy back systematic judge bias — treat the MDE as a lower bound. */
|
|
960
|
+
sharedChannelCaveat?: string;
|
|
961
|
+
/** One actionable sentence for humans and logs. */
|
|
962
|
+
recommendation: string;
|
|
963
|
+
}
|
|
964
|
+
/** Estimate the minimum detectable lift a paired-holdout improvement run can
|
|
965
|
+
* ship at a given budget, from the baseline holdout composites — call it BEFORE
|
|
966
|
+
* spending a search to learn whether the effect you are hunting is even
|
|
967
|
+
* observable at this holdout size and worker variance. */
|
|
968
|
+
declare function powerPreflight(opts: PowerPreflightOptions): PowerPreflight;
|
|
969
|
+
//#endregion
|
|
970
|
+
//#region src/pareto.d.ts
|
|
971
|
+
/**
|
|
972
|
+
* Pareto frontier — multi-objective optimization over candidate runs.
|
|
973
|
+
*
|
|
974
|
+
* Lifted from ADC pareto.ts and blueprint-agent frontier.ts. When you're
|
|
975
|
+
* trading off (cost, latency, quality) or (passRate, tokenBudget,
|
|
976
|
+
* ttfb), you rarely have a single "winner" — you have a set of
|
|
977
|
+
* non-dominated candidates. This module exposes:
|
|
978
|
+
*
|
|
979
|
+
* - `paretoFrontier`: filter a set of candidates to the non-dominated ones
|
|
980
|
+
* - `dominates`: does A dominate B across all objectives?
|
|
981
|
+
*
|
|
982
|
+
* Each objective is declared with a direction: 'maximize' (higher=better)
|
|
983
|
+
* or 'minimize' (lower=better). Candidates are any object; pass an
|
|
984
|
+
* `objective(candidate)` accessor.
|
|
985
|
+
*/
|
|
986
|
+
type Direction = 'maximize' | 'minimize';
|
|
987
|
+
interface Objective<T> {
|
|
988
|
+
/** Stable label used in reports. */
|
|
989
|
+
name: string;
|
|
990
|
+
direction: Direction;
|
|
991
|
+
value: (candidate: T) => number;
|
|
992
|
+
}
|
|
993
|
+
interface ParetoResult<T> {
|
|
994
|
+
frontier: T[];
|
|
995
|
+
dominated: T[];
|
|
996
|
+
/** Index map: frontier[i] dominates each of dominatedBy[i]. */
|
|
997
|
+
dominanceMap: Array<{
|
|
998
|
+
dominator: T;
|
|
999
|
+
dominated: T[];
|
|
1000
|
+
}>;
|
|
1001
|
+
}
|
|
1002
|
+
/** Does candidate A weakly dominate B — strictly better on at least one objective and no worse on any? */
|
|
1003
|
+
declare function dominates<T>(a: T, b: T, objectives: Objective<T>[]): boolean;
|
|
1004
|
+
/**
|
|
1005
|
+
* Compute the non-dominated frontier. Candidates with NaN/Infinity on any
|
|
1006
|
+
* objective are excluded (can't rank them). A candidate enters the frontier
|
|
1007
|
+
* iff no other candidate dominates it.
|
|
1008
|
+
*/
|
|
1009
|
+
declare function paretoFrontier<T>(candidates: T[], objectives: Objective<T>[]): ParetoResult<T>;
|
|
1010
|
+
/**
|
|
1011
|
+
* Weighted-sum scalarisation. Use as a tie-break / single-winner selector
|
|
1012
|
+
* when callers don't want to consume a frontier. Each objective contributes
|
|
1013
|
+
* its normalised value (0..1 via min-max across the candidate pool) times
|
|
1014
|
+
* its weight; missing weights default to 1/N.
|
|
1015
|
+
*
|
|
1016
|
+
* Direction is honoured automatically — `minimize` axes have their values
|
|
1017
|
+
* inverted before scaling so "higher scalar = better" always holds.
|
|
1018
|
+
*/
|
|
1019
|
+
declare function scalarScore<T>(candidates: T[], objectives: Objective<T>[], options?: {
|
|
1020
|
+
weights?: Partial<Record<string, number>>;
|
|
1021
|
+
}): Array<{
|
|
1022
|
+
candidate: T;
|
|
1023
|
+
score: number;
|
|
1024
|
+
}>;
|
|
1025
|
+
/**
|
|
1026
|
+
* NSGA-II crowding distance — secondary sort for ties on the frontier.
|
|
1027
|
+
*
|
|
1028
|
+
* When the Pareto front collapses to a single point (or many candidates tie
|
|
1029
|
+
* on dominance), naive selection picks arbitrarily and the population
|
|
1030
|
+
* degenerates over generations. NSGA-II preserves diversity by preferring
|
|
1031
|
+
* candidates with more empty space around them on the frontier.
|
|
1032
|
+
*
|
|
1033
|
+
* Returns an array of `{ candidate, distance }` in the SAME order as the
|
|
1034
|
+
* input. Higher distance = more isolated = should be preferred when
|
|
1035
|
+
* preserving diversity.
|
|
1036
|
+
*/
|
|
1037
|
+
declare function crowdingDistance<T>(candidates: T[], objectives: Objective<T>[]): Array<{
|
|
1038
|
+
candidate: T;
|
|
1039
|
+
distance: number;
|
|
1040
|
+
}>;
|
|
1041
|
+
/**
|
|
1042
|
+
* Pareto frontier with tie-break by crowding distance — the canonical
|
|
1043
|
+
* NSGA-II selection step. Returns the frontier sorted by descending crowding
|
|
1044
|
+
* distance so callers can `.slice(0, k)` to pick K diverse winners.
|
|
1045
|
+
*/
|
|
1046
|
+
declare function paretoFrontierWithCrowding<T>(candidates: T[], objectives: Objective<T>[]): Array<{
|
|
1047
|
+
candidate: T;
|
|
1048
|
+
distance: number;
|
|
1049
|
+
}>;
|
|
1050
|
+
//#endregion
|
|
1051
|
+
//#region src/campaign/gates/promotion-policy.d.ts
|
|
1052
|
+
/** Where an objective's per-cell scalar comes from. `composite` reads the
|
|
1053
|
+
* judge's composite; `dimension` reads a named per-dimension score. */
|
|
1054
|
+
type ObjectiveSource = {
|
|
1055
|
+
kind: 'composite';
|
|
1056
|
+
} | {
|
|
1057
|
+
kind: 'dimension';
|
|
1058
|
+
dimension: string;
|
|
1059
|
+
};
|
|
1060
|
+
interface PromotionObjective {
|
|
1061
|
+
/** Stable label used in reports + `contributingGates`. */
|
|
1062
|
+
name: string;
|
|
1063
|
+
source: ObjectiveSource;
|
|
1064
|
+
/** 'maximize' (quality dims) or 'minimize' (error/risk/length dims). Orients
|
|
1065
|
+
* the paired delta so a positive bootstrap always means "candidate better". */
|
|
1066
|
+
direction: Direction;
|
|
1067
|
+
/** The good-direction paired-delta CI lower bound must EXCEED this to count
|
|
1068
|
+
* as a significant gain on this axis. Interpreted in the judge's native
|
|
1069
|
+
* scale. Default 0 (⇒ "confidently better"). */
|
|
1070
|
+
gainThreshold?: number;
|
|
1071
|
+
/** A floor breach (regression) is declared when the good-direction CI lower
|
|
1072
|
+
* bound is below −floorTolerance. When omitted it auto-scales off observed
|
|
1073
|
+
* magnitudes (0.05 on [0,1], 5 on 0-100), matching `dimensionRegressions`. */
|
|
1074
|
+
floorTolerance?: number;
|
|
1075
|
+
}
|
|
1076
|
+
/** Per-axis verdict from the good-direction paired bootstrap. */
|
|
1077
|
+
type AxisVerdict = 'improved' | 'regressed' | 'flat' | 'few_runs';
|
|
1078
|
+
interface AxisEvidence {
|
|
1079
|
+
name: string;
|
|
1080
|
+
source: ObjectiveSource;
|
|
1081
|
+
direction: Direction;
|
|
1082
|
+
/** Paired bootstrap on the GOOD-DIRECTION delta (oriented by `direction`):
|
|
1083
|
+
* a positive value means the candidate is better on this axis. */
|
|
1084
|
+
bootstrap: PairedBootstrapResult;
|
|
1085
|
+
/** Paired observations contributing to this axis. */
|
|
1086
|
+
n: number;
|
|
1087
|
+
gainThreshold: number;
|
|
1088
|
+
floorTolerance: number;
|
|
1089
|
+
verdict: AxisVerdict;
|
|
1090
|
+
}
|
|
1091
|
+
interface EvidenceVector {
|
|
1092
|
+
/** One entry per objective — NOTHING averaged across axes. */
|
|
1093
|
+
axes: AxisEvidence[];
|
|
1094
|
+
/** Smallest paired n across axes that produced observations — the binding
|
|
1095
|
+
* evidence-sufficiency constraint. 0 when no axis produced observations. */
|
|
1096
|
+
minN: number;
|
|
1097
|
+
/** Aggregate per-side cost from the gate context (a constraint input, not a
|
|
1098
|
+
* CI axis — see the module header). */
|
|
1099
|
+
cost: {
|
|
1100
|
+
candidate: number;
|
|
1101
|
+
baseline: number;
|
|
1102
|
+
};
|
|
1103
|
+
}
|
|
1104
|
+
/** A promotion strategy: a pure function from the evidence vector to a verdict.
|
|
1105
|
+
* Many policies can run over the same `EvidenceVector` and disagree — that's
|
|
1106
|
+
* the point (competing strategies, shared evidence). */
|
|
1107
|
+
type PromotionPolicy = (ev: EvidenceVector) => GateResult;
|
|
1108
|
+
interface BuildEvidenceVectorOptions {
|
|
1109
|
+
/** Minimum paired observations before an axis can claim significance; below
|
|
1110
|
+
* it the axis is `few_runs`. Default 3. */
|
|
1111
|
+
minProductiveRuns?: number;
|
|
1112
|
+
/** Confidence level for every axis bootstrap. Default 0.95. */
|
|
1113
|
+
confidence?: number;
|
|
1114
|
+
/** Bootstrap resamples. Default 2000. */
|
|
1115
|
+
resamples?: number;
|
|
1116
|
+
/** Fixed bootstrap seed for a deterministic, reproducible verdict. Default 1337. */
|
|
1117
|
+
seed?: number;
|
|
1118
|
+
}
|
|
1119
|
+
/**
|
|
1120
|
+
* The Evidence Bus. For each objective, pair candidate vs baseline by full
|
|
1121
|
+
* cellId and bootstrap a CI on the good-direction paired delta. Reuses the
|
|
1122
|
+
* exact `pairHoldout` + `pairedBootstrap` machinery the held-out gate uses, so
|
|
1123
|
+
* a single source of truth governs pairing granularity + scale handling.
|
|
1124
|
+
*/
|
|
1125
|
+
declare function buildEvidenceVector<TArtifact, TScenario extends Scenario>(ctx: GateContext<TArtifact, TScenario>, objectives: PromotionObjective[], opts?: BuildEvidenceVectorOptions): EvidenceVector;
|
|
1126
|
+
/**
|
|
1127
|
+
* The default strategy: symmetric multi-objective Pareto significance. Ship iff
|
|
1128
|
+
* the candidate weakly dominates the baseline at the confidence level — no axis
|
|
1129
|
+
* credibly worse AND ≥1 axis credibly better. Floor breach on any axis → hold
|
|
1130
|
+
* (anti-Goodhart, dominates everything). Insufficient evidence on any axis →
|
|
1131
|
+
* need_more_work. Statistically equivalent → hold (never ship noise).
|
|
1132
|
+
*/
|
|
1133
|
+
declare const paretoPolicy: PromotionPolicy;
|
|
1134
|
+
interface ParetoSignificanceGateOptions extends BuildEvidenceVectorOptions {
|
|
1135
|
+
/** The objective vector. Every axis is both a gain source and a safety floor. */
|
|
1136
|
+
objectives: PromotionObjective[];
|
|
1137
|
+
/** Strategy applied to the evidence vector. Default `paretoPolicy`. Override
|
|
1138
|
+
* to run a stricter/looser strategy over the SAME bus (competing policies). */
|
|
1139
|
+
policy?: PromotionPolicy;
|
|
1140
|
+
/** Override the gate name in reports. */
|
|
1141
|
+
name?: string;
|
|
1142
|
+
}
|
|
1143
|
+
/**
|
|
1144
|
+
* Wrap the bus + a policy as a `Gate`. Plugs into the existing
|
|
1145
|
+
* `runImprovementLoop({ gate })` slot and composes via `composeGate`; default
|
|
1146
|
+
* loop behavior is unchanged because consumers opt in by passing this gate.
|
|
1147
|
+
*/
|
|
1148
|
+
declare function paretoSignificanceGate<TArtifact = unknown, TScenario extends Scenario = Scenario>(options: ParetoSignificanceGateOptions): Gate<TArtifact, TScenario>;
|
|
1149
|
+
//#endregion
|
|
1150
|
+
//#region src/campaign/optimizer-model.d.ts
|
|
1151
|
+
type OptimizerModelBudget = ExternalOptimizerModelBudget;
|
|
1152
|
+
/** One metered OpenAI-compatible model connection shared by official optimizers. */
|
|
1153
|
+
interface OpenAICompatibleOptimizerModel {
|
|
1154
|
+
model: string;
|
|
1155
|
+
baseUrl: string;
|
|
1156
|
+
apiKey: string;
|
|
1157
|
+
budget: OptimizerModelBudget;
|
|
1158
|
+
}
|
|
1159
|
+
//#endregion
|
|
1160
|
+
//#region src/campaign/gepa-optimization-method.d.ts
|
|
1161
|
+
/** Shared settings for one bounded GEPA engine invocation. */
|
|
1162
|
+
interface GepaEngineOptions {
|
|
1163
|
+
/** GEPA engine name. GEPA validates names available in its Python runtime. */
|
|
1164
|
+
engine: string;
|
|
1165
|
+
/** Required cap for this engine's own model or CLI spend. */
|
|
1166
|
+
maxProposerCostUsd: number;
|
|
1167
|
+
/** Maximum concurrent evaluations inside this engine. Default: 1. */
|
|
1168
|
+
maxConcurrency?: number;
|
|
1169
|
+
/** Stop the engine after it reaches this score. */
|
|
1170
|
+
stopAtScore?: number;
|
|
1171
|
+
/** Isolate agent-based engines. Default: true. */
|
|
1172
|
+
sandbox?: boolean;
|
|
1173
|
+
/**
|
|
1174
|
+
* JSON-safe configuration for the registered GEPA engine.
|
|
1175
|
+
* Python callables and class instances cannot cross the process boundary.
|
|
1176
|
+
*/
|
|
1177
|
+
engineConfig?: Record<string, unknown>;
|
|
1178
|
+
}
|
|
1179
|
+
/** One independently budgeted GEPA engine invocation. */
|
|
1180
|
+
interface GepaEngineRun extends GepaEngineOptions {
|
|
1181
|
+
/** Maximum callback evaluations this engine may consume. */
|
|
1182
|
+
maxEvaluations: number;
|
|
1183
|
+
}
|
|
1184
|
+
/** An engine in an adaptive run. All engines share the recipe evaluation limit. */
|
|
1185
|
+
type GepaAdaptiveEngineRun = GepaEngineOptions;
|
|
1186
|
+
/**
|
|
1187
|
+
* A direct mapping to a GEPA optimization recipe.
|
|
1188
|
+
*
|
|
1189
|
+
* GEPA owns every search and composition operation represented here. Tangle
|
|
1190
|
+
* supplies the candidate, data, execution callback, judges, and budgets.
|
|
1191
|
+
*/
|
|
1192
|
+
type GepaOptimizationRecipe = {
|
|
1193
|
+
kind: 'engine';
|
|
1194
|
+
run: GepaEngineRun;
|
|
1195
|
+
} | {
|
|
1196
|
+
kind: 'sequential';
|
|
1197
|
+
runs: readonly GepaEngineRun[];
|
|
1198
|
+
} | {
|
|
1199
|
+
kind: 'adaptive-sequential';
|
|
1200
|
+
runs: readonly GepaAdaptiveEngineRun[];
|
|
1201
|
+
/** One evaluation budget shared by every adaptive stage. */
|
|
1202
|
+
maxEvaluations: number;
|
|
1203
|
+
/** Switch engines after this many evaluations without improvement. */
|
|
1204
|
+
plateauEvaluations: number;
|
|
1205
|
+
patience?: number;
|
|
1206
|
+
minEvaluationsPerStage?: number;
|
|
1207
|
+
improvementEpsilon?: number;
|
|
1208
|
+
cycle?: boolean;
|
|
1209
|
+
maxSwitches?: number;
|
|
1210
|
+
maxConcurrency?: number;
|
|
1211
|
+
} | {
|
|
1212
|
+
kind: 'best-of';
|
|
1213
|
+
runs: readonly GepaEngineRun[];
|
|
1214
|
+
maxWorkers?: number;
|
|
1215
|
+
} | {
|
|
1216
|
+
kind: 'vote';
|
|
1217
|
+
runs: readonly GepaEngineRun[];
|
|
1218
|
+
maxWorkers?: number;
|
|
1219
|
+
} | {
|
|
1220
|
+
kind: 'omni';
|
|
1221
|
+
explore: readonly GepaEngineRun[];
|
|
1222
|
+
continueWith: GepaEngineRun;
|
|
1223
|
+
maxWorkers?: number;
|
|
1224
|
+
};
|
|
1225
|
+
/** The command that runs the Python GEPA bridge. */
|
|
1226
|
+
type GepaRunnerCommand = ExternalOptimizerRunnerCommand;
|
|
1227
|
+
interface GepaOptimizationMethodConfig<TScenario extends Scenario, TArtifact = unknown> {
|
|
1228
|
+
/** Unique comparison-method name. Default identifies the GEPA recipe. */
|
|
1229
|
+
name?: string;
|
|
1230
|
+
/** A direct GEPA recipe. */
|
|
1231
|
+
recipe: GepaOptimizationRecipe;
|
|
1232
|
+
/** Plain-language goal shown to the external optimizer. */
|
|
1233
|
+
objective: string;
|
|
1234
|
+
/** Stable identity for the dispatch, judges, model settings, and scoring logic. */
|
|
1235
|
+
evaluationId: string;
|
|
1236
|
+
/** Optional bounded context about the surface and task. */
|
|
1237
|
+
background?: string;
|
|
1238
|
+
/**
|
|
1239
|
+
* Public dotted Python modules imported before GEPA resolves engine names.
|
|
1240
|
+
* Each module should call GEPA's official `register_engine()` API at import.
|
|
1241
|
+
*/
|
|
1242
|
+
engineModules?: readonly string[];
|
|
1243
|
+
/** Reject external candidates longer than this. Default: 200,000 characters. */
|
|
1244
|
+
maxCandidateChars?: number;
|
|
1245
|
+
/** Reject serialized score evidence longer than this. Default: 100,000 characters. */
|
|
1246
|
+
maxEvidenceChars?: number;
|
|
1247
|
+
/** End the bridge process after this many milliseconds. Default: 30 minutes. */
|
|
1248
|
+
timeoutMs?: number;
|
|
1249
|
+
/**
|
|
1250
|
+
* OpenAI-compatible model used by standard GEPA reflection.
|
|
1251
|
+
* Calls pass through Agent Eval's local model proxy. Every recipe engine must
|
|
1252
|
+
* be `gepa` when this is set.
|
|
1253
|
+
*/
|
|
1254
|
+
optimizer?: OpenAICompatibleOptimizerModel;
|
|
1255
|
+
/**
|
|
1256
|
+
* Decide what the external optimizer may read for a train or selection case.
|
|
1257
|
+
* The returned value must be JSON-serializable. The final comparison cases
|
|
1258
|
+
* are not accepted by this API and cannot be serialized here.
|
|
1259
|
+
*/
|
|
1260
|
+
describeScenario?: (scenario: TScenario) => unknown;
|
|
1261
|
+
/** Optional bounded artifact evidence returned to GEPA after each evaluation. */
|
|
1262
|
+
describeArtifact?: (artifact: TArtifact, scenario: TScenario) => unknown;
|
|
1263
|
+
/** Default: `never`. Compatible runs resume only when explicitly enabled. */
|
|
1264
|
+
resume?: ExternalOptimizerResumeMode;
|
|
1265
|
+
/**
|
|
1266
|
+
* Required for resumable direct GEPA runs because upstream state uses Python
|
|
1267
|
+
* pickle. Enable only for state created locally in a directory you control.
|
|
1268
|
+
*/
|
|
1269
|
+
trustResumeState?: boolean;
|
|
1270
|
+
runner?: GepaRunnerCommand;
|
|
1271
|
+
}
|
|
1272
|
+
/**
|
|
1273
|
+
* Turn an optional GEPA installation into an `OptimizationMethod`.
|
|
1274
|
+
*
|
|
1275
|
+
* GEPA receives only serialized train and selection cases. The caller's final
|
|
1276
|
+
* test partition stays inside `compareOptimizationMethods`, which invokes this
|
|
1277
|
+
* method without a test-set field. The local callback routes every candidate
|
|
1278
|
+
* evaluation through the same dispatch and judges used by other methods.
|
|
1279
|
+
*/
|
|
1280
|
+
declare function gepaOptimizationMethod<TScenario extends Scenario, TArtifact>(config: GepaOptimizationMethodConfig<TScenario, TArtifact>): OptimizationMethod<TScenario, TArtifact>;
|
|
1281
|
+
//#endregion
|
|
1282
|
+
//#region src/campaign/presets/run-eval.d.ts
|
|
1283
|
+
interface RunEvalOptions<TScenario extends Scenario, TArtifact> extends Omit<RunCampaignOptions<TScenario, TArtifact>, 'runDir'> {
|
|
1284
|
+
runDir: string;
|
|
1285
|
+
}
|
|
1286
|
+
/**
|
|
1287
|
+
* Simplest evaluation preset: run scenarios through dispatch, score with judges, and return a `CampaignResult` — no optimizer, no gate, no PR.
|
|
1288
|
+
*/
|
|
1289
|
+
declare function runEval<TScenario extends Scenario, TArtifact>(opts: RunEvalOptions<TScenario, TArtifact>): Promise<CampaignResult<TArtifact, TScenario>>;
|
|
1290
|
+
//#endregion
|
|
1291
|
+
//#region src/campaign/presets/run-optimization.d.ts
|
|
1292
|
+
interface PremeasuredOptimizationBaseline<TArtifact, TScenario extends Scenario> {
|
|
1293
|
+
/** Hash of the exact surface that produced `campaign`. */
|
|
1294
|
+
surfaceHash: string;
|
|
1295
|
+
/** Complete prior measurement reused by identity, including artifactsByPath. */
|
|
1296
|
+
campaign: CampaignResult<TArtifact, TScenario>;
|
|
1297
|
+
}
|
|
1298
|
+
interface RunOptimizationBaseOptions<TScenario extends Scenario, TArtifact> extends Omit<RunCampaignOptions<TScenario, TArtifact>, 'dispatch'> {
|
|
1299
|
+
/** Initial mutable surface (typically system prompt or addendum). */
|
|
1300
|
+
baselineSurface: MutableSurface;
|
|
1301
|
+
/**
|
|
1302
|
+
* Complete prior measurement of `baselineSurface`. When present,
|
|
1303
|
+
* `runOptimization` validates its surface, scenario split, seed, reps, and
|
|
1304
|
+
* normal campaign coverage, then skips the baseline campaign entirely — no
|
|
1305
|
+
* dispatch or resumability-cache lookup. Candidate campaigns still run
|
|
1306
|
+
* normally. Prior spend remains in the imported campaign aggregates and is
|
|
1307
|
+
* not added again to this continuation's CostLedger.
|
|
1308
|
+
*/
|
|
1309
|
+
premeasuredBaseline?: PremeasuredOptimizationBaseline<TArtifact, TScenario>;
|
|
1310
|
+
/** Dispatcher that takes the CURRENT surface + scenario → artifact. */
|
|
1311
|
+
dispatchWithSurface: (surface: MutableSurface, scenario: TScenario, ctx: Parameters<RunCampaignOptions<TScenario, TArtifact>['dispatch']>[1]) => Promise<TArtifact>;
|
|
1312
|
+
/** The candidate-generation strategy. */
|
|
1313
|
+
proposer: SurfaceProposer;
|
|
1314
|
+
populationSize: number;
|
|
1315
|
+
maxGenerations: number;
|
|
1316
|
+
/** Candidate campaigns run at once. Default 1. Total concurrent cells are
|
|
1317
|
+
* bounded by candidateConcurrency * maxConcurrency. */
|
|
1318
|
+
candidateConcurrency?: number;
|
|
1319
|
+
/** DEPTH knob forwarded to the proposer's `propose()` — max iterations the
|
|
1320
|
+
* agentic generator may take per candidate. */
|
|
1321
|
+
maxImprovementShots?: number;
|
|
1322
|
+
/** Optional analysis report forwarded to `propose()`. Opaque here; the
|
|
1323
|
+
* proposer types it. */
|
|
1324
|
+
report?: unknown;
|
|
1325
|
+
/** Structured findings forwarded to `propose()` as `ctx.findings`. A
|
|
1326
|
+
* findings producer emits these from the
|
|
1327
|
+
* generation's traces; findings-grounded proposers consume them. Opaque here;
|
|
1328
|
+
* the proposer types its `TFindings`. Empty when no producer is wired. */
|
|
1329
|
+
findings?: unknown[];
|
|
1330
|
+
/** Per-generation findings producer. Runs once on the BASELINE campaign
|
|
1331
|
+
* (as `generation: -1`, the baseline convention) before generation 0
|
|
1332
|
+
* proposes — so even a single-generation run proposes with trace context —
|
|
1333
|
+
* and then after each generation's candidates are scored with that
|
|
1334
|
+
* generation's results; whatever it returns REPLACES `ctx.findings` for the
|
|
1335
|
+
* NEXT `propose()`, so the diagnosis is refreshed each round instead
|
|
1336
|
+
* of being a static one-shot. Generic by design: the substrate does not
|
|
1337
|
+
* import an analyst — the consumer plugs its trace-analyst registry / HALO
|
|
1338
|
+
* here (reading the per-candidate `runDir` traces). When absent, findings
|
|
1339
|
+
* stay the static `opts.findings`. */
|
|
1340
|
+
analyzeGeneration?: (input: {
|
|
1341
|
+
generation: number;
|
|
1342
|
+
runDir: string;
|
|
1343
|
+
candidates: Array<{
|
|
1344
|
+
surfaceHash: string;
|
|
1345
|
+
campaign: CampaignResult<TArtifact, TScenario>;
|
|
1346
|
+
composite: number | null;
|
|
1347
|
+
}>;
|
|
1348
|
+
history: GenerationRecord[];
|
|
1349
|
+
/** Shared run spend account and receipt attribution phase. */
|
|
1350
|
+
costLedger?: CostLedgerHandle;
|
|
1351
|
+
costPhase?: string;
|
|
1352
|
+
}) => Promise<unknown[]>;
|
|
1353
|
+
/**
|
|
1354
|
+
* Optional override for how the WINNER is selected among coverage-complete
|
|
1355
|
+
* candidates (and how the incumbent bar is set). Returns a lexicographic rank
|
|
1356
|
+
* key — each element higher-is-better; candidates are ranked by descending key
|
|
1357
|
+
* (`compareRankKeys`) and the top must STRICTLY beat the incumbent's key to
|
|
1358
|
+
* promote. Defaults to `[campaignMeanComposite(campaign)]`, i.e. the historical
|
|
1359
|
+
* scalar-mean ranking (single-element key ⇒ identical behavior).
|
|
1360
|
+
*
|
|
1361
|
+
* A binary-with-replicates consumer (e.g. swe-arena, whose ship-gate counts an
|
|
1362
|
+
* instance resolved only when EVERY replicate resolved) passes a fail-closed
|
|
1363
|
+
* key built from the SAME reduction its gate uses, so winner-selection and the
|
|
1364
|
+
* ship-gate rank on the identical metric and can never invert — the selector
|
|
1365
|
+
* cannot promote a flaky per-cell-mean candidate the gate would reject over a
|
|
1366
|
+
* fail-closed candidate the gate would accept. Only the winner CHOICE changes;
|
|
1367
|
+
* the descriptive `composite` (mean) on every record and the Pareto objective
|
|
1368
|
+
* vectors are untouched, so proposer diversity and reporting are unaffected.
|
|
1369
|
+
*/
|
|
1370
|
+
selectionRankKey?: (campaign: CampaignResult<TArtifact, TScenario>) => number[];
|
|
1371
|
+
}
|
|
1372
|
+
type RunOptimizationOptions<TScenario extends Scenario, TArtifact> = RunOptimizationBaseOptions<TScenario, TArtifact>;
|
|
1373
|
+
interface RunOptimizationResult<TArtifact, TScenario extends Scenario> {
|
|
1374
|
+
generations: Array<{
|
|
1375
|
+
record: GenerationRecord;
|
|
1376
|
+
surfaces: Array<{
|
|
1377
|
+
surfaceHash: string;
|
|
1378
|
+
surface: MutableSurface;
|
|
1379
|
+
campaign: CampaignResult<TArtifact, TScenario>;
|
|
1380
|
+
}>;
|
|
1381
|
+
}>;
|
|
1382
|
+
winnerSurface: MutableSurface;
|
|
1383
|
+
winnerSurfaceHash: string;
|
|
1384
|
+
/** Proposer label for the promoted surface. Present when the winning
|
|
1385
|
+
* candidate came from a `ProposedCandidate` (a reflective proposer);
|
|
1386
|
+
* absent when the winner is the baseline or a bare-surface mutator. */
|
|
1387
|
+
winnerLabel?: string;
|
|
1388
|
+
/** Proposer rationale for the promoted surface — the "because Z" that
|
|
1389
|
+
* motivated the winning change. Survives to `SelfImproveResult` and the
|
|
1390
|
+
* emitted provenance record. Absent when the winner is the baseline. */
|
|
1391
|
+
winnerRationale?: string;
|
|
1392
|
+
baselineCampaign: CampaignResult<TArtifact, TScenario>;
|
|
1393
|
+
/** Run-wide spend, including agents, proposers, analysts, and judges. */
|
|
1394
|
+
cost: CostLedgerSummary;
|
|
1395
|
+
/** The GEPA Pareto frontier across every scored surface (baseline + all
|
|
1396
|
+
* generations) by per-scenario objective vector — the non-dominated set.
|
|
1397
|
+
* Each generation's `propose()` received the frontier-so-far as
|
|
1398
|
+
* `ctx.paretoParents`; this is the final frontier. A surface here that is
|
|
1399
|
+
* NOT the winner is uniquely best on some scenario the winner loses on. */
|
|
1400
|
+
paretoFrontier: ParetoParent[];
|
|
1401
|
+
}
|
|
1402
|
+
/**
|
|
1403
|
+
* Improvement loop body: N generations of propose → campaign → rank, maintaining a Pareto frontier and one global incumbent across generations.
|
|
1404
|
+
*/
|
|
1405
|
+
declare function runOptimization<TScenario extends Scenario, TArtifact>(opts: RunOptimizationOptions<TScenario, TArtifact>): Promise<RunOptimizationResult<TArtifact, TScenario>>;
|
|
1406
|
+
//#endregion
|
|
1407
|
+
//#region src/campaign/presets/run-improvement-loop.d.ts
|
|
1408
|
+
type RunImprovementLoopOptions<TScenario extends Scenario, TArtifact> = RunOptimizationOptions<TScenario, TArtifact> & {
|
|
1409
|
+
/** Holdout scenarios kept OUT of the training optimization pool — used
|
|
1410
|
+
* ONLY to score baseline vs winner for the gate. */
|
|
1411
|
+
holdoutScenarios: TScenario[];
|
|
1412
|
+
/** Holdout policy. Default `'measured'`: baseline + winner are re-scored on
|
|
1413
|
+
* `holdoutScenarios` and the gate decides on that held-out comparison.
|
|
1414
|
+
* `'deferred'`: the improvement-set (search) campaigns run exactly as usual,
|
|
1415
|
+
* but ZERO holdout cells are dispatched, the gate is forced to `'hold'`, and
|
|
1416
|
+
* the result + provenance record carry `holdout: 'deferred'` with NO
|
|
1417
|
+
* held-out lift — for callers that measure the held-out comparison in a
|
|
1418
|
+
* separate later run instead of faking a static holdout scenario and
|
|
1419
|
+
* recording a meaningless lift. */
|
|
1420
|
+
holdout?: 'measured' | 'deferred';
|
|
1421
|
+
/** Promotion gate. Substrate strongly recommends `defaultProductionGate`
|
|
1422
|
+
* for production wiring (composes red-team / reward-hacking / canary /
|
|
1423
|
+
* heldout). */
|
|
1424
|
+
gate: Gate<TArtifact, TScenario>;
|
|
1425
|
+
/** What to do when the gate ships:
|
|
1426
|
+
* - `'pr'`: open a PR via `openAutoPr`
|
|
1427
|
+
* - `'none'`: just report — caller decides what to do with the winner
|
|
1428
|
+
* Live-runtime self-mutation is intentionally unsupported. */
|
|
1429
|
+
autoOnPromote: 'pr' | 'none';
|
|
1430
|
+
/** GH owner / repo for the auto-PR. Required when autoOnPromote === 'pr'. */
|
|
1431
|
+
ghOwner?: string;
|
|
1432
|
+
ghRepo?: string;
|
|
1433
|
+
/** Placebo control. When supplied AND the winner differs from baseline, the
|
|
1434
|
+
* loop scores a THIRD holdout arm: the winner surface with its content
|
|
1435
|
+
* footprint-matched-blanked by this function (typically via `neutralizeText`).
|
|
1436
|
+
* Its scores are exposed to the gate as `ctx.neutralizedJudgeScores`, letting
|
|
1437
|
+
* a `neutralizationGate` reject a win whose lift survives blanking the content
|
|
1438
|
+
* (decorative — driven by footprint, not content). Costs one extra holdout
|
|
1439
|
+
* campaign; omit to skip. Return a byte/layout-matched blank of the winner. */
|
|
1440
|
+
neutralize?: (winnerSurface: MutableSurface, baselineSurface: MutableSurface) => MutableSurface;
|
|
1441
|
+
};
|
|
1442
|
+
interface RunImprovementLoopResult<TArtifact, TScenario extends Scenario> extends RunOptimizationResult<TArtifact, TScenario> {
|
|
1443
|
+
baselineOnHoldout: CampaignResult<TArtifact, TScenario>;
|
|
1444
|
+
winnerOnHoldout: CampaignResult<TArtifact, TScenario>;
|
|
1445
|
+
neutralizedOnHoldout?: CampaignResult<TArtifact, TScenario>;
|
|
1446
|
+
neutralizedSurface?: MutableSurface;
|
|
1447
|
+
gateResult: Awaited<ReturnType<Gate<TArtifact, TScenario>['decide']>>;
|
|
1448
|
+
/** Present iff the loop ran with `holdout: 'deferred'`. When set,
|
|
1449
|
+
* `baselineOnHoldout`/`winnerOnHoldout` are the shared EMPTY campaign (zero
|
|
1450
|
+
* cells dispatched) and the gate verdict is the forced `'hold'`. */
|
|
1451
|
+
holdout?: 'deferred';
|
|
1452
|
+
/** Unified baseline→winner surface diff. Computed UNCONDITIONALLY (not only
|
|
1453
|
+
* when `autoOnPromote === 'pr'`) so the diff that the gate decided on is
|
|
1454
|
+
* always present on the result + in the emitted provenance record. Empty
|
|
1455
|
+
* string when winner == baseline (no change to diff). */
|
|
1456
|
+
promotedDiff: string;
|
|
1457
|
+
prResult?: ReturnType<typeof openAutoPr>;
|
|
1458
|
+
}
|
|
1459
|
+
/**
|
|
1460
|
+
* Gated-promotion shell over `runOptimization`: scores the winner against the baseline on a holdout set, runs the release gate, and optionally opens a PR.
|
|
1461
|
+
*/
|
|
1462
|
+
declare function runImprovementLoop<TScenario extends Scenario, TArtifact>(opts: RunImprovementLoopOptions<TScenario, TArtifact>): Promise<RunImprovementLoopResult<TArtifact, TScenario>>;
|
|
1463
|
+
//#endregion
|
|
1464
|
+
//#region src/campaign/provenance.d.ts
|
|
1465
|
+
interface LoopProvenanceCandidate {
|
|
1466
|
+
/** Generation index this candidate was proposed in. */
|
|
1467
|
+
generation: number;
|
|
1468
|
+
/** 16-char loop-identity fingerprint (matches `GenerationCandidate.surfaceHash`). */
|
|
1469
|
+
surfaceHash: string;
|
|
1470
|
+
/** Full sha256 content hash — byte-identical-verifiable. */
|
|
1471
|
+
contentHash: string;
|
|
1472
|
+
/** Exact scored rows that produced this candidate's search result. */
|
|
1473
|
+
campaignDigest: `sha256:${string}`;
|
|
1474
|
+
/** Proposer label, when the proposer returned a `ProposedCandidate`. */
|
|
1475
|
+
label?: string;
|
|
1476
|
+
/** Proposer rationale — the "because Z". When the proposer returned a bare
|
|
1477
|
+
* surface (blind mutator) this is absent. */
|
|
1478
|
+
rationale?: string;
|
|
1479
|
+
/** Exact complete incumbent this candidate mutated. */
|
|
1480
|
+
parentSurfaceHash: string;
|
|
1481
|
+
/** Search-split composite of the exact parent. */
|
|
1482
|
+
parentComposite: number;
|
|
1483
|
+
/** Search-split composite change relative to the exact parent. */
|
|
1484
|
+
observedDeltaFromParent?: number;
|
|
1485
|
+
/** Whether the candidate completed every designed cell and could be selected. */
|
|
1486
|
+
eligibleForPromotion: boolean;
|
|
1487
|
+
/** Designed-denominator receipt retained even for incomplete candidates. */
|
|
1488
|
+
coverage: NonNullable<GenerationCandidate['coverage']>;
|
|
1489
|
+
/** Mean composite this candidate scored on the search split, or null when unscorable. */
|
|
1490
|
+
composite: number | null;
|
|
1491
|
+
/** Whether this candidate was promoted out of its generation. */
|
|
1492
|
+
promoted: boolean;
|
|
1493
|
+
}
|
|
1494
|
+
interface LoopProvenanceBackend {
|
|
1495
|
+
/** `assertRealBackend`-grade verdict over the worker call records. */
|
|
1496
|
+
verdict: 'real' | 'mixed' | 'stub';
|
|
1497
|
+
/** Number of worker LLM calls captured (the audit's "worker call count"). */
|
|
1498
|
+
workerCallCount: number;
|
|
1499
|
+
/** Distinct model ids observed across worker calls. */
|
|
1500
|
+
models: string[];
|
|
1501
|
+
totalInputTokens: number;
|
|
1502
|
+
totalOutputTokens: number;
|
|
1503
|
+
totalCostUsd: number;
|
|
1504
|
+
}
|
|
1505
|
+
interface LoopProvenanceEvidence {
|
|
1506
|
+
search: {
|
|
1507
|
+
splitDigest: `sha256:${string}`;
|
|
1508
|
+
baselineCampaignDigest: `sha256:${string}`;
|
|
1509
|
+
};
|
|
1510
|
+
holdout: {
|
|
1511
|
+
splitDigest: `sha256:${string}`;
|
|
1512
|
+
baselineCampaignDigest: `sha256:${string}`;
|
|
1513
|
+
winnerCampaignDigest: `sha256:${string}`;
|
|
1514
|
+
neutralized?: {
|
|
1515
|
+
contentHash: `sha256:${string}`;
|
|
1516
|
+
campaignDigest: `sha256:${string}`;
|
|
1517
|
+
composite: number;
|
|
1518
|
+
lift: number;
|
|
1519
|
+
};
|
|
1520
|
+
};
|
|
1521
|
+
costReceiptsDigest: `sha256:${string}`;
|
|
1522
|
+
}
|
|
1523
|
+
interface LoopProvenanceOptimizationMethod {
|
|
1524
|
+
name: string;
|
|
1525
|
+
cost: ComparisonCost;
|
|
1526
|
+
durationMs?: number;
|
|
1527
|
+
provenance?: OptimizationMethodProvenance;
|
|
1528
|
+
}
|
|
1529
|
+
/**
|
|
1530
|
+
* The durable provenance record. Aligns to the hosted `EvalRunEvent` path but
|
|
1531
|
+
* ADDS the rationale + the explicit baseline→candidate diff (both omitted from
|
|
1532
|
+
* the bare hosted event) + backend provenance.
|
|
1533
|
+
*/
|
|
1534
|
+
interface LoopProvenanceRecord {
|
|
1535
|
+
schema: 'tangle.loop-provenance';
|
|
1536
|
+
/** SHA-256 over the canonical record with this field omitted. */
|
|
1537
|
+
recordDigest: `sha256:${string}`;
|
|
1538
|
+
runId: string;
|
|
1539
|
+
runDir: string;
|
|
1540
|
+
timestamp: string;
|
|
1541
|
+
/** Baseline + winner surface content hashes — distinguishable, byte-verifiable. */
|
|
1542
|
+
baselineContentHash: string;
|
|
1543
|
+
winnerContentHash: string;
|
|
1544
|
+
/** Proposer label/rationale for the promoted change. Absent ⇒ winner == baseline. */
|
|
1545
|
+
winnerLabel?: string;
|
|
1546
|
+
winnerRationale?: string;
|
|
1547
|
+
/** The explicit baseline→winner unified diff the gate decided on. */
|
|
1548
|
+
diff: string;
|
|
1549
|
+
/** Every candidate across every generation, with its rationale and structured cause. */
|
|
1550
|
+
candidates: LoopProvenanceCandidate[];
|
|
1551
|
+
/** Complete external method identity and spend, when one authored the candidate. */
|
|
1552
|
+
optimizationMethod?: LoopProvenanceOptimizationMethod;
|
|
1553
|
+
/** Exact campaign, split, surface, and receipt identities behind every summary. */
|
|
1554
|
+
evidence: LoopProvenanceEvidence;
|
|
1555
|
+
/** Baseline composite on the search split that generated the candidates. */
|
|
1556
|
+
baselineSearchComposite: number;
|
|
1557
|
+
/** The gate verdict — decision + reasons + contributing gates + delta. */
|
|
1558
|
+
gate: {
|
|
1559
|
+
decision: GateDecision;
|
|
1560
|
+
reasons: string[];
|
|
1561
|
+
delta?: number;
|
|
1562
|
+
contributingGates: GateContribution[];
|
|
1563
|
+
};
|
|
1564
|
+
/** Present iff the loop ran with `holdout: 'deferred'` — the held-out
|
|
1565
|
+
* comparison was intentionally not measured in this run, so the holdout
|
|
1566
|
+
* composites and `heldOutLift` are ABSENT rather than recorded as a
|
|
1567
|
+
* meaningless 0. */
|
|
1568
|
+
holdout?: 'deferred';
|
|
1569
|
+
/** baseline-on-holdout composite mean. Absent when `holdout === 'deferred'`. */
|
|
1570
|
+
baselineHoldoutComposite?: number;
|
|
1571
|
+
/** winner-on-holdout composite mean. Absent when `holdout === 'deferred'`. */
|
|
1572
|
+
winnerHoldoutComposite?: number;
|
|
1573
|
+
/** winnerHoldout - baselineHoldout — RECOMPUTABLE from this record. Absent
|
|
1574
|
+
* when `holdout === 'deferred'` (no held-out measurement ran). */
|
|
1575
|
+
heldOutLift?: number;
|
|
1576
|
+
/** Backend provenance: stub-vs-real verdict + worker call count + models. */
|
|
1577
|
+
backend: LoopProvenanceBackend;
|
|
1578
|
+
totalCostUsd: number;
|
|
1579
|
+
totalDurationMs: number;
|
|
1580
|
+
}
|
|
1581
|
+
interface BuildLoopProvenanceArgs<TArtifact, TScenario extends Scenario> {
|
|
1582
|
+
runId: string;
|
|
1583
|
+
runDir: string;
|
|
1584
|
+
timestamp: string;
|
|
1585
|
+
baselineSurface: MutableSurface;
|
|
1586
|
+
winnerSurface: MutableSurface;
|
|
1587
|
+
winnerLabel?: string;
|
|
1588
|
+
winnerRationale?: string;
|
|
1589
|
+
/** Exact baseline campaign on the search split. */
|
|
1590
|
+
baselineSearchCampaign: CampaignResult<TArtifact, TScenario>;
|
|
1591
|
+
/** Per-generation candidate records straight off the loop result. */
|
|
1592
|
+
generations: Array<{
|
|
1593
|
+
generationIndex: number;
|
|
1594
|
+
candidates: GenerationCandidate[];
|
|
1595
|
+
promoted: string[];
|
|
1596
|
+
/** Surfaces measured this generation, keyed by surface hash so the content
|
|
1597
|
+
* hash can be computed and the loop identity rechecked from real bytes. */
|
|
1598
|
+
surfaces: Array<{
|
|
1599
|
+
surfaceHash: string;
|
|
1600
|
+
surface: MutableSurface;
|
|
1601
|
+
campaign: CampaignResult<TArtifact, TScenario>;
|
|
1602
|
+
}>;
|
|
1603
|
+
}>;
|
|
1604
|
+
gate: GateResult;
|
|
1605
|
+
/** Holdout policy the loop ran with. `'deferred'` ⇒ the holdout campaigns
|
|
1606
|
+
* below are the shared empty campaign and the record omits the holdout
|
|
1607
|
+
* composites + `heldOutLift`. Default `'measured'`. */
|
|
1608
|
+
holdout?: 'measured' | 'deferred';
|
|
1609
|
+
baselineOnHoldout: CampaignResult<TArtifact, TScenario>;
|
|
1610
|
+
winnerOnHoldout: CampaignResult<TArtifact, TScenario>;
|
|
1611
|
+
neutralizedSurface?: MutableSurface;
|
|
1612
|
+
neutralizedOnHoldout?: CampaignResult<TArtifact, TScenario>;
|
|
1613
|
+
/** Settled run-wide receipts — agent calls are the source for backend provenance. */
|
|
1614
|
+
costReceipts: ReadonlyArray<CostReceipt>;
|
|
1615
|
+
totalCostUsd: number;
|
|
1616
|
+
totalDurationMs: number;
|
|
1617
|
+
optimizationMethod?: LoopProvenanceOptimizationMethod;
|
|
1618
|
+
}
|
|
1619
|
+
interface LoopProvenanceArgsFromResult<TArtifact, TScenario extends Scenario> {
|
|
1620
|
+
runId: string;
|
|
1621
|
+
runDir: string;
|
|
1622
|
+
timestamp: string;
|
|
1623
|
+
baselineSurface: MutableSurface;
|
|
1624
|
+
result: RunImprovementLoopResult<TArtifact, TScenario>;
|
|
1625
|
+
costReceipts: ReadonlyArray<CostReceipt>;
|
|
1626
|
+
totalCostUsd: number;
|
|
1627
|
+
totalDurationMs: number;
|
|
1628
|
+
}
|
|
1629
|
+
/** One translation from a completed improvement loop into durable evidence. */
|
|
1630
|
+
declare function loopProvenanceArgsFromResult<TArtifact, TScenario extends Scenario>(input: LoopProvenanceArgsFromResult<TArtifact, TScenario>): BuildLoopProvenanceArgs<TArtifact, TScenario>;
|
|
1631
|
+
/** Build the durable provenance record from a completed loop result. */
|
|
1632
|
+
declare function buildLoopProvenanceRecord<TArtifact, TScenario extends Scenario>(args: BuildLoopProvenanceArgs<TArtifact, TScenario>): LoopProvenanceRecord;
|
|
1633
|
+
/** Digest the exact campaign fields that can affect a measured comparison. */
|
|
1634
|
+
declare function campaignMeasurementDigest<TArtifact, TScenario extends Scenario>(campaign: CampaignResult<TArtifact, TScenario>): `sha256:${string}`;
|
|
1635
|
+
/** Recompute and validate the self-addressed durable record. */
|
|
1636
|
+
declare function verifyLoopProvenanceRecord(record: LoopProvenanceRecord): LoopProvenanceRecord;
|
|
1637
|
+
/** Return the canonical SHA-256 digest of a JSON-serializable value. */
|
|
1638
|
+
declare function canonicalDigest(value: unknown): `sha256:${string}`;
|
|
1639
|
+
/**
|
|
1640
|
+
* Build the loop's OTLP-ingestable spans from a provenance record. One root
|
|
1641
|
+
* span per loop (`tangle.runId`), one span per generation, one span per
|
|
1642
|
+
* candidate (carrying its surfaceHash + label), and one span for the gate
|
|
1643
|
+
* decision (carrying reasons + delta + lift). Candidate + gate spans pivot on
|
|
1644
|
+
* the same `tangle.runId` / `tangle.generation` attributes `/adapters/otel`
|
|
1645
|
+
* reads, so the hosted collector reconstructs the full tree.
|
|
1646
|
+
*
|
|
1647
|
+
* Times are synthesized monotonically off a single base so the span tree is
|
|
1648
|
+
* orderable; the substrate does not retain per-candidate wall-clock starts.
|
|
1649
|
+
*/
|
|
1650
|
+
declare function loopProvenanceSpans(record: LoopProvenanceRecord, opts?: {
|
|
1651
|
+
baseTimeMs?: number;
|
|
1652
|
+
}): TraceSpanEvent[];
|
|
1653
|
+
/** Canonical durable paths under the run dir. */
|
|
1654
|
+
declare function provenanceRecordPath(runDir: string): string;
|
|
1655
|
+
/**
|
|
1656
|
+
* Canonical path for the durable OTLP spans JSONL file under a loop run directory.
|
|
1657
|
+
*/
|
|
1658
|
+
declare function provenanceSpansPath(runDir: string): string;
|
|
1659
|
+
interface EmitLoopProvenanceResult {
|
|
1660
|
+
record: LoopProvenanceRecord;
|
|
1661
|
+
spans: TraceSpanEvent[];
|
|
1662
|
+
/** Absolute paths the record + spans were written to, when storage persists. */
|
|
1663
|
+
recordPath: string;
|
|
1664
|
+
spansPath: string;
|
|
1665
|
+
}
|
|
1666
|
+
interface EmitLoopProvenanceArgs<TArtifact, TScenario extends Scenario> extends BuildLoopProvenanceArgs<TArtifact, TScenario> {
|
|
1667
|
+
/** Storage the record + spans are written through. */
|
|
1668
|
+
storage: CampaignStorage;
|
|
1669
|
+
/** When set, the spans are also shipped to the hosted `/v1/ingest/traces`
|
|
1670
|
+
* endpoint so the collector receives the full loop, not just `cost.*`. */
|
|
1671
|
+
hostedClient?: HostedClient;
|
|
1672
|
+
}
|
|
1673
|
+
/**
|
|
1674
|
+
* Build the provenance record + OTel spans and persist them durably under the
|
|
1675
|
+
* run dir (and ship spans to a hosted collector when one is wired). Returns
|
|
1676
|
+
* both artifacts so the caller can assert on / re-derive from them.
|
|
1677
|
+
*
|
|
1678
|
+
* Fail-loud: the durable write throws on storage failure (a swallowed write is
|
|
1679
|
+
* exactly the "emitted but lost" failure this closes). The hosted span ship is
|
|
1680
|
+
* the one best-effort leg — its failure is logged, not thrown, so an offline
|
|
1681
|
+
* collector never fails the loop (the durable artifact is the source of truth).
|
|
1682
|
+
*/
|
|
1683
|
+
declare function emitLoopProvenance<TArtifact, TScenario extends Scenario>(args: EmitLoopProvenanceArgs<TArtifact, TScenario>): Promise<EmitLoopProvenanceResult>;
|
|
1684
|
+
//#endregion
|
|
1685
|
+
//#region src/campaign/skillopt-optimization-method.d.ts
|
|
1686
|
+
interface SkillOptTrainerConfig {
|
|
1687
|
+
epochs: number;
|
|
1688
|
+
batchSize: number;
|
|
1689
|
+
accumulation?: number;
|
|
1690
|
+
editBudget?: number;
|
|
1691
|
+
minEditBudget?: number;
|
|
1692
|
+
analystWorkers?: number;
|
|
1693
|
+
minibatchSize?: number;
|
|
1694
|
+
mergeBatchSize?: number;
|
|
1695
|
+
maxAnalystRounds?: number;
|
|
1696
|
+
evaluationWorkers?: number;
|
|
1697
|
+
learningRateSchedule?: 'constant' | 'linear' | 'cosine' | 'autonomous';
|
|
1698
|
+
learningRateControl?: 'fixed' | 'autonomous' | 'none';
|
|
1699
|
+
updateMode?: 'patch' | 'rewrite_from_suggestions' | 'full_rewrite_minibatch';
|
|
1700
|
+
failureOnly?: boolean;
|
|
1701
|
+
useSlowUpdate?: boolean;
|
|
1702
|
+
useMetaSkill?: boolean;
|
|
1703
|
+
/**
|
|
1704
|
+
* Additional flat SkillOpt trainer settings. Tangle overwrites data,
|
|
1705
|
+
* output, split, seed, validation, and activation settings.
|
|
1706
|
+
*/
|
|
1707
|
+
overrides?: Record<string, unknown>;
|
|
1708
|
+
}
|
|
1709
|
+
type SkillOptRunnerCommand = ExternalOptimizerRunnerCommand;
|
|
1710
|
+
interface SkillOptOptimizationMethodConfig<TScenario extends Scenario, TArtifact = unknown> {
|
|
1711
|
+
name?: string;
|
|
1712
|
+
/** Goal included with every described train and selection case. */
|
|
1713
|
+
objective: string;
|
|
1714
|
+
background?: string;
|
|
1715
|
+
/** Stable identity for the dispatch, judges, model settings, and scoring logic. */
|
|
1716
|
+
evaluationId: string;
|
|
1717
|
+
trainer: SkillOptTrainerConfig;
|
|
1718
|
+
/**
|
|
1719
|
+
* OpenAI-compatible model connection and hard limits for SkillOpt's own
|
|
1720
|
+
* optimizer calls.
|
|
1721
|
+
*/
|
|
1722
|
+
optimizer: OpenAICompatibleOptimizerModel;
|
|
1723
|
+
/** Hard cap on candidate-case callback requests. */
|
|
1724
|
+
maxEvaluations: number;
|
|
1725
|
+
/** Scores at or above this value count as hard successes. Default: 1. */
|
|
1726
|
+
hardScoreThreshold?: number;
|
|
1727
|
+
maxCandidateChars?: number;
|
|
1728
|
+
/** Maximum serialized scenario plus evaluation evidence. Default: 100,000. */
|
|
1729
|
+
maxEvidenceChars?: number;
|
|
1730
|
+
timeoutMs?: number;
|
|
1731
|
+
describeScenario?: (scenario: TScenario) => unknown;
|
|
1732
|
+
describeArtifact?: (artifact: TArtifact, scenario: TScenario) => unknown;
|
|
1733
|
+
resume?: ExternalOptimizerResumeMode;
|
|
1734
|
+
runner?: SkillOptRunnerCommand;
|
|
1735
|
+
}
|
|
1736
|
+
/** Run Microsoft's SkillOpt trainer as a complete optimization method. */
|
|
1737
|
+
declare function skillOptOptimizationMethod<TScenario extends Scenario, TArtifact>(config: SkillOptOptimizationMethodConfig<TScenario, TArtifact>): OptimizationMethod<TScenario, TArtifact>;
|
|
1738
|
+
//#endregion
|
|
1739
|
+
export { Objective as $, optimizationTokenUsageFromSummary as $t, RunEvalOptions as A, CanarySeverity as At, OptimizerModelBudget as B, ComparisonCost as Bt, RunImprovementLoopOptions as C, ReferenceEquivalenceJudgeResult as Cn, scoreRedTeamOutput as Ct, RunOptimizationOptions as D, LlmJudgeDimension as Dn, CanaryKind as Dt, PremeasuredOptimizationBaseline as E, runReferenceEquivalenceJudge as En, CanaryEvaluation as Et, GepaOptimizationMethodConfig as F, ExternalTextOptimizerContext as Ft, ObjectiveSource as G, OptimizationMethodProvenance as Gt, AxisVerdict as H, OptimizationMethodComparison as Ht, GepaOptimizationRecipe as I, ExternalTextOptimizerResult as It, PromotionPolicy as J, OptimizationMethodScore as Jt, ParetoSignificanceGateOptions as K, OptimizationMethodResult as Kt, GepaRunnerCommand as L, ExternalOptimizationExample as Lt, GepaAdaptiveEngineRun as M, composeGate as Mt, GepaEngineOptions as N, externalTextOptimizationMethod as Nt, RunOptimizationResult as O, LlmJudgeOptions as On, CanaryOptions as Ot, GepaEngineRun as P, ExternalTextOptimizationMethodConfig as Pt, Direction as Q, costFromLedgerSummary as Qt, gepaOptimizationMethod as R, ExternalTextEvaluationResponse as Rt, verifyLoopProvenanceRecord as S, ReferenceEquivalenceJudgeOptions as Sn, redTeamReport as St, runImprovementLoop as T, createReferenceEquivalenceJudge as Tn, CanaryAlert as Tt, BuildEvidenceVectorOptions as U, OptimizationMethodInput as Ut, AxisEvidence as V, OptimizationMethod as Vt, EvidenceVector as W, OptimizationMethodPairwise as Wt, paretoPolicy as X, OptimizationTokenUsage as Xt, buildEvidenceVector as Y, OptimizationPackageSource as Yt, paretoSignificanceGate as Z, compareOptimizationMethods as Zt, emitLoopProvenance as _, OpenAutoPrResult as _n, RedTeamCategory as _t, BuildLoopProvenanceArgs as a, planCampaignRun as an, scalarScore as at, provenanceRecordPath as b, REFERENCE_EQUIVALENCE_JUDGE_VERSION as bn, RedTeamReport as bt, LoopProvenanceArgsFromResult as c, createRunCostLedger as cn, powerPreflight as ct, LoopProvenanceEvidence as d, assertCampaignDesign as dn, DefaultProductionGateCheck as dt, CampaignCellFailureReceipt as en, ParetoResult as et, LoopProvenanceOptimizationMethod as f, assertCampaignSplitIdentity as fn, DefaultProductionGateOptions as ft, canonicalDigest as g, OpenAutoPrOptions as gn, RedTeamCase as gt, campaignMeasurementDigest as h, campaignSplitDigestFromIdentities as hn, DEFAULT_RED_TEAM_CORPUS as ht, skillOptOptimizationMethod as i, RunCampaignOptions as in, paretoFrontierWithCrowding as it, runEval as j, runCanaries as jt, runOptimization as k, llmJudge as kn, CanaryReport as kt, LoopProvenanceBackend as l, fsCampaignStorage as ln, HeldOutGateOptions as lt, buildLoopProvenanceRecord as m, campaignSplitDigest as mn, defaultProductionGate as mt, SkillOptRunnerCommand as n, CampaignRunPlanCell as nn, dominates as nt, EmitLoopProvenanceArgs as o, runCampaign as on, PowerPreflight as ot, LoopProvenanceRecord as p, campaignScenarioIdentity as pn, DefaultProductionRewardHackingOptions as pt, PromotionObjective as q, OptimizationMethodRunOptions as qt, SkillOptTrainerConfig as r, PlanCampaignRunOptions as rn, paretoFrontier as rt, EmitLoopProvenanceResult as s, CampaignStorage as sn, PowerPreflightOptions as st, SkillOptOptimizationMethodConfig as t, CampaignRunPlan as tn, crowdingDistance as tt, LoopProvenanceCandidate as u, inMemoryCampaignStorage as un, heldOutGate as ut, loopProvenanceArgsFromResult as v, openAutoPr as vn, RedTeamFinding as vt, RunImprovementLoopResult as w, ReferenceEquivalenceScenario as wn, toolNamesForRun as wt, provenanceSpansPath as x, ReferenceEquivalenceJudgeInput as xn, redTeamDataset as xt, loopProvenanceSpans as y, REFERENCE_EQUIVALENCE_INPUT_LIMITS as yn, RedTeamPayload as yt, OpenAICompatibleOptimizerModel as z, CompareOptimizationMethodsOptions as zt };
|
|
1740
|
+
//# sourceMappingURL=skillopt-optimization-method-CWKVTnks.d.ts.map
|