@tangle-network/agent-eval 0.129.0 → 0.130.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +14 -0
- package/README.md +1 -0
- package/dist/active-curriculum-C4mk67HP.js +214 -0
- package/dist/active-curriculum-C4mk67HP.js.map +1 -0
- package/dist/adversarial-smnADNFS.d.ts +21 -0
- package/dist/adversarial-smnADNFS.d.ts.map +1 -0
- package/dist/analyst/index.d.ts +81 -2872
- package/dist/analyst/index.d.ts.map +1 -0
- package/dist/analyst/index.js +319 -360
- package/dist/analyst/index.js.map +1 -1
- package/dist/analyst-BkTS3C58.d.ts +89 -0
- package/dist/analyst-BkTS3C58.d.ts.map +1 -0
- package/dist/analyst-LsnNpSkm.js +152 -0
- package/dist/analyst-LsnNpSkm.js.map +1 -0
- package/dist/analyze-runs-C1CavBMk.js +1067 -0
- package/dist/analyze-runs-C1CavBMk.js.map +1 -0
- package/dist/analyze-runs-FsgYCinh.d.ts +72 -0
- package/dist/analyze-runs-FsgYCinh.d.ts.map +1 -0
- package/dist/attribute-vocabulary-DLJ6303h.d.ts +54 -0
- package/dist/attribute-vocabulary-DLJ6303h.d.ts.map +1 -0
- package/dist/authenticity/index.d.ts +81 -79
- package/dist/authenticity/index.d.ts.map +1 -0
- package/dist/authenticity/index.js +209 -193
- package/dist/authenticity/index.js.map +1 -1
- package/dist/baseline-HsBvw_dk.js +550 -0
- package/dist/baseline-HsBvw_dk.js.map +1 -0
- package/dist/baseline-hG3K85h4.d.ts +125 -0
- package/dist/baseline-hG3K85h4.d.ts.map +1 -0
- package/dist/belief-state/index.d.ts +448 -1188
- package/dist/belief-state/index.d.ts.map +1 -0
- package/dist/belief-state/index.js +1617 -1709
- package/dist/belief-state/index.js.map +1 -1
- package/dist/benchmarks/index.d.ts +2 -891
- package/dist/benchmarks/index.js +2 -60
- package/dist/benchmarks-DviOvUNr.js +754 -0
- package/dist/benchmarks-DviOvUNr.js.map +1 -0
- package/dist/builder-eval/index.d.ts +150 -662
- package/dist/builder-eval/index.d.ts.map +1 -0
- package/dist/builder-eval/index.js +356 -345
- package/dist/builder-eval/index.js.map +1 -1
- package/dist/calibration-CNWWA6K8.js +94 -0
- package/dist/calibration-CNWWA6K8.js.map +1 -0
- package/dist/campaign/index.d.ts +5 -6381
- package/dist/campaign/index.js +3 -213
- package/dist/campaign-CBKZvQ1H.js +3885 -0
- package/dist/campaign-CBKZvQ1H.js.map +1 -0
- package/dist/cli.d.ts +1 -1
- package/dist/cli.js +164 -175
- package/dist/cli.js.map +1 -1
- package/dist/client-C97NMzqi.d.ts +581 -0
- package/dist/client-C97NMzqi.d.ts.map +1 -0
- package/dist/client-CYzbdJOZ.js +637 -0
- package/dist/client-CYzbdJOZ.js.map +1 -0
- package/dist/code-agent-session-BjkMTQ7H.js +1390 -0
- package/dist/code-agent-session-BjkMTQ7H.js.map +1 -0
- package/dist/code-agent-session-DqqgOJaz.d.ts +143 -0
- package/dist/code-agent-session-DqqgOJaz.d.ts.map +1 -0
- package/dist/concurrency-DIxRZF_J.js +85 -0
- package/dist/concurrency-DIxRZF_J.js.map +1 -0
- package/dist/contract/index.d.ts +645 -5565
- package/dist/contract/index.d.ts.map +1 -0
- package/dist/contract/index.js +1730 -1938
- package/dist/contract/index.js.map +1 -1
- package/dist/control.d.ts +3 -1030
- package/dist/control.js +2 -33
- package/dist/cost-ledger-DIgQUFZZ.js +801 -0
- package/dist/cost-ledger-DIgQUFZZ.js.map +1 -0
- package/dist/cost-ledger-Dye6jCgg.d.ts +236 -0
- package/dist/cost-ledger-Dye6jCgg.d.ts.map +1 -0
- package/dist/dataset-BvtnC8Dc.d.ts +115 -0
- package/dist/dataset-BvtnC8Dc.d.ts.map +1 -0
- package/dist/default-registry-C-vFCSEc.js +2579 -0
- package/dist/default-registry-C-vFCSEc.js.map +1 -0
- package/dist/default-registry-CNPo-Vsb.d.ts +540 -0
- package/dist/default-registry-CNPo-Vsb.d.ts.map +1 -0
- package/dist/emitter-CPBAhxum.js +266 -0
- package/dist/emitter-CPBAhxum.js.map +1 -0
- package/dist/emitter-DGQGoLyj.d.ts +113 -0
- package/dist/emitter-DGQGoLyj.d.ts.map +1 -0
- package/dist/errors-8YnH8WlF.js +64 -0
- package/dist/errors-8YnH8WlF.js.map +1 -0
- package/dist/errors-CEk209JS.d.ts +76 -0
- package/dist/errors-CEk209JS.d.ts.map +1 -0
- package/dist/eval-campaign-DEm6c8ru.js +349 -0
- package/dist/eval-campaign-DEm6c8ru.js.map +1 -0
- package/dist/execution-tracks-CpgFPpS5.js +93 -0
- package/dist/execution-tracks-CpgFPpS5.js.map +1 -0
- package/dist/exporters-q9iL-2Jf.js +148 -0
- package/dist/exporters-q9iL-2Jf.js.map +1 -0
- package/dist/extract-usage-BrQ8mCLX.js +155 -0
- package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
- package/dist/failure-cluster-CqcvCcdR.d.ts +59 -0
- package/dist/failure-cluster-CqcvCcdR.d.ts.map +1 -0
- package/dist/feedback-trajectory-CVaeREXV.d.ts +340 -0
- package/dist/feedback-trajectory-CVaeREXV.d.ts.map +1 -0
- package/dist/fuzz.d.ts +320 -646
- package/dist/fuzz.d.ts.map +1 -0
- package/dist/fuzz.js +670 -618
- package/dist/fuzz.js.map +1 -1
- package/dist/hf-dataset-DBJXXoY1.js +763 -0
- package/dist/hf-dataset-DBJXXoY1.js.map +1 -0
- package/dist/hosted/index.d.ts +11 -831
- package/dist/hosted/index.d.ts.map +1 -0
- package/dist/hosted/index.js +2 -37
- package/dist/index-2JJSA6-r2.d.ts +926 -0
- package/dist/index-2JJSA6-r2.d.ts.map +1 -0
- package/dist/index-6N0aYmpW.d.ts +217 -0
- package/dist/index-6N0aYmpW.d.ts.map +1 -0
- package/dist/index-BAvgST_9.d.ts +131 -0
- package/dist/index-BAvgST_9.d.ts.map +1 -0
- package/dist/index-BvnJuTGD.d.ts +68 -0
- package/dist/index-BvnJuTGD.d.ts.map +1 -0
- package/dist/index-C61Wi7yg.d.ts +547 -0
- package/dist/index-C61Wi7yg.d.ts.map +1 -0
- package/dist/index-CAPUUKaM.d.ts +335 -0
- package/dist/index-CAPUUKaM.d.ts.map +1 -0
- package/dist/index-DE5fb3EC.d.ts +2244 -0
- package/dist/index-DE5fb3EC.d.ts.map +1 -0
- package/dist/index-DSC51roc.d.ts +102 -0
- package/dist/index-DSC51roc.d.ts.map +1 -0
- package/dist/index.d.ts +3755 -15555
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +11182 -11216
- package/dist/index.js.map +1 -1
- package/dist/integrity-BzRbCHzi.js +424 -0
- package/dist/integrity-BzRbCHzi.js.map +1 -0
- package/dist/integrity-rmVhXWA7.d.ts +61 -0
- package/dist/integrity-rmVhXWA7.d.ts.map +1 -0
- package/dist/judge-calibration-DFtEMlde.d.ts +146 -0
- package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
- package/dist/ledger-core/index.d.ts +2 -0
- package/dist/ledger-core/index.js +2 -0
- package/dist/ledger-core-DtZz1RG0.js +388 -0
- package/dist/ledger-core-DtZz1RG0.js.map +1 -0
- package/dist/llm-client--GR4JbZE.js +687 -0
- package/dist/llm-client--GR4JbZE.js.map +1 -0
- package/dist/llm-client-B_nIBlYo.d.ts +290 -0
- package/dist/llm-client-B_nIBlYo.d.ts.map +1 -0
- package/dist/matrix/index.d.ts +3 -155
- package/dist/matrix/index.js +2 -8
- package/dist/matrix-BzQnu2S6.js +270 -0
- package/dist/matrix-BzQnu2S6.js.map +1 -0
- package/dist/meta-eval/index.d.ts +4 -1027
- package/dist/meta-eval/index.js +393 -390
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/metrics-C9YY1OcL.js +239 -0
- package/dist/metrics-C9YY1OcL.js.map +1 -0
- package/dist/mint-yN2M2eh0.js +201 -0
- package/dist/mint-yN2M2eh0.js.map +1 -0
- package/dist/multi-layer-verifier-BHY1gWAc.d.ts +138 -0
- package/dist/multi-layer-verifier-BHY1gWAc.d.ts.map +1 -0
- package/dist/multishot/index.d.ts +273 -480
- package/dist/multishot/index.d.ts.map +1 -0
- package/dist/multishot/index.js +595 -548
- package/dist/multishot/index.js.map +1 -1
- package/dist/off-policy-DvgzvtIx.js +220 -0
- package/dist/off-policy-DvgzvtIx.js.map +1 -0
- package/dist/off-policy-mskQw8Mb.d.ts +153 -0
- package/dist/off-policy-mskQw8Mb.d.ts.map +1 -0
- package/dist/openapi.json +1 -1
- package/dist/opencode-sqlite-BGrHeDu3.js +318 -0
- package/dist/opencode-sqlite-BGrHeDu3.js.map +1 -0
- package/dist/outcome-store-BYHIuO0e.d.ts +65 -0
- package/dist/outcome-store-BYHIuO0e.d.ts.map +1 -0
- package/dist/outcome-store-ChBKlTd_.js +75 -0
- package/dist/outcome-store-ChBKlTd_.js.map +1 -0
- package/dist/paired-arms-D9D0wXj2.js +260 -0
- package/dist/paired-arms-D9D0wXj2.js.map +1 -0
- package/dist/pipelines/index.d.ts +95 -532
- package/dist/pipelines/index.d.ts.map +1 -0
- package/dist/pipelines/index.js +497 -478
- package/dist/pipelines/index.js.map +1 -1
- package/dist/pre-registration-DakwTRXk.js +96 -0
- package/dist/pre-registration-DakwTRXk.js.map +1 -0
- package/dist/propose-review-control-LhIWGzHl.js +1458 -0
- package/dist/propose-review-control-LhIWGzHl.js.map +1 -0
- package/dist/query-CJ_DX8vl.d.ts +28 -0
- package/dist/query-CJ_DX8vl.d.ts.map +1 -0
- package/dist/query-Di7eEQ79.js +83 -0
- package/dist/query-Di7eEQ79.js.map +1 -0
- package/dist/raw-provider-sink-BQd7mzyT.js +201 -0
- package/dist/raw-provider-sink-BQd7mzyT.js.map +1 -0
- package/dist/raw-provider-sink-BU29Sh8h.d.ts +134 -0
- package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +1 -0
- package/dist/redact-7Aq1ukl-.js +107 -0
- package/dist/redact-7Aq1ukl-.js.map +1 -0
- package/dist/release-report-Cz9NKH39.js +603 -0
- package/dist/release-report-Cz9NKH39.js.map +1 -0
- package/dist/release-report-mvB2G4_J.d.ts +244 -0
- package/dist/release-report-mvB2G4_J.d.ts.map +1 -0
- package/dist/replay-GnyotH0J.js +1741 -0
- package/dist/replay-GnyotH0J.js.map +1 -0
- package/dist/replay-OoidtG1E.d.ts +749 -0
- package/dist/replay-OoidtG1E.d.ts.map +1 -0
- package/dist/reporting.d.ts +6 -1312
- package/dist/reporting.js +6 -51
- package/dist/researcher-CwTdwXG1.d.ts +314 -0
- package/dist/researcher-CwTdwXG1.d.ts.map +1 -0
- package/dist/reward-hacking-eAnOsynk.d.ts +224 -0
- package/dist/reward-hacking-eAnOsynk.d.ts.map +1 -0
- package/dist/reward-hacking-qipEpKvY.js +596 -0
- package/dist/reward-hacking-qipEpKvY.js.map +1 -0
- package/dist/reward-nw2xZGZG.js +137 -0
- package/dist/reward-nw2xZGZG.js.map +1 -0
- package/dist/rl.d.ts +760 -4010
- package/dist/rl.d.ts.map +1 -0
- package/dist/rl.js +2325 -1958
- package/dist/rl.js.map +1 -1
- package/dist/rolldown-runtime-8H4AJuhK.js +14 -0
- package/dist/rollout/index.d.ts +3 -2087
- package/dist/rollout/index.js +8 -168
- package/dist/rollout-BOYjemfR.js +624 -0
- package/dist/rollout-BOYjemfR.js.map +1 -0
- package/dist/rubric-predictive-validity-B3xmbmS1.js +141 -0
- package/dist/rubric-predictive-validity-B3xmbmS1.js.map +1 -0
- package/dist/rubric-predictive-validity-Ku_clp_1.d.ts +74 -0
- package/dist/rubric-predictive-validity-Ku_clp_1.d.ts.map +1 -0
- package/dist/run-evidence-ClRX_8A9.d.ts +225 -0
- package/dist/run-evidence-ClRX_8A9.d.ts.map +1 -0
- package/dist/run-record-BuoE80Dq.js +467 -0
- package/dist/{chunk-56TAVBOK.js.map → run-record-BuoE80Dq.js.map} +1 -1
- package/dist/run-record-CnZu_gjl.d.ts +357 -0
- package/dist/run-record-CnZu_gjl.d.ts.map +1 -0
- package/dist/run-score-iEEAWiBY.js +41 -0
- package/dist/run-score-iEEAWiBY.js.map +1 -0
- package/dist/runtime-trajectory-1gyaTOoC.js +93 -0
- package/dist/runtime-trajectory-1gyaTOoC.js.map +1 -0
- package/dist/runtime-trajectory-BvSZcCHD.d.ts +50 -0
- package/dist/runtime-trajectory-BvSZcCHD.d.ts.map +1 -0
- package/dist/schema-BtVldJ3T.d.ts +206 -0
- package/dist/schema-BtVldJ3T.d.ts.map +1 -0
- package/dist/schema-C6DW4ZHR.js +821 -0
- package/dist/schema-C6DW4ZHR.js.map +1 -0
- package/dist/schema-CRhEY1SO.js +69 -0
- package/dist/schema-CRhEY1SO.js.map +1 -0
- package/dist/schema-Cef2cFmb.d.ts +408 -0
- package/dist/schema-Cef2cFmb.d.ts.map +1 -0
- package/dist/semantic-concept-judge-B6cWNJ2K.js +725 -0
- package/dist/semantic-concept-judge-B6cWNJ2K.js.map +1 -0
- package/dist/sequential-Br0mAPHA.js +148 -0
- package/dist/sequential-Br0mAPHA.js.map +1 -0
- package/dist/sequential-CYwq6Ff_.d.ts +141 -0
- package/dist/sequential-CYwq6Ff_.d.ts.map +1 -0
- package/dist/series-convergence-CjO2QdRW.js +43 -0
- package/dist/series-convergence-CjO2QdRW.js.map +1 -0
- package/dist/series-convergence-ofsqPWhs.d.ts +35 -0
- package/dist/series-convergence-ofsqPWhs.d.ts.map +1 -0
- package/dist/server-m5D9cvnG.js +1040 -0
- package/dist/server-m5D9cvnG.js.map +1 -0
- package/dist/skill-usage-C_pXm7lP.d.ts +534 -0
- package/dist/skill-usage-C_pXm7lP.d.ts.map +1 -0
- package/dist/skillopt-optimization-method-B7wX7XkF.d.ts +1739 -0
- package/dist/skillopt-optimization-method-B7wX7XkF.d.ts.map +1 -0
- package/dist/skillopt-optimization-method-D4ODwFVV.js +7712 -0
- package/dist/skillopt-optimization-method-D4ODwFVV.js.map +1 -0
- package/dist/statistics-Cmj6nynr.d.ts +514 -0
- package/dist/statistics-Cmj6nynr.d.ts.map +1 -0
- package/dist/statistics-CnnxdpOg.js +1437 -0
- package/dist/statistics-CnnxdpOg.js.map +1 -0
- package/dist/store-CT9YIIve.d.ts +117 -0
- package/dist/store-CT9YIIve.d.ts.map +1 -0
- package/dist/store-CxJry_cs.d.ts +229 -0
- package/dist/store-CxJry_cs.d.ts.map +1 -0
- package/dist/storyboard/index.d.ts +84 -203
- package/dist/storyboard/index.d.ts.map +1 -0
- package/dist/storyboard/index.js +609 -542
- package/dist/storyboard/index.js.map +1 -1
- package/dist/summary-report-BNs5nmXI.js +862 -0
- package/dist/summary-report-BNs5nmXI.js.map +1 -0
- package/dist/summary-report-Cj9gdw4i.d.ts +407 -0
- package/dist/summary-report-Cj9gdw4i.d.ts.map +1 -0
- package/dist/supervisor-run/index.d.ts +2 -959
- package/dist/supervisor-run/index.js +2 -65
- package/dist/supervisor-run-_lnTLM3z.js +1679 -0
- package/dist/supervisor-run-_lnTLM3z.js.map +1 -0
- package/dist/task-failure-attributes-CQZlB3et.js +311 -0
- package/dist/task-failure-attributes-CQZlB3et.js.map +1 -0
- package/dist/test-graded-scenario-BsqWLmPt.js +318 -0
- package/dist/test-graded-scenario-BsqWLmPt.js.map +1 -0
- package/dist/test-graded-scenario-D1TaI2va.d.ts +141 -0
- package/dist/test-graded-scenario-D1TaI2va.d.ts.map +1 -0
- package/dist/tools-BmuN627J.js +1085 -0
- package/dist/tools-BmuN627J.js.map +1 -0
- package/dist/trace-attributes.d.ts +2 -52
- package/dist/trace-attributes.js +131 -61
- package/dist/trace-attributes.js.map +1 -1
- package/dist/traces.d.ts +12 -2365
- package/dist/traces.js +12 -252
- package/dist/trajectory-D_7rLrvE.js +56 -0
- package/dist/trajectory-D_7rLrvE.js.map +1 -0
- package/dist/types-DGsxbAEd.d.ts +387 -0
- package/dist/types-DGsxbAEd.d.ts.map +1 -0
- package/dist/types-k9tZGKUg.d.ts +640 -0
- package/dist/types-k9tZGKUg.d.ts.map +1 -0
- package/dist/verdict-Dps8_okt.d.ts +37 -0
- package/dist/verdict-Dps8_okt.d.ts.map +1 -0
- package/dist/wire/index.d.ts +702 -1173
- package/dist/wire/index.d.ts.map +1 -0
- package/dist/wire/index.js +2 -81
- package/package.json +17 -9
- package/dist/benchmarks/index.js.map +0 -1
- package/dist/campaign/index.js.map +0 -1
- package/dist/chunk-2QU3YOPR.js +0 -7374
- package/dist/chunk-2QU3YOPR.js.map +0 -1
- package/dist/chunk-3OCR4R5I.js +0 -728
- package/dist/chunk-3OCR4R5I.js.map +0 -1
- package/dist/chunk-3RF76KTD.js +0 -84
- package/dist/chunk-3RF76KTD.js.map +0 -1
- package/dist/chunk-56TAVBOK.js +0 -698
- package/dist/chunk-5DTSBUL2.js +0 -159
- package/dist/chunk-5DTSBUL2.js.map +0 -1
- package/dist/chunk-7FO3TNPI.js +0 -232
- package/dist/chunk-7FO3TNPI.js.map +0 -1
- package/dist/chunk-7ZZMD7UK.js +0 -386
- package/dist/chunk-7ZZMD7UK.js.map +0 -1
- package/dist/chunk-BOD4O7OF.js +0 -40
- package/dist/chunk-BOD4O7OF.js.map +0 -1
- package/dist/chunk-BSO5JDQH.js +0 -2335
- package/dist/chunk-BSO5JDQH.js.map +0 -1
- package/dist/chunk-C6LXANRU.js +0 -1550
- package/dist/chunk-C6LXANRU.js.map +0 -1
- package/dist/chunk-DODXQREJ.js +0 -752
- package/dist/chunk-DODXQREJ.js.map +0 -1
- package/dist/chunk-DRYIUNWY.js +0 -622
- package/dist/chunk-DRYIUNWY.js.map +0 -1
- package/dist/chunk-E7QXT7SX.js +0 -183
- package/dist/chunk-E7QXT7SX.js.map +0 -1
- package/dist/chunk-EG66UGL4.js +0 -341
- package/dist/chunk-EG66UGL4.js.map +0 -1
- package/dist/chunk-FXTVJPYD.js +0 -576
- package/dist/chunk-FXTVJPYD.js.map +0 -1
- package/dist/chunk-G7MGMCZD.js +0 -153
- package/dist/chunk-G7MGMCZD.js.map +0 -1
- package/dist/chunk-GGE4NNQT.js +0 -65
- package/dist/chunk-GGE4NNQT.js.map +0 -1
- package/dist/chunk-H23X7XKK.js +0 -181
- package/dist/chunk-H23X7XKK.js.map +0 -1
- package/dist/chunk-HHWE3POT.js +0 -94
- package/dist/chunk-HHWE3POT.js.map +0 -1
- package/dist/chunk-HPWUNB47.js +0 -289
- package/dist/chunk-HPWUNB47.js.map +0 -1
- package/dist/chunk-IYCLP2N2.js +0 -766
- package/dist/chunk-IYCLP2N2.js.map +0 -1
- package/dist/chunk-JHCHEVET.js +0 -274
- package/dist/chunk-JHCHEVET.js.map +0 -1
- package/dist/chunk-JQSF5DQT.js +0 -701
- package/dist/chunk-JQSF5DQT.js.map +0 -1
- package/dist/chunk-K4DBDHLK.js +0 -158
- package/dist/chunk-K4DBDHLK.js.map +0 -1
- package/dist/chunk-K6N6XJJX.js +0 -306
- package/dist/chunk-K6N6XJJX.js.map +0 -1
- package/dist/chunk-M4YBQKIJ.js +0 -1040
- package/dist/chunk-M4YBQKIJ.js.map +0 -1
- package/dist/chunk-MA6HLL3S.js +0 -65
- package/dist/chunk-MA6HLL3S.js.map +0 -1
- package/dist/chunk-MAZ26DC7.js +0 -99
- package/dist/chunk-MAZ26DC7.js.map +0 -1
- package/dist/chunk-NPCTHQIO.js +0 -91
- package/dist/chunk-NPCTHQIO.js.map +0 -1
- package/dist/chunk-NY44NC4A.js +0 -1056
- package/dist/chunk-NY44NC4A.js.map +0 -1
- package/dist/chunk-OIUOT4QD.js +0 -44
- package/dist/chunk-OIUOT4QD.js.map +0 -1
- package/dist/chunk-ONWEPEDO.js +0 -57
- package/dist/chunk-ONWEPEDO.js.map +0 -1
- package/dist/chunk-OWN5NPMC.js +0 -152
- package/dist/chunk-OWN5NPMC.js.map +0 -1
- package/dist/chunk-P6FYH6K4.js +0 -1161
- package/dist/chunk-P6FYH6K4.js.map +0 -1
- package/dist/chunk-PC4UYEBM.js +0 -166
- package/dist/chunk-PC4UYEBM.js.map +0 -1
- package/dist/chunk-PC5DOSM7.js +0 -579
- package/dist/chunk-PC5DOSM7.js.map +0 -1
- package/dist/chunk-PXE2VKMX.js +0 -140
- package/dist/chunk-PXE2VKMX.js.map +0 -1
- package/dist/chunk-PZ5AY32C.js +0 -10
- package/dist/chunk-PZ5AY32C.js.map +0 -1
- package/dist/chunk-QB6BDBP2.js +0 -4464
- package/dist/chunk-QB6BDBP2.js.map +0 -1
- package/dist/chunk-RXHCETDZ.js +0 -536
- package/dist/chunk-RXHCETDZ.js.map +0 -1
- package/dist/chunk-RZTMDUO7.js +0 -49
- package/dist/chunk-RZTMDUO7.js.map +0 -1
- package/dist/chunk-SFLLL76A.js +0 -669
- package/dist/chunk-SFLLL76A.js.map +0 -1
- package/dist/chunk-SZLVEKMJ.js +0 -1446
- package/dist/chunk-SZLVEKMJ.js.map +0 -1
- package/dist/chunk-T4SQEITX.js +0 -95
- package/dist/chunk-T4SQEITX.js.map +0 -1
- package/dist/chunk-T6RLYGAD.js +0 -158
- package/dist/chunk-T6RLYGAD.js.map +0 -1
- package/dist/chunk-TJVT4QFF.js +0 -911
- package/dist/chunk-TJVT4QFF.js.map +0 -1
- package/dist/chunk-TQ7LNKZ3.js +0 -136
- package/dist/chunk-TQ7LNKZ3.js.map +0 -1
- package/dist/chunk-U4L7JRPZ.js +0 -1706
- package/dist/chunk-U4L7JRPZ.js.map +0 -1
- package/dist/chunk-U4PHLT2N.js +0 -419
- package/dist/chunk-U4PHLT2N.js.map +0 -1
- package/dist/chunk-VCZ5FQYW.js +0 -928
- package/dist/chunk-VCZ5FQYW.js.map +0 -1
- package/dist/chunk-VI2UW6B6.js +0 -162
- package/dist/chunk-VI2UW6B6.js.map +0 -1
- package/dist/chunk-VQMK5FMP.js +0 -247
- package/dist/chunk-VQMK5FMP.js.map +0 -1
- package/dist/chunk-WGXIEX7P.js +0 -116
- package/dist/chunk-WGXIEX7P.js.map +0 -1
- package/dist/chunk-WVATSFCP.js +0 -1553
- package/dist/chunk-WVATSFCP.js.map +0 -1
- package/dist/chunk-X4YIBDER.js +0 -1662
- package/dist/chunk-X4YIBDER.js.map +0 -1
- package/dist/chunk-YQN4ICPP.js +0 -355
- package/dist/chunk-YQN4ICPP.js.map +0 -1
- package/dist/chunk-ZET2UAYW.js +0 -89
- package/dist/chunk-ZET2UAYW.js.map +0 -1
- package/dist/chunk-ZHTZ4EYI.js +0 -1212
- package/dist/chunk-ZHTZ4EYI.js.map +0 -1
- package/dist/control.js.map +0 -1
- package/dist/hosted/index.js.map +0 -1
- package/dist/matrix/index.js.map +0 -1
- package/dist/reporting.js.map +0 -1
- package/dist/rollout/index.js.map +0 -1
- package/dist/run-campaign-OJJ7CZF4.js +0 -18
- package/dist/run-campaign-OJJ7CZF4.js.map +0 -1
- package/dist/supervisor-run/index.js.map +0 -1
- package/dist/traces.js.map +0 -1
- package/dist/wire/index.js.map +0 -1
package/dist/control.d.ts
CHANGED
|
@@ -1,1030 +1,3 @@
|
|
|
1
|
-
|
|
2
|
-
|
|
3
|
-
|
|
4
|
-
wallMs?: number;
|
|
5
|
-
calls?: number;
|
|
6
|
-
usd?: number;
|
|
7
|
-
}
|
|
8
|
-
interface RunOutcome$1 {
|
|
9
|
-
score?: number;
|
|
10
|
-
pass?: boolean;
|
|
11
|
-
failureClass?: FailureClass;
|
|
12
|
-
notes?: string;
|
|
13
|
-
}
|
|
14
|
-
/**
|
|
15
|
-
* Layer — optional classification in a nested build workflow.
|
|
16
|
-
* `builder`: the meta-agent editing a project (e.g. agent-builder Forge chat).
|
|
17
|
-
* `app-build`: sandbox harness that compiled + tested the generated scaffold.
|
|
18
|
-
* `app-runtime`: a run of the generated agent against a domain scenario.
|
|
19
|
-
* `meta`: any meta-eval (judge replay, correlation analysis).
|
|
20
|
-
*/
|
|
21
|
-
type RunLayer = 'builder' | 'app-build' | 'app-runtime' | 'meta' | 'custom';
|
|
22
|
-
interface Run {
|
|
23
|
-
runId: string;
|
|
24
|
-
/**
|
|
25
|
-
* Stable identifier of the scenario being executed.
|
|
26
|
-
*
|
|
27
|
-
* Always populated on the persisted Run — but `TraceEmitter.startRun` accepts
|
|
28
|
-
* input WITHOUT this field, substituting a sensible default
|
|
29
|
-
* (`run.layer ?? run.tags?.['kind'] ?? 'runtime'`) when the caller has no
|
|
30
|
-
* curated scenario to anchor to (runtime / operator / meta-eval runs). This
|
|
31
|
-
* keeps the persisted shape unambiguous for downstream filters + aggregations
|
|
32
|
-
* while removing the boilerplate of inventing placeholder ids at the call site.
|
|
33
|
-
*/
|
|
34
|
-
scenarioId: string;
|
|
35
|
-
variantId?: string;
|
|
36
|
-
datasetVersion?: string;
|
|
37
|
-
/** Git SHA of agent code at run time. */
|
|
38
|
-
codeSha?: string;
|
|
39
|
-
/** Hash of the prompt template + any system prompt. */
|
|
40
|
-
promptSha?: string;
|
|
41
|
-
/** Model id + date + system-prompt hash, concatenated. */
|
|
42
|
-
modelFingerprint?: string;
|
|
43
|
-
seed?: number;
|
|
44
|
-
/** Arbitrary environment markers (shell, docker version, tz). */
|
|
45
|
-
envFingerprint?: Record<string, string>;
|
|
46
|
-
/** Version of the redaction rules applied to this run. */
|
|
47
|
-
redactionVersion?: string;
|
|
48
|
-
/** Parent run in a nested build workflow. A builder run's children are
|
|
49
|
-
* app-build runs; those children are app-runtime runs. */
|
|
50
|
-
parentRunId?: string;
|
|
51
|
-
/** Stable project identifier — groups runs across chats + sessions. */
|
|
52
|
-
projectId?: string;
|
|
53
|
-
/** Chat/conversation identifier within a project. */
|
|
54
|
-
chatId?: string;
|
|
55
|
-
/** Layer classification — hint for aggregation; not enforced. */
|
|
56
|
-
layer?: RunLayer;
|
|
57
|
-
startedAt: number;
|
|
58
|
-
endedAt?: number;
|
|
59
|
-
status: RunStatus;
|
|
60
|
-
outcome?: RunOutcome$1;
|
|
61
|
-
budget?: BudgetSpec;
|
|
62
|
-
/** Free-form labels for downstream grouping. */
|
|
63
|
-
tags?: Record<string, string>;
|
|
64
|
-
}
|
|
65
|
-
type SpanKind = 'agent' | 'llm' | 'tool' | 'retrieval' | 'judge' | 'sandbox' | 'custom';
|
|
66
|
-
type SpanStatus = 'ok' | 'error';
|
|
67
|
-
interface SpanBase {
|
|
68
|
-
spanId: string;
|
|
69
|
-
parentSpanId?: string;
|
|
70
|
-
runId: string;
|
|
71
|
-
kind: SpanKind;
|
|
72
|
-
name: string;
|
|
73
|
-
startedAt: number;
|
|
74
|
-
endedAt?: number;
|
|
75
|
-
status?: SpanStatus;
|
|
76
|
-
error?: string;
|
|
77
|
-
/** Anything not covered by typed fields. Kept deliberately free-form. */
|
|
78
|
-
attributes?: Record<string, unknown>;
|
|
79
|
-
}
|
|
80
|
-
interface Message {
|
|
81
|
-
role: 'system' | 'user' | 'assistant' | 'tool';
|
|
82
|
-
content: string;
|
|
83
|
-
tokens?: number;
|
|
84
|
-
/** Multi-modal content descriptors; blobs themselves live in Artifacts. */
|
|
85
|
-
images?: Array<{
|
|
86
|
-
artifactId?: string;
|
|
87
|
-
url?: string;
|
|
88
|
-
mime?: string;
|
|
89
|
-
}>;
|
|
90
|
-
}
|
|
91
|
-
interface LlmSpan extends SpanBase {
|
|
92
|
-
kind: 'llm';
|
|
93
|
-
model: string;
|
|
94
|
-
messages: Message[];
|
|
95
|
-
output?: string;
|
|
96
|
-
inputTokens?: number;
|
|
97
|
-
/** All generated tokens, including the reasoning subset when present. */
|
|
98
|
-
outputTokens?: number;
|
|
99
|
-
cachedTokens?: number;
|
|
100
|
-
cacheWriteTokens?: number;
|
|
101
|
-
/** Reasoning-token subset of `outputTokens`. */
|
|
102
|
-
reasoningTokens?: number;
|
|
103
|
-
costUsd?: number;
|
|
104
|
-
finishReason?: string;
|
|
105
|
-
}
|
|
106
|
-
interface ToolSpan extends SpanBase {
|
|
107
|
-
kind: 'tool';
|
|
108
|
-
toolName: string;
|
|
109
|
-
args: unknown;
|
|
110
|
-
/** False when the source observed the call but did not capture its arguments. */
|
|
111
|
-
argsCaptured?: boolean;
|
|
112
|
-
result?: unknown;
|
|
113
|
-
latencyMs?: number;
|
|
114
|
-
}
|
|
115
|
-
interface RetrievalSpan extends SpanBase {
|
|
116
|
-
kind: 'retrieval';
|
|
117
|
-
query: string;
|
|
118
|
-
hits: Array<{
|
|
119
|
-
docId: string;
|
|
120
|
-
score: number;
|
|
121
|
-
content?: string;
|
|
122
|
-
}>;
|
|
123
|
-
}
|
|
124
|
-
interface JudgeSpan extends SpanBase {
|
|
125
|
-
kind: 'judge';
|
|
126
|
-
judgeId: string;
|
|
127
|
-
/** Span this judgment applies to. */
|
|
128
|
-
targetSpanId: string;
|
|
129
|
-
dimension: string;
|
|
130
|
-
/** Numeric score (free-range; interpretation up to the judge). */
|
|
131
|
-
score: number;
|
|
132
|
-
rationale?: string;
|
|
133
|
-
evidence?: string;
|
|
134
|
-
}
|
|
135
|
-
interface SandboxSpan extends SpanBase {
|
|
136
|
-
kind: 'sandbox';
|
|
137
|
-
image?: string;
|
|
138
|
-
command?: string;
|
|
139
|
-
exitCode?: number;
|
|
140
|
-
testsTotal?: number;
|
|
141
|
-
testsPassed?: number;
|
|
142
|
-
stdoutHash?: string;
|
|
143
|
-
stderrHash?: string;
|
|
144
|
-
/** Duration in ms; the harness fills this explicitly (endedAt - startedAt may miss setup). */
|
|
145
|
-
wallMs?: number;
|
|
146
|
-
}
|
|
147
|
-
interface GenericSpan extends SpanBase {
|
|
148
|
-
kind: 'agent' | 'custom';
|
|
149
|
-
}
|
|
150
|
-
type Span = LlmSpan | ToolSpan | RetrievalSpan | JudgeSpan | SandboxSpan | GenericSpan;
|
|
151
|
-
type EventKind = 'log' | 'error' | 'budget_decrement' | 'budget_breach' | 'state_mutation' | 'policy_violation' | 'redaction_applied' | 'custom';
|
|
152
|
-
interface TraceEvent {
|
|
153
|
-
eventId: string;
|
|
154
|
-
runId: string;
|
|
155
|
-
spanId?: string;
|
|
156
|
-
kind: EventKind;
|
|
157
|
-
timestamp: number;
|
|
158
|
-
payload: Record<string, unknown>;
|
|
159
|
-
}
|
|
160
|
-
interface BudgetLedgerEntry {
|
|
161
|
-
runId: string;
|
|
162
|
-
dimension: keyof BudgetSpec;
|
|
163
|
-
limit: number;
|
|
164
|
-
consumed: number;
|
|
165
|
-
remaining: number;
|
|
166
|
-
timestamp: number;
|
|
167
|
-
breached: boolean;
|
|
168
|
-
/** Span that triggered this entry, if any. */
|
|
169
|
-
spanId?: string;
|
|
170
|
-
}
|
|
171
|
-
interface Artifact {
|
|
172
|
-
artifactId: string;
|
|
173
|
-
runId: string;
|
|
174
|
-
spanId?: string;
|
|
175
|
-
contentType: string;
|
|
176
|
-
sizeBytes: number;
|
|
177
|
-
/** sha256 in hex. */
|
|
178
|
-
hash: string;
|
|
179
|
-
/** External storage URL (R2, S3, filesystem path). */
|
|
180
|
-
storageUrl?: string;
|
|
181
|
-
/** Inline content for small blobs — keep under ~64KB. */
|
|
182
|
-
inlineContent?: string;
|
|
183
|
-
}
|
|
184
|
-
type FailureClass = 'success' | 'reasoning_error' | 'tool_selection_error' | 'tool_argument_error' | 'tool_recovery_failure' | 'hallucination' | 'instruction_following' | 'safety_refusal_miss' | 'policy_violation' | 'budget_exceeded' | 'format_drift' | 'permission_escalation' | 'pii_leak' | 'cost_overrun' | 'timeout' | 'sandbox_failure' | 'missing_user_data' | 'missing_domain_data' | 'missing_codebase_context' | 'missing_runtime_context' | 'missing_credentials' | 'missing_integration_connection' | 'missing_integration_scope' | 'integration_approval_required' | 'integration_auth_expired' | 'integration_provider_failure' | 'bad_integration_manifest' | 'unsafe_integration_write_denied' | 'stale_external_data' | 'bad_retrieval' | 'insufficient_evidence' | 'contradictory_evidence' | 'ambiguous_user_intent' | 'knowledge_readiness_blocked' | 'unknown';
|
|
185
|
-
|
|
186
|
-
interface RunFilter {
|
|
187
|
-
scenarioId?: string;
|
|
188
|
-
variantId?: string;
|
|
189
|
-
status?: RunStatus;
|
|
190
|
-
since?: number;
|
|
191
|
-
until?: number;
|
|
192
|
-
tag?: {
|
|
193
|
-
key: string;
|
|
194
|
-
value: string;
|
|
195
|
-
};
|
|
196
|
-
parentRunId?: string;
|
|
197
|
-
projectId?: string;
|
|
198
|
-
chatId?: string;
|
|
199
|
-
layer?: RunLayer;
|
|
200
|
-
}
|
|
201
|
-
interface SpanFilter {
|
|
202
|
-
runId?: string;
|
|
203
|
-
parentSpanId?: string;
|
|
204
|
-
kind?: SpanKind;
|
|
205
|
-
name?: string;
|
|
206
|
-
toolName?: string;
|
|
207
|
-
judgeId?: string;
|
|
208
|
-
since?: number;
|
|
209
|
-
until?: number;
|
|
210
|
-
}
|
|
211
|
-
interface EventFilter {
|
|
212
|
-
runId?: string;
|
|
213
|
-
spanId?: string;
|
|
214
|
-
kind?: EventKind;
|
|
215
|
-
since?: number;
|
|
216
|
-
until?: number;
|
|
217
|
-
}
|
|
218
|
-
interface TraceStore {
|
|
219
|
-
appendRun(run: Run): Promise<void>;
|
|
220
|
-
updateRun(runId: string, patch: Partial<Run>): Promise<void>;
|
|
221
|
-
appendSpan(span: Span): Promise<void>;
|
|
222
|
-
updateSpan(spanId: string, patch: Partial<Span>): Promise<void>;
|
|
223
|
-
appendEvent(event: TraceEvent): Promise<void>;
|
|
224
|
-
appendArtifact(artifact: Artifact): Promise<void>;
|
|
225
|
-
appendBudgetEntry(entry: BudgetLedgerEntry): Promise<void>;
|
|
226
|
-
getRun(runId: string): Promise<Run | undefined>;
|
|
227
|
-
listRuns(filter?: RunFilter): Promise<Run[]>;
|
|
228
|
-
spans(filter?: SpanFilter): Promise<Span[]>;
|
|
229
|
-
events(filter?: EventFilter): Promise<TraceEvent[]>;
|
|
230
|
-
budget(runId: string): Promise<BudgetLedgerEntry[]>;
|
|
231
|
-
artifacts(runId: string): Promise<Artifact[]>;
|
|
232
|
-
}
|
|
233
|
-
|
|
234
|
-
/**
|
|
235
|
-
* TraceEmitter — hierarchical span builder that auto-parents using an
|
|
236
|
-
* internal stack. One emitter per Run; emitters do NOT share state.
|
|
237
|
-
*
|
|
238
|
-
* Convenience methods (`llm`, `tool`, `retrieval`, `judge`, `sandbox`)
|
|
239
|
-
* return a `SpanHandle` with `.end()` / `.fail()` so callers don't
|
|
240
|
-
* have to thread spanIds manually. For async workflows that can't use
|
|
241
|
-
* the stack (e.g. fan-out parallel calls), pass `parentSpanId`
|
|
242
|
-
* explicitly.
|
|
243
|
-
*/
|
|
244
|
-
|
|
245
|
-
interface SpanHandle<S extends Span = Span> {
|
|
246
|
-
span: S;
|
|
247
|
-
end(patch?: Partial<S>): Promise<void>;
|
|
248
|
-
fail(error: string | Error, patch?: Partial<S>): Promise<void>;
|
|
249
|
-
}
|
|
250
|
-
interface RunCompleteHookContext {
|
|
251
|
-
runId: string;
|
|
252
|
-
emitter: TraceEmitter;
|
|
253
|
-
store: TraceStore;
|
|
254
|
-
/** Outcome the caller passed to `endRun` (undefined for `abortRun`). */
|
|
255
|
-
outcome?: RunOutcome$1;
|
|
256
|
-
/** Final run status. */
|
|
257
|
-
status: 'completed' | 'failed' | 'aborted';
|
|
258
|
-
}
|
|
259
|
-
type RunCompleteHook = (ctx: RunCompleteHookContext) => Promise<void> | void;
|
|
260
|
-
interface TraceEmitterOptions {
|
|
261
|
-
runId?: string;
|
|
262
|
-
/** Inject a clock for deterministic tests. */
|
|
263
|
-
now?: () => number;
|
|
264
|
-
/** Inject an id generator for deterministic tests. */
|
|
265
|
-
id?: () => string;
|
|
266
|
-
/**
|
|
267
|
-
* Hooks fired after `endRun` / `abortRun` writes the final run state.
|
|
268
|
-
* Designed for trace-analyst auto-execution, integrity assertions, and
|
|
269
|
-
* outbound notifications. Hooks run sequentially in the order supplied.
|
|
270
|
-
*
|
|
271
|
-
* By default a hook that throws is swallowed and logged as a `note` event
|
|
272
|
-
* on the run — auto-orchestration must not crash the underlying flow.
|
|
273
|
-
* Set `hookErrors: 'throw'` to propagate.
|
|
274
|
-
*/
|
|
275
|
-
onRunComplete?: RunCompleteHook[];
|
|
276
|
-
/** `'swallow'` (default) | `'throw'`. */
|
|
277
|
-
hookErrors?: 'swallow' | 'throw';
|
|
278
|
-
}
|
|
279
|
-
declare class TraceEmitter {
|
|
280
|
-
private store;
|
|
281
|
-
private stack;
|
|
282
|
-
private _runId;
|
|
283
|
-
private now;
|
|
284
|
-
private id;
|
|
285
|
-
private hooks;
|
|
286
|
-
private hookErrors;
|
|
287
|
-
constructor(store: TraceStore, options?: TraceEmitterOptions);
|
|
288
|
-
get runId(): string;
|
|
289
|
-
get traceStore(): TraceStore;
|
|
290
|
-
/** Append a hook after construction (e.g. attach the trace analyst). */
|
|
291
|
-
addRunCompleteHook(hook: RunCompleteHook): void;
|
|
292
|
-
/**
|
|
293
|
-
* Begin a Run.
|
|
294
|
-
*
|
|
295
|
-
* `scenarioId` is required on the persisted Run shape — every Run downstream
|
|
296
|
-
* gets a non-empty scenarioId so filters and aggregations stay simple — but
|
|
297
|
-
* the INPUT here accepts it as optional. When omitted, startRun substitutes
|
|
298
|
-
* a sensible default (`run.layer ?? run.tags?.['kind'] ?? 'runtime'`) so
|
|
299
|
-
* runtime / operator / meta-eval runs that have no curated-scenario corpus
|
|
300
|
-
* to anchor to don't have to invent placeholder strings at the call site.
|
|
301
|
-
*/
|
|
302
|
-
startRun(run: Omit<Run, 'runId' | 'scenarioId' | 'startedAt' | 'status'> & {
|
|
303
|
-
scenarioId?: string;
|
|
304
|
-
}): Promise<Run>;
|
|
305
|
-
endRun(outcome?: RunOutcome$1): Promise<void>;
|
|
306
|
-
abortRun(reason: string): Promise<void>;
|
|
307
|
-
private runHooks;
|
|
308
|
-
span<S extends Span = Span>(init: {
|
|
309
|
-
kind: SpanKind;
|
|
310
|
-
name: string;
|
|
311
|
-
parentSpanId?: string;
|
|
312
|
-
attributes?: Record<string, unknown>;
|
|
313
|
-
} & Partial<Omit<S, 'spanId' | 'runId' | 'startedAt' | 'kind' | 'name'>>): Promise<SpanHandle<S>>;
|
|
314
|
-
private handle;
|
|
315
|
-
private pop;
|
|
316
|
-
llm(init: Omit<LlmSpan, 'spanId' | 'runId' | 'kind' | 'startedAt'>): Promise<SpanHandle<LlmSpan>>;
|
|
317
|
-
tool(init: Omit<ToolSpan, 'spanId' | 'runId' | 'kind' | 'startedAt'>): Promise<SpanHandle<ToolSpan>>;
|
|
318
|
-
retrieval(init: Omit<RetrievalSpan, 'spanId' | 'runId' | 'kind' | 'startedAt'>): Promise<SpanHandle<RetrievalSpan>>;
|
|
319
|
-
recordJudge(verdict: Omit<JudgeSpan, 'spanId' | 'runId' | 'kind' | 'startedAt' | 'endedAt'>): Promise<JudgeSpan>;
|
|
320
|
-
sandbox(init: Omit<SandboxSpan, 'spanId' | 'runId' | 'kind' | 'startedAt'>): Promise<SpanHandle<SandboxSpan>>;
|
|
321
|
-
emit(event: {
|
|
322
|
-
kind: EventKind;
|
|
323
|
-
spanId?: string;
|
|
324
|
-
payload?: Record<string, unknown>;
|
|
325
|
-
}): Promise<TraceEvent>;
|
|
326
|
-
recordBudget(entry: Omit<BudgetLedgerEntry, 'runId' | 'timestamp'> & {
|
|
327
|
-
timestamp?: number;
|
|
328
|
-
}): Promise<BudgetLedgerEntry>;
|
|
329
|
-
recordArtifact(artifact: Omit<Artifact, 'artifactId' | 'runId'>): Promise<Artifact>;
|
|
330
|
-
/**
|
|
331
|
-
* Runs `fn` inside a span; auto-ends on success, auto-fails on throw.
|
|
332
|
-
* Returns the fn's return value. Use this for the 95% case.
|
|
333
|
-
*/
|
|
334
|
-
within<T>(init: Parameters<TraceEmitter['span']>[0], fn: (handle: SpanHandle) => Promise<T>): Promise<T>;
|
|
335
|
-
}
|
|
336
|
-
|
|
337
|
-
/**
|
|
338
|
-
* Policy-based agent control runtime.
|
|
339
|
-
*
|
|
340
|
-
* This is the minimal reusable loop behind driver-agent patterns:
|
|
341
|
-
*
|
|
342
|
-
* observe state -> validate -> decide next action -> act -> observe -> ...
|
|
343
|
-
*
|
|
344
|
-
* It deliberately does not model named "topologies". Direct execution,
|
|
345
|
-
* critic/revise, driver intervention, specialist calls, and human escalation
|
|
346
|
-
* are all just actions chosen by the control policy.
|
|
347
|
-
*/
|
|
348
|
-
|
|
349
|
-
type ControlSeverity = 'info' | 'warning' | 'error' | 'critical';
|
|
350
|
-
type ControlActionFailureMode = 'continue' | 'stop';
|
|
351
|
-
interface ControlEvalResult {
|
|
352
|
-
/** Stable validator or judge id. */
|
|
353
|
-
id: string;
|
|
354
|
-
/** Whether this check passed. */
|
|
355
|
-
passed: boolean;
|
|
356
|
-
/** Optional normalized score. 1 = best, 0 = worst. */
|
|
357
|
-
score?: number;
|
|
358
|
-
/** Objective validators should usually be "error" or "critical" when failed. */
|
|
359
|
-
severity?: ControlSeverity;
|
|
360
|
-
/** Human-readable result. */
|
|
361
|
-
detail?: string;
|
|
362
|
-
/** Small evidence string or pointer. Avoid large payloads. */
|
|
363
|
-
evidence?: string;
|
|
364
|
-
/** True when the result came from deterministic state, not LLM judgment. */
|
|
365
|
-
objective?: boolean;
|
|
366
|
-
/** Structured details for downstream control policies and reports. */
|
|
367
|
-
metadata?: Record<string, unknown>;
|
|
368
|
-
}
|
|
369
|
-
interface ControlBudget {
|
|
370
|
-
maxSteps: number;
|
|
371
|
-
maxWallMs?: number;
|
|
372
|
-
maxCostUsd?: number;
|
|
373
|
-
}
|
|
374
|
-
interface ControlStopPolicies<TState, TAction> {
|
|
375
|
-
/**
|
|
376
|
-
* Stop after N consecutive steps with no state fingerprint change and
|
|
377
|
-
* less than `minScoreDelta` score movement. Disabled when omitted.
|
|
378
|
-
*/
|
|
379
|
-
maxNoProgressSteps?: number;
|
|
380
|
-
/**
|
|
381
|
-
* Stop after the same action fingerprint is selected N consecutive
|
|
382
|
-
* times. Disabled when omitted.
|
|
383
|
-
*/
|
|
384
|
-
maxRepeatedActions?: number;
|
|
385
|
-
/** Minimum score movement that counts as progress. Default 0.001. */
|
|
386
|
-
minScoreDelta?: number;
|
|
387
|
-
/** Override the default JSON/string fingerprint for state comparisons. */
|
|
388
|
-
stateFingerprint?: (state: TState) => string;
|
|
389
|
-
/** Override the default JSON/string fingerprint for repeated-action checks. */
|
|
390
|
-
actionFingerprint?: (action: TAction) => string;
|
|
391
|
-
}
|
|
392
|
-
interface ControlContext<TState, TAction, TActionResult, TEval extends ControlEvalResult = ControlEvalResult> {
|
|
393
|
-
intent: string;
|
|
394
|
-
state: TState;
|
|
395
|
-
evals: TEval[];
|
|
396
|
-
history: ControlStep<TState, TAction, TActionResult, TEval>[];
|
|
397
|
-
budget: ControlBudget;
|
|
398
|
-
stepIndex: number;
|
|
399
|
-
wallMs: number;
|
|
400
|
-
spentCostUsd: number;
|
|
401
|
-
remainingCostUsd?: number;
|
|
402
|
-
abortSignal: AbortSignal;
|
|
403
|
-
emitter?: TraceEmitter;
|
|
404
|
-
}
|
|
405
|
-
type ControlDecision<TAction> = {
|
|
406
|
-
type: 'continue';
|
|
407
|
-
action: TAction;
|
|
408
|
-
reason?: string;
|
|
409
|
-
} | {
|
|
410
|
-
type: 'stop';
|
|
411
|
-
reason: string;
|
|
412
|
-
pass?: boolean;
|
|
413
|
-
score?: number;
|
|
414
|
-
/** Canonical task-failure class when this stop represents a failed task. */
|
|
415
|
-
failureClass?: FailureClass;
|
|
416
|
-
};
|
|
417
|
-
interface StopDecision {
|
|
418
|
-
stop: boolean;
|
|
419
|
-
pass: boolean;
|
|
420
|
-
reason: string;
|
|
421
|
-
score?: number;
|
|
422
|
-
failureClass?: FailureClass;
|
|
423
|
-
}
|
|
424
|
-
interface ControlActionOutcome<TActionResult> {
|
|
425
|
-
ok: boolean;
|
|
426
|
-
result?: TActionResult;
|
|
427
|
-
error?: string;
|
|
428
|
-
costUsd?: number;
|
|
429
|
-
durationMs: number;
|
|
430
|
-
}
|
|
431
|
-
interface ControlRuntimeError {
|
|
432
|
-
phase: 'observe' | 'validate' | 'decide' | 'act' | 'stop-policy' | 'on-step' | 'trace';
|
|
433
|
-
stepIndex: number;
|
|
434
|
-
message: string;
|
|
435
|
-
}
|
|
436
|
-
interface ControlStep<TState, TAction, TActionResult, TEval extends ControlEvalResult = ControlEvalResult> {
|
|
437
|
-
index: number;
|
|
438
|
-
decision: ControlDecision<TAction>;
|
|
439
|
-
beforeState: TState;
|
|
440
|
-
afterState: TState;
|
|
441
|
-
evalsBefore: TEval[];
|
|
442
|
-
evalsAfter: TEval[];
|
|
443
|
-
actionOutcome?: ControlActionOutcome<TActionResult>;
|
|
444
|
-
startedAt: string;
|
|
445
|
-
endedAt: string;
|
|
446
|
-
}
|
|
447
|
-
interface ControlRunResult<TState, TAction, TActionResult, TEval extends ControlEvalResult = ControlEvalResult> {
|
|
448
|
-
intent: string;
|
|
449
|
-
pass: boolean;
|
|
450
|
-
completed: boolean;
|
|
451
|
-
reason: string;
|
|
452
|
-
score?: number;
|
|
453
|
-
steps: ControlStep<TState, TAction, TActionResult, TEval>[];
|
|
454
|
-
finalState: TState | undefined;
|
|
455
|
-
finalEvals: TEval[];
|
|
456
|
-
wallMs: number;
|
|
457
|
-
spentCostUsd: number;
|
|
458
|
-
/** null when the run executed without a TraceEmitter wired (no run record was persisted). */
|
|
459
|
-
runId: string | null;
|
|
460
|
-
failureClass?: FailureClass;
|
|
461
|
-
runtimeErrors: ControlRuntimeError[];
|
|
462
|
-
stoppedBy: 'policy' | 'stop-policy' | 'budget' | 'abort' | 'runtime-error';
|
|
463
|
-
}
|
|
464
|
-
interface ControlRuntimeConfig<TState, TAction, TActionResult, TEval extends ControlEvalResult = ControlEvalResult> {
|
|
465
|
-
intent: string;
|
|
466
|
-
budget?: Partial<ControlBudget>;
|
|
467
|
-
signal?: AbortSignal;
|
|
468
|
-
/** Defaults to `continue`: action failures are recorded, then the policy gets another chance. */
|
|
469
|
-
actionFailure?: ControlActionFailureMode;
|
|
470
|
-
/**
|
|
471
|
-
* Extract cost from an action result. Used for `maxCostUsd` budget
|
|
472
|
-
* enforcement and trace budget ledger emission.
|
|
473
|
-
*/
|
|
474
|
-
getActionCostUsd?: (ctx: {
|
|
475
|
-
action: TAction;
|
|
476
|
-
result: TActionResult;
|
|
477
|
-
state: TState;
|
|
478
|
-
evals: TEval[];
|
|
479
|
-
history: ControlStep<TState, TAction, TActionResult, TEval>[];
|
|
480
|
-
}) => number | undefined;
|
|
481
|
-
/** Read typed task/product state. Prefer structured state over transcript-only context. */
|
|
482
|
-
observe: (ctx: {
|
|
483
|
-
history: ControlStep<TState, TAction, TActionResult, TEval>[];
|
|
484
|
-
abortSignal: AbortSignal;
|
|
485
|
-
}) => Promise<TState> | TState;
|
|
486
|
-
/** Objective validators first, subjective judges only where objective state is insufficient. */
|
|
487
|
-
validate: (ctx: {
|
|
488
|
-
intent: string;
|
|
489
|
-
state: TState;
|
|
490
|
-
history: ControlStep<TState, TAction, TActionResult, TEval>[];
|
|
491
|
-
abortSignal: AbortSignal;
|
|
492
|
-
}) => Promise<TEval[]> | TEval[];
|
|
493
|
-
/** Choose the next control action. Can call a worker, ask user, run critic, inspect state, or stop. */
|
|
494
|
-
decide: (ctx: ControlContext<TState, TAction, TActionResult, TEval>) => Promise<ControlDecision<TAction>> | ControlDecision<TAction>;
|
|
495
|
-
/** Execute the action selected by the policy. */
|
|
496
|
-
act: (action: TAction, ctx: ControlContext<TState, TAction, TActionResult, TEval>) => Promise<TActionResult> | TActionResult;
|
|
497
|
-
/** Final stopping policy. Called before decide and after each action. */
|
|
498
|
-
shouldStop?: (ctx: ControlContext<TState, TAction, TActionResult, TEval>) => Promise<StopDecision> | StopDecision;
|
|
499
|
-
/** Optional hook for tracing or live progress updates. */
|
|
500
|
-
onStep?: (step: ControlStep<TState, TAction, TActionResult, TEval>) => Promise<void> | void;
|
|
501
|
-
/** Optional generic stuck-loop policies. Custom `shouldStop` still runs first. */
|
|
502
|
-
stopPolicies?: ControlStopPolicies<TState, TAction>;
|
|
503
|
-
/** Optional trace sink. Emits one run plus one span per control step. */
|
|
504
|
-
store?: TraceStore;
|
|
505
|
-
scenarioId?: string;
|
|
506
|
-
projectId?: string;
|
|
507
|
-
variantId?: string;
|
|
508
|
-
}
|
|
509
|
-
declare function runAgentControlLoop<TState, TAction, TActionResult, TEval extends ControlEvalResult = ControlEvalResult>(config: ControlRuntimeConfig<TState, TAction, TActionResult, TEval>): Promise<ControlRunResult<TState, TAction, TActionResult, TEval>>;
|
|
510
|
-
declare function stopOnNoProgress<TState, TAction>(maxNoProgressSteps: number, options?: Omit<ControlStopPolicies<TState, TAction>, 'maxNoProgressSteps'>): ControlStopPolicies<TState, TAction>;
|
|
511
|
-
declare function stopOnRepeatedAction<TState, TAction>(maxRepeatedActions: number, options?: Omit<ControlStopPolicies<TState, TAction>, 'maxRepeatedActions'>): ControlStopPolicies<TState, TAction>;
|
|
512
|
-
declare function objectiveEval(input: Omit<ControlEvalResult, 'objective'>): ControlEvalResult;
|
|
513
|
-
declare function subjectiveEval(input: Omit<ControlEvalResult, 'objective'>): ControlEvalResult;
|
|
514
|
-
declare function allCriticalPassed(evals: ControlEvalResult[]): boolean;
|
|
515
|
-
|
|
516
|
-
type FeedbackLabelSource = 'user' | 'judge' | 'environment' | 'metric' | 'policy' | 'system';
|
|
517
|
-
type FeedbackLabelKind = 'approve' | 'reject' | 'select' | 'edit' | 'rank' | 'rate' | 'comment' | 'metric_outcome' | 'policy_block' | 'revision_request';
|
|
518
|
-
type FeedbackSeverity = 'info' | 'warning' | 'error' | 'critical';
|
|
519
|
-
interface ProposedSideEffect {
|
|
520
|
-
type: string;
|
|
521
|
-
risk?: 'low' | 'medium' | 'high';
|
|
522
|
-
costUsd?: number;
|
|
523
|
-
externalSideEffect?: boolean;
|
|
524
|
-
requiresApproval?: boolean;
|
|
525
|
-
metadata?: Record<string, unknown>;
|
|
526
|
-
}
|
|
527
|
-
interface FeedbackLabel {
|
|
528
|
-
id?: string;
|
|
529
|
-
source: FeedbackLabelSource;
|
|
530
|
-
kind: FeedbackLabelKind;
|
|
531
|
-
value: unknown;
|
|
532
|
-
reason?: string;
|
|
533
|
-
severity?: FeedbackSeverity;
|
|
534
|
-
createdAt: string;
|
|
535
|
-
metadata?: Record<string, unknown>;
|
|
536
|
-
}
|
|
537
|
-
|
|
538
|
-
interface ActionExecutionPolicy {
|
|
539
|
-
allowedTypes?: string[];
|
|
540
|
-
blockedTypes?: string[];
|
|
541
|
-
alwaysRequireApprovalTypes?: string[];
|
|
542
|
-
autoApproveTypes?: string[];
|
|
543
|
-
requireApprovalForExternalSideEffects?: boolean;
|
|
544
|
-
requireApprovalAboveCostUsd?: number;
|
|
545
|
-
maxActionCostUsd?: number;
|
|
546
|
-
remainingBudgetUsd?: number;
|
|
547
|
-
expectedOutcomeRequired?: boolean;
|
|
548
|
-
killCriteriaRequired?: boolean;
|
|
549
|
-
}
|
|
550
|
-
interface ActionPolicyDecision {
|
|
551
|
-
allowed: boolean;
|
|
552
|
-
blocked: boolean;
|
|
553
|
-
requiresApproval: boolean;
|
|
554
|
-
reasons: string[];
|
|
555
|
-
label?: FeedbackLabel;
|
|
556
|
-
}
|
|
557
|
-
declare function evaluateActionPolicy(action: ProposedSideEffect, policy?: ActionExecutionPolicy, options?: {
|
|
558
|
-
createdAt?: string;
|
|
559
|
-
}): ActionPolicyDecision;
|
|
560
|
-
|
|
561
|
-
/**
|
|
562
|
-
* Propose / Verify / Review — the core multi-shot primitive.
|
|
563
|
-
*
|
|
564
|
-
* shot N: propose(state, priorReview) → new state
|
|
565
|
-
* verify(state) → pass/fail, optional layers
|
|
566
|
-
* review(state, verification, memory) → observations + next-shot
|
|
567
|
-
* instruction + shouldContinue
|
|
568
|
-
* memory.append(entry)
|
|
569
|
-
*
|
|
570
|
-
* Roles are strictly separated:
|
|
571
|
-
*
|
|
572
|
-
* - The WORKER is whatever the caller wraps in `propose`. It is
|
|
573
|
-
* stateful — caller owns its resume/session mechanism.
|
|
574
|
-
* - The VERIFIER grades the state. It produces the ground truth.
|
|
575
|
-
* The reviewer cannot overturn or downgrade a verification layer.
|
|
576
|
-
* - The REVIEWER is stateless per call. Its continuity is the
|
|
577
|
-
* `ReviewMemoryStore` — durable JSONL by default, or any store
|
|
578
|
-
* implementing the interface. It reads memory + trace summary +
|
|
579
|
-
* verification and directs the NEXT proposer shot.
|
|
580
|
-
*
|
|
581
|
-
* This shape is load-bearing. The reviewer never grades; the verifier
|
|
582
|
-
* never directs. Two processes, two prompts, two concerns — which is
|
|
583
|
-
* what keeps the loop from confirmation-biasing itself into "all
|
|
584
|
-
* passed" when it didn't.
|
|
585
|
-
*
|
|
586
|
-
* Short-circuits and soft-fails are both first-class:
|
|
587
|
-
* - verify.pass === true → reviewer LLM call is skipped, memory
|
|
588
|
-
* records a success entry, loop exits.
|
|
589
|
-
* - review throws → the shot still counts; the loop uses the
|
|
590
|
-
* last-known instruction (or `fallbackInstruction`) for the next
|
|
591
|
-
* propose call. A transient reviewer failure must NEVER abort a
|
|
592
|
-
* valid arc.
|
|
593
|
-
*
|
|
594
|
-
* Composable: `propose` itself can be another `runProposeReview` call.
|
|
595
|
-
* That's the dogfooding path — a harness built on this primitive is in
|
|
596
|
-
* turn evaluable by it.
|
|
597
|
-
*/
|
|
598
|
-
|
|
599
|
-
interface Verification {
|
|
600
|
-
pass: boolean;
|
|
601
|
-
score?: number;
|
|
602
|
-
failingLayers?: string[];
|
|
603
|
-
details?: unknown;
|
|
604
|
-
}
|
|
605
|
-
interface Review {
|
|
606
|
-
observations: string;
|
|
607
|
-
diagnosis: string;
|
|
608
|
-
nextShotInstruction: string;
|
|
609
|
-
shouldContinue: boolean;
|
|
610
|
-
confidence: number;
|
|
611
|
-
}
|
|
612
|
-
interface ReviewMemoryEntry extends Review {
|
|
613
|
-
shot: number;
|
|
614
|
-
timestamp: number;
|
|
615
|
-
verification: {
|
|
616
|
-
pass: boolean;
|
|
617
|
-
score?: number;
|
|
618
|
-
failingLayers?: string[];
|
|
619
|
-
};
|
|
620
|
-
}
|
|
621
|
-
interface ProposeInput<State> {
|
|
622
|
-
shot: number;
|
|
623
|
-
goal: string;
|
|
624
|
-
state: State;
|
|
625
|
-
priorReview: Review | null;
|
|
626
|
-
abortSignal: AbortSignal;
|
|
627
|
-
emitter?: TraceEmitter;
|
|
628
|
-
}
|
|
629
|
-
interface ProposeOutput<State, Summary = unknown> {
|
|
630
|
-
state: State;
|
|
631
|
-
traceSummary?: Summary;
|
|
632
|
-
}
|
|
633
|
-
interface ReviewInput<State, Summary = unknown> {
|
|
634
|
-
shot: number;
|
|
635
|
-
goal: string;
|
|
636
|
-
state: State;
|
|
637
|
-
verification: Verification;
|
|
638
|
-
traceSummary: Summary | undefined;
|
|
639
|
-
memory: ReviewMemoryEntry[];
|
|
640
|
-
}
|
|
641
|
-
type ProposeFn<State, Summary = unknown> = (input: ProposeInput<State>) => Promise<ProposeOutput<State, Summary>>;
|
|
642
|
-
type VerifyFn<State> = (state: State) => Promise<Verification>;
|
|
643
|
-
type ReviewFn<State, Summary = unknown> = (input: ReviewInput<State, Summary>) => Promise<Review>;
|
|
644
|
-
interface ReviewMemoryStore {
|
|
645
|
-
load(): Promise<ReviewMemoryEntry[]>;
|
|
646
|
-
append(entry: ReviewMemoryEntry): Promise<void>;
|
|
647
|
-
}
|
|
648
|
-
interface ProposeReviewConfig<State, Summary = unknown> {
|
|
649
|
-
goal: string;
|
|
650
|
-
initialState: State;
|
|
651
|
-
propose: ProposeFn<State, Summary>;
|
|
652
|
-
verify: VerifyFn<State>;
|
|
653
|
-
review: ReviewFn<State, Summary>;
|
|
654
|
-
/** Hard shot cap. Default 10. */
|
|
655
|
-
maxShots?: number;
|
|
656
|
-
/** Wall-clock cap in ms. Default 10 min. */
|
|
657
|
-
maxWallMs?: number;
|
|
658
|
-
/**
|
|
659
|
-
* If the reviewer returns confidence ≤ floor on `confidenceFloorWindow`
|
|
660
|
-
* consecutive shots, terminate early. Default floor 0.3, window 2.
|
|
661
|
-
* Set window to 0 or floor to <0 to disable.
|
|
662
|
-
*/
|
|
663
|
-
confidenceFloor?: number;
|
|
664
|
-
confidenceFloorWindow?: number;
|
|
665
|
-
/** Defaults to an in-memory store if omitted. */
|
|
666
|
-
memory?: ReviewMemoryStore;
|
|
667
|
-
/** If provided, emit a Run + per-shot spans. */
|
|
668
|
-
store?: TraceStore;
|
|
669
|
-
scenarioId?: string;
|
|
670
|
-
projectId?: string;
|
|
671
|
-
variantId?: string;
|
|
672
|
-
/**
|
|
673
|
-
* Used when the reviewer soft-fails on shot 1 (no prior instruction to
|
|
674
|
-
* fall back to). Default is a generic "inspect failures and fix".
|
|
675
|
-
*/
|
|
676
|
-
fallbackInstruction?: string;
|
|
677
|
-
}
|
|
678
|
-
interface ProposeReviewShot<State, Summary = unknown> {
|
|
679
|
-
shot: number;
|
|
680
|
-
state: State;
|
|
681
|
-
verification: Verification;
|
|
682
|
-
traceSummary: Summary | undefined;
|
|
683
|
-
review: Review;
|
|
684
|
-
reviewAvailable: boolean;
|
|
685
|
-
reviewError?: string;
|
|
686
|
-
durationMs: number;
|
|
687
|
-
}
|
|
688
|
-
interface ProposeReviewReport<State, Summary = unknown> {
|
|
689
|
-
runId: string | null;
|
|
690
|
-
completed: boolean;
|
|
691
|
-
shots: ProposeReviewShot<State, Summary>[];
|
|
692
|
-
finalState: State;
|
|
693
|
-
finalVerification: Verification;
|
|
694
|
-
failureClass?: FailureClass;
|
|
695
|
-
wallMs: number;
|
|
696
|
-
score: number;
|
|
697
|
-
}
|
|
698
|
-
declare function runProposeReview<State, Summary = unknown>(config: ProposeReviewConfig<State, Summary>): Promise<ProposeReviewReport<State, Summary>>;
|
|
699
|
-
|
|
700
|
-
interface ProposeReviewControlState<State, Summary = unknown> {
|
|
701
|
-
shot: number;
|
|
702
|
-
state: State;
|
|
703
|
-
priorReview: Review | null;
|
|
704
|
-
verification: Verification;
|
|
705
|
-
traceSummary?: Summary;
|
|
706
|
-
memory: ReviewMemoryEntry[];
|
|
707
|
-
completed: boolean;
|
|
708
|
-
reviewAvailable: boolean;
|
|
709
|
-
reviewError?: string;
|
|
710
|
-
}
|
|
711
|
-
interface ProposeReviewControlAction {
|
|
712
|
-
type: 'propose-review-shot';
|
|
713
|
-
shot: number;
|
|
714
|
-
}
|
|
715
|
-
interface ProposeReviewControlResult<State, Summary = unknown> {
|
|
716
|
-
state: State;
|
|
717
|
-
verification: Verification;
|
|
718
|
-
traceSummary?: Summary;
|
|
719
|
-
review: Review | null;
|
|
720
|
-
reviewAvailable: boolean;
|
|
721
|
-
reviewError?: string;
|
|
722
|
-
}
|
|
723
|
-
interface ProposeReviewControlConfig<State, Summary = unknown> {
|
|
724
|
-
goal: string;
|
|
725
|
-
initialState: State;
|
|
726
|
-
propose: ProposeFn<State, Summary>;
|
|
727
|
-
verify: VerifyFn<State>;
|
|
728
|
-
review: ReviewFn<State, Summary>;
|
|
729
|
-
maxShots?: number;
|
|
730
|
-
maxWallMs?: number;
|
|
731
|
-
memory?: ReviewMemoryStore;
|
|
732
|
-
store?: TraceStore;
|
|
733
|
-
scenarioId?: string;
|
|
734
|
-
projectId?: string;
|
|
735
|
-
variantId?: string;
|
|
736
|
-
fallbackInstruction?: string;
|
|
737
|
-
confidenceFloor?: number;
|
|
738
|
-
confidenceFloorWindow?: number;
|
|
739
|
-
failureClassFromVerification?: (verification: Verification) => FailureClass | undefined;
|
|
740
|
-
actionFailure?: ControlRuntimeConfig<ProposeReviewControlState<State, Summary>, ProposeReviewControlAction, ProposeReviewControlResult<State, Summary>>['actionFailure'];
|
|
741
|
-
}
|
|
742
|
-
declare function runProposeReviewAsControlLoop<State, Summary = unknown>(config: ProposeReviewControlConfig<State, Summary>): Promise<ControlRunResult<ProposeReviewControlState<State, Summary>, ProposeReviewControlAction, ProposeReviewControlResult<State, Summary>>>;
|
|
743
|
-
|
|
744
|
-
type AgentProfileCellSchemaVersion = 'agent-profile-cell/v1';
|
|
745
|
-
type AgentProfileDimensionValue = string | number | boolean | null;
|
|
746
|
-
interface AgentProfileSource {
|
|
747
|
-
/** Runtime/profile contract being fingerprinted, e.g. `agent-interface-profile`. */
|
|
748
|
-
kind: string;
|
|
749
|
-
/** sha256 over the canonical source profile object. */
|
|
750
|
-
hash: string;
|
|
751
|
-
}
|
|
752
|
-
interface AgentProfileHarness {
|
|
753
|
-
id: string;
|
|
754
|
-
version?: string;
|
|
755
|
-
hash?: string;
|
|
756
|
-
}
|
|
757
|
-
interface AgentProfileCell {
|
|
758
|
-
schemaVersion: AgentProfileCellSchemaVersion;
|
|
759
|
-
cellId: string;
|
|
760
|
-
profileId: string;
|
|
761
|
-
sourceProfile: AgentProfileSource;
|
|
762
|
-
harness?: AgentProfileHarness;
|
|
763
|
-
model?: string;
|
|
764
|
-
promptHash?: string;
|
|
765
|
-
dimensions?: Record<string, AgentProfileDimensionValue>;
|
|
766
|
-
}
|
|
767
|
-
|
|
768
|
-
/**
|
|
769
|
-
* Paper-grade RunRecord schema + runtime validator.
|
|
770
|
-
*
|
|
771
|
-
* Every run that participates in a promotion gate, paper table, or
|
|
772
|
-
* researcher loop SHOULD be recorded as a `RunRecord`. The mandatory
|
|
773
|
-
* fields are exactly those the paper "Two Loops, Three Roles" requires
|
|
774
|
-
* for reproducibility: who/what/when/cost/seed/hash, plus the search vs
|
|
775
|
-
* holdout split tag. A task score is optional because execution-only records
|
|
776
|
-
* must preserve missing labels instead of converting errors into zero quality.
|
|
777
|
-
*
|
|
778
|
-
* This is intentionally NOT a replacement for the rich `Run` /
|
|
779
|
-
* `ProposeReviewReport` / `ScenarioResult` types already in the
|
|
780
|
-
* package. Those are runtime structures with full provenance. A
|
|
781
|
-
* `RunRecord` is the analysis-time projection — the JSON-friendly
|
|
782
|
-
* row you'd put in a parquet file or paste into a notebook.
|
|
783
|
-
*
|
|
784
|
-
* Validate at the boundary:
|
|
785
|
-
*
|
|
786
|
-
* const rec = validateRunRecord(rawJson) // throws on missing
|
|
787
|
-
* const ok = isRunRecord(rawJson) // boolean check
|
|
788
|
-
* const rec = parseRunRecordSafe(rawJson) // { ok, value | error }
|
|
789
|
-
*
|
|
790
|
-
* The validator runs in pure TS — zod is intentionally NOT a
|
|
791
|
-
* dependency. Round-trip tested in `tests/run-record.test.ts`.
|
|
792
|
-
*/
|
|
793
|
-
|
|
794
|
-
/** Search/dev/holdout split tag. 'search' is the paper-grade alias for the
|
|
795
|
-
* combined train+test pool that the optimizer is allowed to read. */
|
|
796
|
-
type RunSplitTag = 'search' | 'dev' | 'holdout';
|
|
797
|
-
/**
|
|
798
|
-
* Explicit execution-lifecycle result for a run.
|
|
799
|
-
*
|
|
800
|
-
* This is separate from task quality (`outcome`) and failure classification.
|
|
801
|
-
* Producers set it only from root-run or process evidence.
|
|
802
|
-
*/
|
|
803
|
-
type RunTerminalOutcome = 'succeeded' | 'failed' | 'cancelled' | 'incomplete' | 'unknown';
|
|
804
|
-
interface RunTokenUsage {
|
|
805
|
-
input: number;
|
|
806
|
-
/** All generated tokens charged as output, including reasoning tokens. */
|
|
807
|
-
output: number;
|
|
808
|
-
/** Reasoning-token subset of `output`, when the provider reports it. */
|
|
809
|
-
reasoning?: number;
|
|
810
|
-
/** Prompt tokens served from a provider cache. */
|
|
811
|
-
cached?: number;
|
|
812
|
-
/** Prompt tokens written into a provider cache. */
|
|
813
|
-
cacheWrite?: number;
|
|
814
|
-
}
|
|
815
|
-
/**
|
|
816
|
-
* How a run's USD amount was obtained.
|
|
817
|
-
*/
|
|
818
|
-
type RunCostProvenance = {
|
|
819
|
-
kind: 'observed';
|
|
820
|
-
usd: number;
|
|
821
|
-
} | {
|
|
822
|
-
kind: 'estimated';
|
|
823
|
-
usd: number;
|
|
824
|
-
} | {
|
|
825
|
-
kind: 'uncaptured';
|
|
826
|
-
usd: null;
|
|
827
|
-
};
|
|
828
|
-
interface RunJudgeMetadata {
|
|
829
|
-
model: string;
|
|
830
|
-
promptVersion: string;
|
|
831
|
-
/** [0,1] confidence the judge declared. Constant judge confidence
|
|
832
|
-
* across many runs is a fallback signal (see `canary.ts`). */
|
|
833
|
-
confidence: number;
|
|
834
|
-
/** True if the judge degraded to a fallback path (rules-only,
|
|
835
|
-
* prior-call cache, etc.). The canary uses this to alert. */
|
|
836
|
-
fallback: boolean;
|
|
837
|
-
}
|
|
838
|
-
/**
|
|
839
|
-
* Per-judge / per-dimension breakdown for runs scored by an ensemble of
|
|
840
|
-
* judges over a multi-dimensional rubric.
|
|
841
|
-
*
|
|
842
|
-
* The collapsed `outcome.searchScore` / `holdoutScore` carries the
|
|
843
|
-
* composite the gate uses. The full breakdown belongs here so consumers
|
|
844
|
-
* can answer "which judge disagreed?", "which dimension dragged the
|
|
845
|
-
* composite down?", and "did half the panel fail?" without re-running.
|
|
846
|
-
*
|
|
847
|
-
* `perJudge[judgeId][dim]` is the canonical source; `perDimMean` and
|
|
848
|
-
* `composite` are convenience projections — derivable but precomputed so
|
|
849
|
-
* downstream IRR primitives (`interRaterReliability`,
|
|
850
|
-
* `corpusInterRaterAgreement`) and reporters don't pay the same
|
|
851
|
-
* aggregation twice.
|
|
852
|
-
*
|
|
853
|
-
* Fail-loud discipline: judges that errored out land in `failedJudges`
|
|
854
|
-
* by id. A missing key in `perJudge` is ambiguous (silent zero vs not
|
|
855
|
-
* run); the explicit list makes a partial-failure recorded as such.
|
|
856
|
-
*/
|
|
857
|
-
interface JudgeScoresRecord {
|
|
858
|
-
/** Per-judge per-dimension scores. `{ "kimi-k2.6": { helpfulness: 0.8, clarity: 0.7 }, ... }`. */
|
|
859
|
-
perJudge: Record<string, Record<string, number>>;
|
|
860
|
-
/** Per-dim mean across judges. Convenience — derivable from `perJudge`. */
|
|
861
|
-
perDimMean: Record<string, number>;
|
|
862
|
-
/** Composite mean across successful judges. Mirrors the task score only
|
|
863
|
-
* when `failedJudges` is empty. */
|
|
864
|
-
composite: number;
|
|
865
|
-
/** Judges that errored or returned an unparseable verdict. Recorded
|
|
866
|
-
* by id (e.g. `['glm-5.1']`) so a partial-failure case is explicit,
|
|
867
|
-
* not inferred from missing keys in `perJudge`. */
|
|
868
|
-
failedJudges?: string[];
|
|
869
|
-
/** Free-form notes the judges emitted (joined across judges or
|
|
870
|
-
* first-judge only — consumer's choice). */
|
|
871
|
-
notes?: string;
|
|
872
|
-
}
|
|
873
|
-
interface RunOutcome {
|
|
874
|
-
/** Score on the search/optimization split. Optional for holdout-only and
|
|
875
|
-
* execution-only records. */
|
|
876
|
-
searchScore?: number;
|
|
877
|
-
/** Score on the held-out split. Optional for search-only and execution-only
|
|
878
|
-
* records. When both scores are absent, the run is explicitly unlabeled. */
|
|
879
|
-
holdoutScore?: number;
|
|
880
|
-
/** Bag of any other metric the run produced — judge dimensions,
|
|
881
|
-
* pass/fail counters, latency stats, etc. Numeric only — keeps
|
|
882
|
-
* reporters honest. */
|
|
883
|
-
raw: Record<string, number>;
|
|
884
|
-
/** Per-judge / per-dim breakdown. Consumers writing ensemble
|
|
885
|
-
* judgements populate this; substrate primitives like
|
|
886
|
-
* `interRaterReliability` and `corpusInterRaterAgreement` accept
|
|
887
|
-
* these records as input. Optional — single-judge or scalar-only
|
|
888
|
-
* runs leave it unset. */
|
|
889
|
-
judgeScores?: JudgeScoresRecord;
|
|
890
|
-
/** Authenticity / realness verdict — did the run build the REAL thing on the
|
|
891
|
-
* intended infra, or fake it (see `./authenticity`)? Optional: only domains
|
|
892
|
-
* with an authenticity config populate it. Carried in the corpus so the
|
|
893
|
-
* flywheel / off-policy learning can optimize for real completion, not gamed
|
|
894
|
-
* pass-rate. `score` is 0-1; `gated` is the anti-Goodhart flag — a gated run
|
|
895
|
-
* must not count as a real success regardless of `score`. */
|
|
896
|
-
realness?: {
|
|
897
|
-
score: number;
|
|
898
|
-
gated: boolean;
|
|
899
|
-
reason?: string;
|
|
900
|
-
};
|
|
901
|
-
}
|
|
902
|
-
/**
|
|
903
|
-
* Mandatory paper-grade fields for a single evaluation run. Optional
|
|
904
|
-
* fields are extension points; mandatory fields throw if missing.
|
|
905
|
-
*
|
|
906
|
-
* Hash discipline:
|
|
907
|
-
* - `promptHash` is the sha256 of the EFFECTIVE prompt sent to the
|
|
908
|
-
* model (after any steering bundle merge).
|
|
909
|
-
* - `configHash` is the sha256 of the effective run config (model,
|
|
910
|
-
* temperature, tools, judges, splits). The pair (promptHash,
|
|
911
|
-
* configHash) uniquely identifies an experiment cell.
|
|
912
|
-
*
|
|
913
|
-
* Model snapshot discipline:
|
|
914
|
-
* - `model` MUST encode a snapshot version. Bare aliases like
|
|
915
|
-
* `claude-sonnet-4` or `gpt-4o` are banned — they remap silently.
|
|
916
|
-
* Use `claude-sonnet-4-6@2025-04-15` or `gpt-4o-2024-11-20`.
|
|
917
|
-
*/
|
|
918
|
-
interface RunRecord {
|
|
919
|
-
/** UUID for the run. */
|
|
920
|
-
runId: string;
|
|
921
|
-
/** Logical experiment grouping (a treatment vs a baseline within
|
|
922
|
-
* the same sweep should share `experimentId`). */
|
|
923
|
-
experimentId: string;
|
|
924
|
-
/** Stable identifier for the candidate (variant) being run. The
|
|
925
|
-
* promotion gate compares two `candidateId`s on matched items. */
|
|
926
|
-
candidateId: string;
|
|
927
|
-
/** RNG seed for the run. Always recorded — silent re-seeding is
|
|
928
|
-
* the most common cause of non-reproducible numbers. */
|
|
929
|
-
seed: number;
|
|
930
|
-
/** Model identifier WITH snapshot version. */
|
|
931
|
-
model: string;
|
|
932
|
-
/** sha256 of the effective prompt (post-steering). */
|
|
933
|
-
promptHash: string;
|
|
934
|
-
/** sha256 of the effective config. */
|
|
935
|
-
configHash: string;
|
|
936
|
-
/** Git SHA the harness was run from. */
|
|
937
|
-
commitSha: string;
|
|
938
|
-
/** End-to-end wall-clock duration in milliseconds. */
|
|
939
|
-
wallMs: number;
|
|
940
|
-
/** Time spent queued before execution started, if known. */
|
|
941
|
-
queueMs?: number;
|
|
942
|
-
/** Total USD cost, or null when the producer could not capture one. */
|
|
943
|
-
costUsd: number | null;
|
|
944
|
-
/** Whether `costUsd` came from billing data, a price calculation, or is unavailable. */
|
|
945
|
-
costProvenance: RunCostProvenance;
|
|
946
|
-
/** Token usage breakdown. */
|
|
947
|
-
tokenUsage: RunTokenUsage;
|
|
948
|
-
/** Root-run or process terminal result. Never inferred from a child span. */
|
|
949
|
-
terminalOutcome: RunTerminalOutcome;
|
|
950
|
-
/** Root-run or process failure reason. Valid only for a failed, cancelled,
|
|
951
|
-
* or incomplete terminal result; never populated from a child span. */
|
|
952
|
-
terminalFailureReason?: string;
|
|
953
|
-
/** Judge-side metadata, if a judge was used. */
|
|
954
|
-
judgeMetadata?: RunJudgeMetadata;
|
|
955
|
-
/** Per-split scores + raw bag. */
|
|
956
|
-
outcome: RunOutcome;
|
|
957
|
-
/** Canonical task-failure class drawn from the shared
|
|
958
|
-
* `FAILURE_CLASSES` taxonomy. Producers set it only from task-result
|
|
959
|
-
* evidence. Execution errors belong in
|
|
960
|
-
* `outcome.raw.execution_error_count`. */
|
|
961
|
-
failureClass?: FailureClass;
|
|
962
|
-
/** Free-form task-failure detail scoped under a non-success
|
|
963
|
-
* `failureClass`. It is invalid without that class. */
|
|
964
|
-
failureMode?: string;
|
|
965
|
-
/** Which split this run was drawn from. */
|
|
966
|
-
splitTag: RunSplitTag;
|
|
967
|
-
/**
|
|
968
|
-
* Stable scenario identifier the run observed or was scored against.
|
|
969
|
-
* Comparison primitives match this identity rather than input order.
|
|
970
|
-
*/
|
|
971
|
-
scenarioId: string;
|
|
972
|
-
/**
|
|
973
|
-
* Canonical identity for the agent profile cell that produced this row:
|
|
974
|
-
* profile artifact hash plus optional harness/model/prompt/reporting
|
|
975
|
-
* dimensions. Use `agentProfile.cellId` to group persona sweeps and
|
|
976
|
-
* longitudinal reports by the complete source profile, not by a loose
|
|
977
|
-
* candidate label or opaque config hash.
|
|
978
|
-
*/
|
|
979
|
-
agentProfile?: AgentProfileCell;
|
|
980
|
-
}
|
|
981
|
-
/**
|
|
982
|
-
* Canonical task-result classification.
|
|
983
|
-
*
|
|
984
|
-
* A producer may omit classification, record explicit success, or attach
|
|
985
|
-
* domain-specific detail to a non-success class. Detail can never stand alone.
|
|
986
|
-
* Execution errors belong in `outcome.raw.execution_error_count`.
|
|
987
|
-
*/
|
|
988
|
-
type RunTaskFailure = {
|
|
989
|
-
failureClass?: undefined;
|
|
990
|
-
failureMode?: undefined;
|
|
991
|
-
} | {
|
|
992
|
-
failureClass: 'success';
|
|
993
|
-
failureMode?: undefined;
|
|
994
|
-
} | {
|
|
995
|
-
failureClass: Exclude<FailureClass, 'success'>;
|
|
996
|
-
failureMode?: string;
|
|
997
|
-
};
|
|
998
|
-
|
|
999
|
-
interface RunEvidenceMetadata {
|
|
1000
|
-
experimentId: string;
|
|
1001
|
-
scenarioId: string;
|
|
1002
|
-
candidateId: string;
|
|
1003
|
-
seed: number;
|
|
1004
|
-
model: string;
|
|
1005
|
-
promptHash: string;
|
|
1006
|
-
configHash: string;
|
|
1007
|
-
commitSha: string;
|
|
1008
|
-
splitTag: RunSplitTag;
|
|
1009
|
-
tokenUsage: RunTokenUsage;
|
|
1010
|
-
costProvenance: RunRecord['costProvenance'];
|
|
1011
|
-
queueMs?: number;
|
|
1012
|
-
judgeMetadata?: RunRecord['judgeMetadata'];
|
|
1013
|
-
raw?: Record<string, number>;
|
|
1014
|
-
}
|
|
1015
|
-
type ControlRunToRunRecordOptions = RunEvidenceMetadata & RunTaskFailure & {
|
|
1016
|
-
runId?: string;
|
|
1017
|
-
score?: number;
|
|
1018
|
-
};
|
|
1019
|
-
/**
|
|
1020
|
-
* Project a completed control-loop run into the strict RunRecord shape used by
|
|
1021
|
-
* release gates, optimizer tables, and research reports.
|
|
1022
|
-
*
|
|
1023
|
-
* The control loop owns live execution evidence. The caller still supplies the
|
|
1024
|
-
* experiment-cell metadata because prompt/config hashes, split assignment,
|
|
1025
|
-
* model snapshot, and commit SHA are product/harness concerns.
|
|
1026
|
-
*/
|
|
1027
|
-
declare function controlRunToRunRecord<TState, TAction, TActionResult, TEval extends ControlEvalResult = ControlEvalResult>(run: ControlRunResult<TState, TAction, TActionResult, TEval>, options: ControlRunToRunRecordOptions): RunRecord;
|
|
1028
|
-
declare function scoreFromEvals(evals: readonly ControlEvalResult[]): number | undefined;
|
|
1029
|
-
|
|
1030
|
-
export { type ActionExecutionPolicy, type ActionPolicyDecision, type ControlActionFailureMode, type ControlActionOutcome, type ControlBudget, type ControlContext, type ControlDecision, type ControlEvalResult, type ControlRunResult, type ControlRunToRunRecordOptions, type ControlRuntimeConfig, type ControlRuntimeError, type ControlSeverity, type ControlStep, type ControlStopPolicies, type ProposeReviewConfig, type ProposeReviewControlAction, type ProposeReviewControlConfig, type ProposeReviewControlResult, type ProposeReviewControlState, type ProposeReviewReport, type RunEvidenceMetadata, type StopDecision, allCriticalPassed, controlRunToRunRecord, evaluateActionPolicy, objectiveEval, runAgentControlLoop, runProposeReview, runProposeReviewAsControlLoop, scoreFromEvals, stopOnNoProgress, stopOnRepeatedAction, subjectiveEval };
|
|
1
|
+
import { B as ControlRunResult, F as ControlActionOutcome, G as ControlStopPolicies, H as ControlRuntimeError, I as ControlBudget, J as objectiveEval, K as StopDecision, L as ControlContext, P as ControlActionFailureMode, Q as subjectiveEval, R as ControlDecision, U as ControlSeverity, V as ControlRuntimeConfig, W as ControlStep, X as stopOnNoProgress, Y as runAgentControlLoop, Z as stopOnRepeatedAction, q as allCriticalPassed, z as ControlEvalResult } from "./feedback-trajectory-CVaeREXV.js";
|
|
2
|
+
import { A as ActionExecutionPolicy, M as evaluateActionPolicy, _ as ProposeReviewReport, a as ProposeReviewControlAction, c as ProposeReviewControlState, g as ProposeReviewConfig, i as scoreFromEvals, j as ActionPolicyDecision, k as runProposeReview, n as RunEvidenceMetadata, o as ProposeReviewControlConfig, r as controlRunToRunRecord, s as ProposeReviewControlResult, t as ControlRunToRunRecordOptions, u as runProposeReviewAsControlLoop } from "./run-evidence-ClRX_8A9.js";
|
|
3
|
+
export { type ActionExecutionPolicy, type ActionPolicyDecision, type ControlActionFailureMode, type ControlActionOutcome, type ControlBudget, type ControlContext, type ControlDecision, type ControlEvalResult, type ControlRunResult, type ControlRunToRunRecordOptions, type ControlRuntimeConfig, type ControlRuntimeError, type ControlSeverity, type ControlStep, type ControlStopPolicies, type ProposeReviewConfig, type ProposeReviewControlAction, type ProposeReviewControlConfig, type ProposeReviewControlResult, type ProposeReviewControlState, type ProposeReviewReport, type RunEvidenceMetadata, type StopDecision, allCriticalPassed, controlRunToRunRecord, evaluateActionPolicy, objectiveEval, runAgentControlLoop, runProposeReview, runProposeReviewAsControlLoop, scoreFromEvals, stopOnNoProgress, stopOnRepeatedAction, subjectiveEval };
|