@tangle-network/agent-eval 0.129.0 → 0.130.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +21 -0
- package/README.md +2 -0
- package/dist/active-curriculum-C4mk67HP.js +214 -0
- package/dist/active-curriculum-C4mk67HP.js.map +1 -0
- package/dist/adversarial-smnADNFS.d.ts +21 -0
- package/dist/adversarial-smnADNFS.d.ts.map +1 -0
- package/dist/analyst/index.d.ts +81 -2872
- package/dist/analyst/index.d.ts.map +1 -0
- package/dist/analyst/index.js +319 -360
- package/dist/analyst/index.js.map +1 -1
- package/dist/analyst-BkTS3C58.d.ts +89 -0
- package/dist/analyst-BkTS3C58.d.ts.map +1 -0
- package/dist/analyst-LsnNpSkm.js +152 -0
- package/dist/analyst-LsnNpSkm.js.map +1 -0
- package/dist/analyze-runs-C1CavBMk.js +1067 -0
- package/dist/analyze-runs-C1CavBMk.js.map +1 -0
- package/dist/analyze-runs-FsgYCinh.d.ts +72 -0
- package/dist/analyze-runs-FsgYCinh.d.ts.map +1 -0
- package/dist/attribute-vocabulary-DLJ6303h.d.ts +54 -0
- package/dist/attribute-vocabulary-DLJ6303h.d.ts.map +1 -0
- package/dist/authenticity/index.d.ts +81 -79
- package/dist/authenticity/index.d.ts.map +1 -0
- package/dist/authenticity/index.js +209 -193
- package/dist/authenticity/index.js.map +1 -1
- package/dist/baseline-HsBvw_dk.js +550 -0
- package/dist/baseline-HsBvw_dk.js.map +1 -0
- package/dist/baseline-hG3K85h4.d.ts +125 -0
- package/dist/baseline-hG3K85h4.d.ts.map +1 -0
- package/dist/belief-state/index.d.ts +448 -1188
- package/dist/belief-state/index.d.ts.map +1 -0
- package/dist/belief-state/index.js +1617 -1709
- package/dist/belief-state/index.js.map +1 -1
- package/dist/benchmarks/index.d.ts +2 -891
- package/dist/benchmarks/index.js +2 -60
- package/dist/benchmarks-BJgDGkAD.js +754 -0
- package/dist/benchmarks-BJgDGkAD.js.map +1 -0
- package/dist/builder-eval/index.d.ts +150 -662
- package/dist/builder-eval/index.d.ts.map +1 -0
- package/dist/builder-eval/index.js +356 -345
- package/dist/builder-eval/index.js.map +1 -1
- package/dist/calibration-CNWWA6K8.js +94 -0
- package/dist/calibration-CNWWA6K8.js.map +1 -0
- package/dist/campaign/index.d.ts +5 -6381
- package/dist/campaign/index.js +3 -213
- package/dist/campaign-aKJt6emI.js +3886 -0
- package/dist/campaign-aKJt6emI.js.map +1 -0
- package/dist/cli.d.ts +1 -1
- package/dist/cli.js +164 -175
- package/dist/cli.js.map +1 -1
- package/dist/client-C97NMzqi.d.ts +581 -0
- package/dist/client-C97NMzqi.d.ts.map +1 -0
- package/dist/client-CYzbdJOZ.js +637 -0
- package/dist/client-CYzbdJOZ.js.map +1 -0
- package/dist/code-agent-session-BjkMTQ7H.js +1390 -0
- package/dist/code-agent-session-BjkMTQ7H.js.map +1 -0
- package/dist/code-agent-session-DqqgOJaz.d.ts +143 -0
- package/dist/code-agent-session-DqqgOJaz.d.ts.map +1 -0
- package/dist/concurrency-DIxRZF_J.js +85 -0
- package/dist/concurrency-DIxRZF_J.js.map +1 -0
- package/dist/contract/index.d.ts +645 -5565
- package/dist/contract/index.d.ts.map +1 -0
- package/dist/contract/index.js +1730 -1938
- package/dist/contract/index.js.map +1 -1
- package/dist/control.d.ts +3 -1030
- package/dist/control.js +2 -33
- package/dist/cost-ledger-DIgQUFZZ.js +801 -0
- package/dist/cost-ledger-DIgQUFZZ.js.map +1 -0
- package/dist/cost-ledger-Dye6jCgg.d.ts +236 -0
- package/dist/cost-ledger-Dye6jCgg.d.ts.map +1 -0
- package/dist/dataset-BvtnC8Dc.d.ts +115 -0
- package/dist/dataset-BvtnC8Dc.d.ts.map +1 -0
- package/dist/default-registry-C-vFCSEc.js +2579 -0
- package/dist/default-registry-C-vFCSEc.js.map +1 -0
- package/dist/default-registry-CNPo-Vsb.d.ts +540 -0
- package/dist/default-registry-CNPo-Vsb.d.ts.map +1 -0
- package/dist/emitter-CPBAhxum.js +266 -0
- package/dist/emitter-CPBAhxum.js.map +1 -0
- package/dist/emitter-DGQGoLyj.d.ts +113 -0
- package/dist/emitter-DGQGoLyj.d.ts.map +1 -0
- package/dist/errors-8YnH8WlF.js +64 -0
- package/dist/errors-8YnH8WlF.js.map +1 -0
- package/dist/errors-CEk209JS.d.ts +76 -0
- package/dist/errors-CEk209JS.d.ts.map +1 -0
- package/dist/eval-campaign-DEm6c8ru.js +349 -0
- package/dist/eval-campaign-DEm6c8ru.js.map +1 -0
- package/dist/execution-tracks-CpgFPpS5.js +93 -0
- package/dist/execution-tracks-CpgFPpS5.js.map +1 -0
- package/dist/exporters-q9iL-2Jf.js +148 -0
- package/dist/exporters-q9iL-2Jf.js.map +1 -0
- package/dist/extract-usage-BrQ8mCLX.js +155 -0
- package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
- package/dist/failure-cluster-CqcvCcdR.d.ts +59 -0
- package/dist/failure-cluster-CqcvCcdR.d.ts.map +1 -0
- package/dist/feedback-trajectory-CVaeREXV.d.ts +340 -0
- package/dist/feedback-trajectory-CVaeREXV.d.ts.map +1 -0
- package/dist/fuzz.d.ts +320 -646
- package/dist/fuzz.d.ts.map +1 -0
- package/dist/fuzz.js +670 -618
- package/dist/fuzz.js.map +1 -1
- package/dist/hf-dataset-DBJXXoY1.js +763 -0
- package/dist/hf-dataset-DBJXXoY1.js.map +1 -0
- package/dist/hosted/index.d.ts +11 -831
- package/dist/hosted/index.d.ts.map +1 -0
- package/dist/hosted/index.js +2 -37
- package/dist/index-2JJSA6-r2.d.ts +926 -0
- package/dist/index-2JJSA6-r2.d.ts.map +1 -0
- package/dist/index-6N0aYmpW.d.ts +217 -0
- package/dist/index-6N0aYmpW.d.ts.map +1 -0
- package/dist/index-BAvgST_9.d.ts +131 -0
- package/dist/index-BAvgST_9.d.ts.map +1 -0
- package/dist/index-BvnJuTGD.d.ts +68 -0
- package/dist/index-BvnJuTGD.d.ts.map +1 -0
- package/dist/index-C61Wi7yg.d.ts +547 -0
- package/dist/index-C61Wi7yg.d.ts.map +1 -0
- package/dist/index-CD_WZ_Xr.d.ts +2250 -0
- package/dist/index-CD_WZ_Xr.d.ts.map +1 -0
- package/dist/index-DSC51roc.d.ts +102 -0
- package/dist/index-DSC51roc.d.ts.map +1 -0
- package/dist/index-Em67JBjs.d.ts +335 -0
- package/dist/index-Em67JBjs.d.ts.map +1 -0
- package/dist/index.d.ts +3755 -15555
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +11182 -11216
- package/dist/index.js.map +1 -1
- package/dist/integrity-BzRbCHzi.js +424 -0
- package/dist/integrity-BzRbCHzi.js.map +1 -0
- package/dist/integrity-rmVhXWA7.d.ts +61 -0
- package/dist/integrity-rmVhXWA7.d.ts.map +1 -0
- package/dist/judge-calibration-DFtEMlde.d.ts +146 -0
- package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
- package/dist/ledger-core/index.d.ts +2 -0
- package/dist/ledger-core/index.js +2 -0
- package/dist/ledger-core-DtZz1RG0.js +388 -0
- package/dist/ledger-core-DtZz1RG0.js.map +1 -0
- package/dist/llm-client--GR4JbZE.js +687 -0
- package/dist/llm-client--GR4JbZE.js.map +1 -0
- package/dist/llm-client-B_nIBlYo.d.ts +290 -0
- package/dist/llm-client-B_nIBlYo.d.ts.map +1 -0
- package/dist/matrix/index.d.ts +3 -155
- package/dist/matrix/index.js +2 -8
- package/dist/matrix-BzQnu2S6.js +270 -0
- package/dist/matrix-BzQnu2S6.js.map +1 -0
- package/dist/meta-eval/index.d.ts +4 -1027
- package/dist/meta-eval/index.js +393 -390
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/metrics-C9YY1OcL.js +239 -0
- package/dist/metrics-C9YY1OcL.js.map +1 -0
- package/dist/mint-yN2M2eh0.js +201 -0
- package/dist/mint-yN2M2eh0.js.map +1 -0
- package/dist/multi-layer-verifier-BHY1gWAc.d.ts +138 -0
- package/dist/multi-layer-verifier-BHY1gWAc.d.ts.map +1 -0
- package/dist/multishot/index.d.ts +273 -480
- package/dist/multishot/index.d.ts.map +1 -0
- package/dist/multishot/index.js +595 -548
- package/dist/multishot/index.js.map +1 -1
- package/dist/off-policy-DvgzvtIx.js +220 -0
- package/dist/off-policy-DvgzvtIx.js.map +1 -0
- package/dist/off-policy-mskQw8Mb.d.ts +153 -0
- package/dist/off-policy-mskQw8Mb.d.ts.map +1 -0
- package/dist/openapi.json +1 -1
- package/dist/opencode-sqlite-BGrHeDu3.js +318 -0
- package/dist/opencode-sqlite-BGrHeDu3.js.map +1 -0
- package/dist/outcome-store-BYHIuO0e.d.ts +65 -0
- package/dist/outcome-store-BYHIuO0e.d.ts.map +1 -0
- package/dist/outcome-store-ChBKlTd_.js +75 -0
- package/dist/outcome-store-ChBKlTd_.js.map +1 -0
- package/dist/paired-arms-D9D0wXj2.js +260 -0
- package/dist/paired-arms-D9D0wXj2.js.map +1 -0
- package/dist/pipelines/index.d.ts +95 -532
- package/dist/pipelines/index.d.ts.map +1 -0
- package/dist/pipelines/index.js +497 -478
- package/dist/pipelines/index.js.map +1 -1
- package/dist/pre-registration-DakwTRXk.js +96 -0
- package/dist/pre-registration-DakwTRXk.js.map +1 -0
- package/dist/propose-review-control-LhIWGzHl.js +1458 -0
- package/dist/propose-review-control-LhIWGzHl.js.map +1 -0
- package/dist/query-CJ_DX8vl.d.ts +28 -0
- package/dist/query-CJ_DX8vl.d.ts.map +1 -0
- package/dist/query-Di7eEQ79.js +83 -0
- package/dist/query-Di7eEQ79.js.map +1 -0
- package/dist/raw-provider-sink-BQd7mzyT.js +201 -0
- package/dist/raw-provider-sink-BQd7mzyT.js.map +1 -0
- package/dist/raw-provider-sink-BU29Sh8h.d.ts +134 -0
- package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +1 -0
- package/dist/redact-7Aq1ukl-.js +107 -0
- package/dist/redact-7Aq1ukl-.js.map +1 -0
- package/dist/release-report-Cz9NKH39.js +603 -0
- package/dist/release-report-Cz9NKH39.js.map +1 -0
- package/dist/release-report-mvB2G4_J.d.ts +244 -0
- package/dist/release-report-mvB2G4_J.d.ts.map +1 -0
- package/dist/replay-GnyotH0J.js +1741 -0
- package/dist/replay-GnyotH0J.js.map +1 -0
- package/dist/replay-OoidtG1E.d.ts +749 -0
- package/dist/replay-OoidtG1E.d.ts.map +1 -0
- package/dist/reporting.d.ts +6 -1312
- package/dist/reporting.js +6 -51
- package/dist/researcher-CwTdwXG1.d.ts +314 -0
- package/dist/researcher-CwTdwXG1.d.ts.map +1 -0
- package/dist/reward-hacking-eAnOsynk.d.ts +224 -0
- package/dist/reward-hacking-eAnOsynk.d.ts.map +1 -0
- package/dist/reward-hacking-qipEpKvY.js +596 -0
- package/dist/reward-hacking-qipEpKvY.js.map +1 -0
- package/dist/reward-nw2xZGZG.js +137 -0
- package/dist/reward-nw2xZGZG.js.map +1 -0
- package/dist/rl.d.ts +760 -4010
- package/dist/rl.d.ts.map +1 -0
- package/dist/rl.js +2325 -1958
- package/dist/rl.js.map +1 -1
- package/dist/rolldown-runtime-8H4AJuhK.js +14 -0
- package/dist/rollout/index.d.ts +3 -2087
- package/dist/rollout/index.js +8 -168
- package/dist/rollout-BOYjemfR.js +624 -0
- package/dist/rollout-BOYjemfR.js.map +1 -0
- package/dist/rubric-predictive-validity-B3xmbmS1.js +141 -0
- package/dist/rubric-predictive-validity-B3xmbmS1.js.map +1 -0
- package/dist/rubric-predictive-validity-Ku_clp_1.d.ts +74 -0
- package/dist/rubric-predictive-validity-Ku_clp_1.d.ts.map +1 -0
- package/dist/run-evidence-ClRX_8A9.d.ts +225 -0
- package/dist/run-evidence-ClRX_8A9.d.ts.map +1 -0
- package/dist/run-record-BuoE80Dq.js +467 -0
- package/dist/{chunk-56TAVBOK.js.map → run-record-BuoE80Dq.js.map} +1 -1
- package/dist/run-record-CnZu_gjl.d.ts +357 -0
- package/dist/run-record-CnZu_gjl.d.ts.map +1 -0
- package/dist/run-score-iEEAWiBY.js +41 -0
- package/dist/run-score-iEEAWiBY.js.map +1 -0
- package/dist/runtime-trajectory-1gyaTOoC.js +93 -0
- package/dist/runtime-trajectory-1gyaTOoC.js.map +1 -0
- package/dist/runtime-trajectory-BvSZcCHD.d.ts +50 -0
- package/dist/runtime-trajectory-BvSZcCHD.d.ts.map +1 -0
- package/dist/schema-BtVldJ3T.d.ts +206 -0
- package/dist/schema-BtVldJ3T.d.ts.map +1 -0
- package/dist/schema-C6DW4ZHR.js +821 -0
- package/dist/schema-C6DW4ZHR.js.map +1 -0
- package/dist/schema-CRhEY1SO.js +69 -0
- package/dist/schema-CRhEY1SO.js.map +1 -0
- package/dist/schema-Cef2cFmb.d.ts +408 -0
- package/dist/schema-Cef2cFmb.d.ts.map +1 -0
- package/dist/semantic-concept-judge-B6cWNJ2K.js +725 -0
- package/dist/semantic-concept-judge-B6cWNJ2K.js.map +1 -0
- package/dist/sequential-Br0mAPHA.js +148 -0
- package/dist/sequential-Br0mAPHA.js.map +1 -0
- package/dist/sequential-CYwq6Ff_.d.ts +141 -0
- package/dist/sequential-CYwq6Ff_.d.ts.map +1 -0
- package/dist/series-convergence-CjO2QdRW.js +43 -0
- package/dist/series-convergence-CjO2QdRW.js.map +1 -0
- package/dist/series-convergence-ofsqPWhs.d.ts +35 -0
- package/dist/series-convergence-ofsqPWhs.d.ts.map +1 -0
- package/dist/server-m5D9cvnG.js +1040 -0
- package/dist/server-m5D9cvnG.js.map +1 -0
- package/dist/skill-usage-C_pXm7lP.d.ts +534 -0
- package/dist/skill-usage-C_pXm7lP.d.ts.map +1 -0
- package/dist/skillopt-optimization-method-CUmHkGbI.js +7718 -0
- package/dist/skillopt-optimization-method-CUmHkGbI.js.map +1 -0
- package/dist/skillopt-optimization-method-CWKVTnks.d.ts +1740 -0
- package/dist/skillopt-optimization-method-CWKVTnks.d.ts.map +1 -0
- package/dist/statistics-Cmj6nynr.d.ts +514 -0
- package/dist/statistics-Cmj6nynr.d.ts.map +1 -0
- package/dist/statistics-CnnxdpOg.js +1437 -0
- package/dist/statistics-CnnxdpOg.js.map +1 -0
- package/dist/store-CT9YIIve.d.ts +117 -0
- package/dist/store-CT9YIIve.d.ts.map +1 -0
- package/dist/store-CxJry_cs.d.ts +229 -0
- package/dist/store-CxJry_cs.d.ts.map +1 -0
- package/dist/storyboard/index.d.ts +84 -203
- package/dist/storyboard/index.d.ts.map +1 -0
- package/dist/storyboard/index.js +609 -542
- package/dist/storyboard/index.js.map +1 -1
- package/dist/summary-report-BNs5nmXI.js +862 -0
- package/dist/summary-report-BNs5nmXI.js.map +1 -0
- package/dist/summary-report-Cj9gdw4i.d.ts +407 -0
- package/dist/summary-report-Cj9gdw4i.d.ts.map +1 -0
- package/dist/supervisor-run/index.d.ts +2 -959
- package/dist/supervisor-run/index.js +2 -65
- package/dist/supervisor-run-_lnTLM3z.js +1679 -0
- package/dist/supervisor-run-_lnTLM3z.js.map +1 -0
- package/dist/task-failure-attributes-CQZlB3et.js +311 -0
- package/dist/task-failure-attributes-CQZlB3et.js.map +1 -0
- package/dist/test-graded-scenario-BsqWLmPt.js +318 -0
- package/dist/test-graded-scenario-BsqWLmPt.js.map +1 -0
- package/dist/test-graded-scenario-D1TaI2va.d.ts +141 -0
- package/dist/test-graded-scenario-D1TaI2va.d.ts.map +1 -0
- package/dist/tools-BmuN627J.js +1085 -0
- package/dist/tools-BmuN627J.js.map +1 -0
- package/dist/trace-attributes.d.ts +2 -52
- package/dist/trace-attributes.js +131 -61
- package/dist/trace-attributes.js.map +1 -1
- package/dist/traces.d.ts +12 -2365
- package/dist/traces.js +12 -252
- package/dist/trajectory-D_7rLrvE.js +56 -0
- package/dist/trajectory-D_7rLrvE.js.map +1 -0
- package/dist/types-DGsxbAEd.d.ts +387 -0
- package/dist/types-DGsxbAEd.d.ts.map +1 -0
- package/dist/types-k9tZGKUg.d.ts +640 -0
- package/dist/types-k9tZGKUg.d.ts.map +1 -0
- package/dist/verdict-Dps8_okt.d.ts +37 -0
- package/dist/verdict-Dps8_okt.d.ts.map +1 -0
- package/dist/wire/index.d.ts +702 -1173
- package/dist/wire/index.d.ts.map +1 -0
- package/dist/wire/index.js +2 -81
- package/docs/campaign-proposers.md +1 -0
- package/package.json +17 -9
- package/dist/benchmarks/index.js.map +0 -1
- package/dist/campaign/index.js.map +0 -1
- package/dist/chunk-2QU3YOPR.js +0 -7374
- package/dist/chunk-2QU3YOPR.js.map +0 -1
- package/dist/chunk-3OCR4R5I.js +0 -728
- package/dist/chunk-3OCR4R5I.js.map +0 -1
- package/dist/chunk-3RF76KTD.js +0 -84
- package/dist/chunk-3RF76KTD.js.map +0 -1
- package/dist/chunk-56TAVBOK.js +0 -698
- package/dist/chunk-5DTSBUL2.js +0 -159
- package/dist/chunk-5DTSBUL2.js.map +0 -1
- package/dist/chunk-7FO3TNPI.js +0 -232
- package/dist/chunk-7FO3TNPI.js.map +0 -1
- package/dist/chunk-7ZZMD7UK.js +0 -386
- package/dist/chunk-7ZZMD7UK.js.map +0 -1
- package/dist/chunk-BOD4O7OF.js +0 -40
- package/dist/chunk-BOD4O7OF.js.map +0 -1
- package/dist/chunk-BSO5JDQH.js +0 -2335
- package/dist/chunk-BSO5JDQH.js.map +0 -1
- package/dist/chunk-C6LXANRU.js +0 -1550
- package/dist/chunk-C6LXANRU.js.map +0 -1
- package/dist/chunk-DODXQREJ.js +0 -752
- package/dist/chunk-DODXQREJ.js.map +0 -1
- package/dist/chunk-DRYIUNWY.js +0 -622
- package/dist/chunk-DRYIUNWY.js.map +0 -1
- package/dist/chunk-E7QXT7SX.js +0 -183
- package/dist/chunk-E7QXT7SX.js.map +0 -1
- package/dist/chunk-EG66UGL4.js +0 -341
- package/dist/chunk-EG66UGL4.js.map +0 -1
- package/dist/chunk-FXTVJPYD.js +0 -576
- package/dist/chunk-FXTVJPYD.js.map +0 -1
- package/dist/chunk-G7MGMCZD.js +0 -153
- package/dist/chunk-G7MGMCZD.js.map +0 -1
- package/dist/chunk-GGE4NNQT.js +0 -65
- package/dist/chunk-GGE4NNQT.js.map +0 -1
- package/dist/chunk-H23X7XKK.js +0 -181
- package/dist/chunk-H23X7XKK.js.map +0 -1
- package/dist/chunk-HHWE3POT.js +0 -94
- package/dist/chunk-HHWE3POT.js.map +0 -1
- package/dist/chunk-HPWUNB47.js +0 -289
- package/dist/chunk-HPWUNB47.js.map +0 -1
- package/dist/chunk-IYCLP2N2.js +0 -766
- package/dist/chunk-IYCLP2N2.js.map +0 -1
- package/dist/chunk-JHCHEVET.js +0 -274
- package/dist/chunk-JHCHEVET.js.map +0 -1
- package/dist/chunk-JQSF5DQT.js +0 -701
- package/dist/chunk-JQSF5DQT.js.map +0 -1
- package/dist/chunk-K4DBDHLK.js +0 -158
- package/dist/chunk-K4DBDHLK.js.map +0 -1
- package/dist/chunk-K6N6XJJX.js +0 -306
- package/dist/chunk-K6N6XJJX.js.map +0 -1
- package/dist/chunk-M4YBQKIJ.js +0 -1040
- package/dist/chunk-M4YBQKIJ.js.map +0 -1
- package/dist/chunk-MA6HLL3S.js +0 -65
- package/dist/chunk-MA6HLL3S.js.map +0 -1
- package/dist/chunk-MAZ26DC7.js +0 -99
- package/dist/chunk-MAZ26DC7.js.map +0 -1
- package/dist/chunk-NPCTHQIO.js +0 -91
- package/dist/chunk-NPCTHQIO.js.map +0 -1
- package/dist/chunk-NY44NC4A.js +0 -1056
- package/dist/chunk-NY44NC4A.js.map +0 -1
- package/dist/chunk-OIUOT4QD.js +0 -44
- package/dist/chunk-OIUOT4QD.js.map +0 -1
- package/dist/chunk-ONWEPEDO.js +0 -57
- package/dist/chunk-ONWEPEDO.js.map +0 -1
- package/dist/chunk-OWN5NPMC.js +0 -152
- package/dist/chunk-OWN5NPMC.js.map +0 -1
- package/dist/chunk-P6FYH6K4.js +0 -1161
- package/dist/chunk-P6FYH6K4.js.map +0 -1
- package/dist/chunk-PC4UYEBM.js +0 -166
- package/dist/chunk-PC4UYEBM.js.map +0 -1
- package/dist/chunk-PC5DOSM7.js +0 -579
- package/dist/chunk-PC5DOSM7.js.map +0 -1
- package/dist/chunk-PXE2VKMX.js +0 -140
- package/dist/chunk-PXE2VKMX.js.map +0 -1
- package/dist/chunk-PZ5AY32C.js +0 -10
- package/dist/chunk-PZ5AY32C.js.map +0 -1
- package/dist/chunk-QB6BDBP2.js +0 -4464
- package/dist/chunk-QB6BDBP2.js.map +0 -1
- package/dist/chunk-RXHCETDZ.js +0 -536
- package/dist/chunk-RXHCETDZ.js.map +0 -1
- package/dist/chunk-RZTMDUO7.js +0 -49
- package/dist/chunk-RZTMDUO7.js.map +0 -1
- package/dist/chunk-SFLLL76A.js +0 -669
- package/dist/chunk-SFLLL76A.js.map +0 -1
- package/dist/chunk-SZLVEKMJ.js +0 -1446
- package/dist/chunk-SZLVEKMJ.js.map +0 -1
- package/dist/chunk-T4SQEITX.js +0 -95
- package/dist/chunk-T4SQEITX.js.map +0 -1
- package/dist/chunk-T6RLYGAD.js +0 -158
- package/dist/chunk-T6RLYGAD.js.map +0 -1
- package/dist/chunk-TJVT4QFF.js +0 -911
- package/dist/chunk-TJVT4QFF.js.map +0 -1
- package/dist/chunk-TQ7LNKZ3.js +0 -136
- package/dist/chunk-TQ7LNKZ3.js.map +0 -1
- package/dist/chunk-U4L7JRPZ.js +0 -1706
- package/dist/chunk-U4L7JRPZ.js.map +0 -1
- package/dist/chunk-U4PHLT2N.js +0 -419
- package/dist/chunk-U4PHLT2N.js.map +0 -1
- package/dist/chunk-VCZ5FQYW.js +0 -928
- package/dist/chunk-VCZ5FQYW.js.map +0 -1
- package/dist/chunk-VI2UW6B6.js +0 -162
- package/dist/chunk-VI2UW6B6.js.map +0 -1
- package/dist/chunk-VQMK5FMP.js +0 -247
- package/dist/chunk-VQMK5FMP.js.map +0 -1
- package/dist/chunk-WGXIEX7P.js +0 -116
- package/dist/chunk-WGXIEX7P.js.map +0 -1
- package/dist/chunk-WVATSFCP.js +0 -1553
- package/dist/chunk-WVATSFCP.js.map +0 -1
- package/dist/chunk-X4YIBDER.js +0 -1662
- package/dist/chunk-X4YIBDER.js.map +0 -1
- package/dist/chunk-YQN4ICPP.js +0 -355
- package/dist/chunk-YQN4ICPP.js.map +0 -1
- package/dist/chunk-ZET2UAYW.js +0 -89
- package/dist/chunk-ZET2UAYW.js.map +0 -1
- package/dist/chunk-ZHTZ4EYI.js +0 -1212
- package/dist/chunk-ZHTZ4EYI.js.map +0 -1
- package/dist/control.js.map +0 -1
- package/dist/hosted/index.js.map +0 -1
- package/dist/matrix/index.js.map +0 -1
- package/dist/reporting.js.map +0 -1
- package/dist/rollout/index.js.map +0 -1
- package/dist/run-campaign-OJJ7CZF4.js +0 -18
- package/dist/run-campaign-OJJ7CZF4.js.map +0 -1
- package/dist/supervisor-run/index.js.map +0 -1
- package/dist/traces.js.map +0 -1
- package/dist/wire/index.js.map +0 -1
|
@@ -1,1362 +1,622 @@
|
|
|
1
|
-
|
|
2
|
-
|
|
3
|
-
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
score?: number;
|
|
10
|
-
pass?: boolean;
|
|
11
|
-
failureClass?: FailureClass;
|
|
12
|
-
notes?: string;
|
|
13
|
-
}
|
|
14
|
-
/**
|
|
15
|
-
* Layer — optional classification in a nested build workflow.
|
|
16
|
-
* `builder`: the meta-agent editing a project (e.g. agent-builder Forge chat).
|
|
17
|
-
* `app-build`: sandbox harness that compiled + tested the generated scaffold.
|
|
18
|
-
* `app-runtime`: a run of the generated agent against a domain scenario.
|
|
19
|
-
* `meta`: any meta-eval (judge replay, correlation analysis).
|
|
20
|
-
*/
|
|
21
|
-
type RunLayer = 'builder' | 'app-build' | 'app-runtime' | 'meta' | 'custom';
|
|
22
|
-
interface Run {
|
|
23
|
-
runId: string;
|
|
24
|
-
/**
|
|
25
|
-
* Stable identifier of the scenario being executed.
|
|
26
|
-
*
|
|
27
|
-
* Always populated on the persisted Run — but `TraceEmitter.startRun` accepts
|
|
28
|
-
* input WITHOUT this field, substituting a sensible default
|
|
29
|
-
* (`run.layer ?? run.tags?.['kind'] ?? 'runtime'`) when the caller has no
|
|
30
|
-
* curated scenario to anchor to (runtime / operator / meta-eval runs). This
|
|
31
|
-
* keeps the persisted shape unambiguous for downstream filters + aggregations
|
|
32
|
-
* while removing the boilerplate of inventing placeholder ids at the call site.
|
|
33
|
-
*/
|
|
34
|
-
scenarioId: string;
|
|
35
|
-
variantId?: string;
|
|
36
|
-
datasetVersion?: string;
|
|
37
|
-
/** Git SHA of agent code at run time. */
|
|
38
|
-
codeSha?: string;
|
|
39
|
-
/** Hash of the prompt template + any system prompt. */
|
|
40
|
-
promptSha?: string;
|
|
41
|
-
/** Model id + date + system-prompt hash, concatenated. */
|
|
42
|
-
modelFingerprint?: string;
|
|
43
|
-
seed?: number;
|
|
44
|
-
/** Arbitrary environment markers (shell, docker version, tz). */
|
|
45
|
-
envFingerprint?: Record<string, string>;
|
|
46
|
-
/** Version of the redaction rules applied to this run. */
|
|
47
|
-
redactionVersion?: string;
|
|
48
|
-
/** Parent run in a nested build workflow. A builder run's children are
|
|
49
|
-
* app-build runs; those children are app-runtime runs. */
|
|
50
|
-
parentRunId?: string;
|
|
51
|
-
/** Stable project identifier — groups runs across chats + sessions. */
|
|
52
|
-
projectId?: string;
|
|
53
|
-
/** Chat/conversation identifier within a project. */
|
|
54
|
-
chatId?: string;
|
|
55
|
-
/** Layer classification — hint for aggregation; not enforced. */
|
|
56
|
-
layer?: RunLayer;
|
|
57
|
-
startedAt: number;
|
|
58
|
-
endedAt?: number;
|
|
59
|
-
status: RunStatus;
|
|
60
|
-
outcome?: RunOutcome$1;
|
|
61
|
-
budget?: BudgetSpec;
|
|
62
|
-
/** Free-form labels for downstream grouping. */
|
|
63
|
-
tags?: Record<string, string>;
|
|
64
|
-
}
|
|
65
|
-
type SpanKind = 'agent' | 'llm' | 'tool' | 'retrieval' | 'judge' | 'sandbox' | 'custom';
|
|
66
|
-
type SpanStatus = 'ok' | 'error';
|
|
67
|
-
interface SpanBase {
|
|
68
|
-
spanId: string;
|
|
69
|
-
parentSpanId?: string;
|
|
70
|
-
runId: string;
|
|
71
|
-
kind: SpanKind;
|
|
72
|
-
name: string;
|
|
73
|
-
startedAt: number;
|
|
74
|
-
endedAt?: number;
|
|
75
|
-
status?: SpanStatus;
|
|
76
|
-
error?: string;
|
|
77
|
-
/** Anything not covered by typed fields. Kept deliberately free-form. */
|
|
78
|
-
attributes?: Record<string, unknown>;
|
|
79
|
-
}
|
|
80
|
-
interface Message {
|
|
81
|
-
role: 'system' | 'user' | 'assistant' | 'tool';
|
|
82
|
-
content: string;
|
|
83
|
-
tokens?: number;
|
|
84
|
-
/** Multi-modal content descriptors; blobs themselves live in Artifacts. */
|
|
85
|
-
images?: Array<{
|
|
86
|
-
artifactId?: string;
|
|
87
|
-
url?: string;
|
|
88
|
-
mime?: string;
|
|
89
|
-
}>;
|
|
90
|
-
}
|
|
91
|
-
interface LlmSpan extends SpanBase {
|
|
92
|
-
kind: 'llm';
|
|
93
|
-
model: string;
|
|
94
|
-
messages: Message[];
|
|
95
|
-
output?: string;
|
|
96
|
-
inputTokens?: number;
|
|
97
|
-
/** All generated tokens, including the reasoning subset when present. */
|
|
98
|
-
outputTokens?: number;
|
|
99
|
-
cachedTokens?: number;
|
|
100
|
-
cacheWriteTokens?: number;
|
|
101
|
-
/** Reasoning-token subset of `outputTokens`. */
|
|
102
|
-
reasoningTokens?: number;
|
|
103
|
-
costUsd?: number;
|
|
104
|
-
finishReason?: string;
|
|
105
|
-
}
|
|
106
|
-
interface ToolSpan extends SpanBase {
|
|
107
|
-
kind: 'tool';
|
|
108
|
-
toolName: string;
|
|
109
|
-
args: unknown;
|
|
110
|
-
/** False when the source observed the call but did not capture its arguments. */
|
|
111
|
-
argsCaptured?: boolean;
|
|
112
|
-
result?: unknown;
|
|
113
|
-
latencyMs?: number;
|
|
114
|
-
}
|
|
115
|
-
interface RetrievalSpan extends SpanBase {
|
|
116
|
-
kind: 'retrieval';
|
|
117
|
-
query: string;
|
|
118
|
-
hits: Array<{
|
|
119
|
-
docId: string;
|
|
120
|
-
score: number;
|
|
121
|
-
content?: string;
|
|
122
|
-
}>;
|
|
123
|
-
}
|
|
124
|
-
interface JudgeSpan extends SpanBase {
|
|
125
|
-
kind: 'judge';
|
|
126
|
-
judgeId: string;
|
|
127
|
-
/** Span this judgment applies to. */
|
|
128
|
-
targetSpanId: string;
|
|
129
|
-
dimension: string;
|
|
130
|
-
/** Numeric score (free-range; interpretation up to the judge). */
|
|
131
|
-
score: number;
|
|
132
|
-
rationale?: string;
|
|
133
|
-
evidence?: string;
|
|
134
|
-
}
|
|
135
|
-
interface SandboxSpan extends SpanBase {
|
|
136
|
-
kind: 'sandbox';
|
|
137
|
-
image?: string;
|
|
138
|
-
command?: string;
|
|
139
|
-
exitCode?: number;
|
|
140
|
-
testsTotal?: number;
|
|
141
|
-
testsPassed?: number;
|
|
142
|
-
stdoutHash?: string;
|
|
143
|
-
stderrHash?: string;
|
|
144
|
-
/** Duration in ms; the harness fills this explicitly (endedAt - startedAt may miss setup). */
|
|
145
|
-
wallMs?: number;
|
|
146
|
-
}
|
|
147
|
-
interface GenericSpan extends SpanBase {
|
|
148
|
-
kind: 'agent' | 'custom';
|
|
149
|
-
}
|
|
150
|
-
type Span = LlmSpan | ToolSpan | RetrievalSpan | JudgeSpan | SandboxSpan | GenericSpan;
|
|
151
|
-
type EventKind = 'log' | 'error' | 'budget_decrement' | 'budget_breach' | 'state_mutation' | 'policy_violation' | 'redaction_applied' | 'custom';
|
|
152
|
-
interface TraceEvent {
|
|
153
|
-
eventId: string;
|
|
154
|
-
runId: string;
|
|
155
|
-
spanId?: string;
|
|
156
|
-
kind: EventKind;
|
|
157
|
-
timestamp: number;
|
|
158
|
-
payload: Record<string, unknown>;
|
|
159
|
-
}
|
|
160
|
-
interface BudgetLedgerEntry {
|
|
161
|
-
runId: string;
|
|
162
|
-
dimension: keyof BudgetSpec;
|
|
163
|
-
limit: number;
|
|
164
|
-
consumed: number;
|
|
165
|
-
remaining: number;
|
|
166
|
-
timestamp: number;
|
|
167
|
-
breached: boolean;
|
|
168
|
-
/** Span that triggered this entry, if any. */
|
|
169
|
-
spanId?: string;
|
|
170
|
-
}
|
|
171
|
-
interface Artifact {
|
|
172
|
-
artifactId: string;
|
|
173
|
-
runId: string;
|
|
174
|
-
spanId?: string;
|
|
175
|
-
contentType: string;
|
|
176
|
-
sizeBytes: number;
|
|
177
|
-
/** sha256 in hex. */
|
|
178
|
-
hash: string;
|
|
179
|
-
/** External storage URL (R2, S3, filesystem path). */
|
|
180
|
-
storageUrl?: string;
|
|
181
|
-
/** Inline content for small blobs — keep under ~64KB. */
|
|
182
|
-
inlineContent?: string;
|
|
183
|
-
}
|
|
184
|
-
type FailureClass = 'success' | 'reasoning_error' | 'tool_selection_error' | 'tool_argument_error' | 'tool_recovery_failure' | 'hallucination' | 'instruction_following' | 'safety_refusal_miss' | 'policy_violation' | 'budget_exceeded' | 'format_drift' | 'permission_escalation' | 'pii_leak' | 'cost_overrun' | 'timeout' | 'sandbox_failure' | 'missing_user_data' | 'missing_domain_data' | 'missing_codebase_context' | 'missing_runtime_context' | 'missing_credentials' | 'missing_integration_connection' | 'missing_integration_scope' | 'integration_approval_required' | 'integration_auth_expired' | 'integration_provider_failure' | 'bad_integration_manifest' | 'unsafe_integration_write_denied' | 'stale_external_data' | 'bad_retrieval' | 'insufficient_evidence' | 'contradictory_evidence' | 'ambiguous_user_intent' | 'knowledge_readiness_blocked' | 'unknown';
|
|
185
|
-
|
|
186
|
-
interface RunFilter {
|
|
187
|
-
scenarioId?: string;
|
|
188
|
-
variantId?: string;
|
|
189
|
-
status?: RunStatus;
|
|
190
|
-
since?: number;
|
|
191
|
-
until?: number;
|
|
192
|
-
tag?: {
|
|
193
|
-
key: string;
|
|
194
|
-
value: string;
|
|
195
|
-
};
|
|
196
|
-
parentRunId?: string;
|
|
197
|
-
projectId?: string;
|
|
198
|
-
chatId?: string;
|
|
199
|
-
layer?: RunLayer;
|
|
200
|
-
}
|
|
201
|
-
interface SpanFilter {
|
|
202
|
-
runId?: string;
|
|
203
|
-
parentSpanId?: string;
|
|
204
|
-
kind?: SpanKind;
|
|
205
|
-
name?: string;
|
|
206
|
-
toolName?: string;
|
|
207
|
-
judgeId?: string;
|
|
208
|
-
since?: number;
|
|
209
|
-
until?: number;
|
|
210
|
-
}
|
|
211
|
-
interface EventFilter {
|
|
212
|
-
runId?: string;
|
|
213
|
-
spanId?: string;
|
|
214
|
-
kind?: EventKind;
|
|
215
|
-
since?: number;
|
|
216
|
-
until?: number;
|
|
217
|
-
}
|
|
218
|
-
interface TraceStore {
|
|
219
|
-
appendRun(run: Run): Promise<void>;
|
|
220
|
-
updateRun(runId: string, patch: Partial<Run>): Promise<void>;
|
|
221
|
-
appendSpan(span: Span): Promise<void>;
|
|
222
|
-
updateSpan(spanId: string, patch: Partial<Span>): Promise<void>;
|
|
223
|
-
appendEvent(event: TraceEvent): Promise<void>;
|
|
224
|
-
appendArtifact(artifact: Artifact): Promise<void>;
|
|
225
|
-
appendBudgetEntry(entry: BudgetLedgerEntry): Promise<void>;
|
|
226
|
-
getRun(runId: string): Promise<Run | undefined>;
|
|
227
|
-
listRuns(filter?: RunFilter): Promise<Run[]>;
|
|
228
|
-
spans(filter?: SpanFilter): Promise<Span[]>;
|
|
229
|
-
events(filter?: EventFilter): Promise<TraceEvent[]>;
|
|
230
|
-
budget(runId: string): Promise<BudgetLedgerEntry[]>;
|
|
231
|
-
artifacts(runId: string): Promise<Artifact[]>;
|
|
232
|
-
}
|
|
233
|
-
|
|
234
|
-
/**
|
|
235
|
-
* Calibration curve — binned "if eval says X, what does reality show?"
|
|
236
|
-
*
|
|
237
|
-
* Companion to correlationStudy. Raw correlation is a single number;
|
|
238
|
-
* the calibration curve shows *where* the eval is well-calibrated vs
|
|
239
|
-
* overconfident / underconfident. Buckets the eval metric, computes
|
|
240
|
-
* mean outcome per bucket, reports expected-calibration-error (ECE).
|
|
241
|
-
*/
|
|
242
|
-
|
|
243
|
-
interface CalibrationBin {
|
|
244
|
-
lower: number;
|
|
245
|
-
upper: number;
|
|
246
|
-
n: number;
|
|
247
|
-
evalMean: number;
|
|
248
|
-
outcomeMean: number;
|
|
249
|
-
/** |outcomeMean − evalMean|; contributes to ECE weighted by n/total. */
|
|
250
|
-
gap: number;
|
|
251
|
-
}
|
|
252
|
-
interface CalibrationReport {
|
|
253
|
-
evalMetric: string;
|
|
254
|
-
outcomeMetric: string;
|
|
255
|
-
n: number;
|
|
256
|
-
bins: CalibrationBin[];
|
|
257
|
-
/** Expected Calibration Error — Σ (n_i/N) × |outcomeMean_i − evalMean_i|. */
|
|
258
|
-
ece: number;
|
|
259
|
-
/** Max bin gap — upper bound on miscalibration. */
|
|
260
|
-
maxGap: number;
|
|
261
|
-
}
|
|
262
|
-
|
|
263
|
-
/**
|
|
264
|
-
* Off-policy evaluation primitives.
|
|
265
|
-
*
|
|
266
|
-
* Standard inverse-probability-weighted (IPS), self-normalized
|
|
267
|
-
* importance-weighted (SNIPS), and doubly-robust (DR) estimators for the
|
|
268
|
-
* value of a *target* policy given trajectories collected under a
|
|
269
|
-
* *behavior* policy. This is the canonical RL eval task: "we have last
|
|
270
|
-
* week's runs, we changed the policy — how would the new one do without
|
|
271
|
-
* re-running?"
|
|
272
|
-
*
|
|
273
|
-
* The math here is textbook (Dudík, Langford, Li 2011 for DR; Swaminathan
|
|
274
|
-
* & Joachims 2015 for SNIPS) but the *application* to LLM-agent
|
|
275
|
-
* evaluation needs care:
|
|
276
|
-
*
|
|
277
|
-
* - The "policy" is the (prompt, tool config, model snapshot) triple.
|
|
278
|
-
* Two policies have the same probability over an action *iff* their
|
|
279
|
-
* LLM call would emit the same token with the same probability —
|
|
280
|
-
* which is generally unknowable without the model log-probs.
|
|
281
|
-
* - For LLM agents, propensity scores must be supplied by the caller
|
|
282
|
-
* (logged in the trace, recovered from token log-probs, or estimated
|
|
283
|
-
* via a learned propensity model). We do NOT estimate propensity here.
|
|
284
|
-
* - Doubly-robust requires two outputs from a Q-function: its prediction
|
|
285
|
-
* for the logged action and its expectation under the target policy.
|
|
286
|
-
* Consumers compute these with a tabular estimate, regression fit, or
|
|
287
|
-
* learned reward model before constructing the trajectories.
|
|
288
|
-
*
|
|
289
|
-
* Bias / variance tradeoffs:
|
|
290
|
-
* - IPS: unbiased; high variance for small overlap, infinite variance
|
|
291
|
-
* when target has support outside behavior.
|
|
292
|
-
* - SNIPS: lower variance, slight bias; usually preferred in practice.
|
|
293
|
-
* - DR: doubly-robust — unbiased if either propensity OR Q-function is
|
|
294
|
-
* correct. Lowest practical variance when Q is decent. Use this.
|
|
295
|
-
*
|
|
296
|
-
* Caveat the panel will land: on the LLM-agent setting, propensity scores
|
|
297
|
-
* recovered from token log-probs are noisy, the action space is enormous,
|
|
298
|
-
* and overlap is often poor. These estimators are useful but not magic;
|
|
299
|
-
* complement with `replayCampaign` (exact replay where the request hashes
|
|
300
|
-
* match) for high-confidence answers and OPE for the gap.
|
|
301
|
-
*/
|
|
302
|
-
interface OffPolicyTrajectory {
|
|
303
|
-
/** Stable id, for traceability through the dataset. */
|
|
304
|
-
runId: string;
|
|
305
|
-
/** Reward observed under the behavior policy (the realized outcome). */
|
|
306
|
-
reward: number;
|
|
307
|
-
/**
|
|
308
|
-
* Behavior-policy probability of the action that was taken. For LLM
|
|
309
|
-
* agents this is typically `exp(sum(token_log_probs))` over the chosen
|
|
310
|
-
* trajectory. Must be in (0, 1].
|
|
311
|
-
*/
|
|
312
|
-
behaviorProb: number;
|
|
313
|
-
/**
|
|
314
|
-
* Target-policy probability of the same action. For replay-style
|
|
315
|
-
* counterfactual evaluation this is what the *new* policy would have
|
|
316
|
-
* assigned to the *old* trajectory. Must be in [0, 1].
|
|
317
|
-
*/
|
|
318
|
-
targetProb: number;
|
|
319
|
-
/**
|
|
320
|
-
* Model-based reward prediction for the action selected by the behavior
|
|
321
|
-
* policy: `Q_hat(context, loggedAction)`. Supply this together with
|
|
322
|
-
* `vHatTarget` for contextual-bandit doubly-robust estimation.
|
|
323
|
-
*/
|
|
324
|
-
qHatChosen?: number | null;
|
|
325
|
-
/**
|
|
326
|
-
* Expected model-based reward under the target policy:
|
|
327
|
-
* `sum_action targetPolicy(action | context) * Q_hat(context, action)`.
|
|
328
|
-
* Supply this together with `qHatChosen`. For an honest evaluation, both
|
|
329
|
-
* values must come from a model cross-fitted or trained outside this row.
|
|
330
|
-
*/
|
|
331
|
-
vHatTarget?: number | null;
|
|
332
|
-
}
|
|
333
|
-
interface OffPolicyContributionCounts {
|
|
334
|
-
/** Contributions using the contextual-bandit doubly-robust formula. */
|
|
335
|
-
dr: number;
|
|
336
|
-
/** Contributions using exact IPS because no reward-model estimate was supplied. */
|
|
337
|
-
ipsFallback: number;
|
|
338
|
-
}
|
|
339
|
-
interface OffPolicyEstimate {
|
|
340
|
-
/** Estimated value of the target policy. */
|
|
341
|
-
value: number;
|
|
342
|
-
/** Standard error of the estimate. */
|
|
343
|
-
standardError: number;
|
|
344
|
-
/** Effective sample size (Kong 1992). Lower = more reliance on a few high-weight samples. */
|
|
345
|
-
effectiveSampleSize: number;
|
|
346
|
-
/** Number of trajectories used. */
|
|
347
|
-
n: number;
|
|
348
|
-
/**
|
|
349
|
-
* Diagnostic: maximum importance weight observed. Large values (>>10x
|
|
350
|
-
* mean) are a red flag — variance is dominated by a few outliers.
|
|
351
|
-
*/
|
|
352
|
-
maxImportanceWeight: number;
|
|
353
|
-
/** Populated by `doublyRobust` to expose which formula each row used. */
|
|
354
|
-
contributionCounts?: OffPolicyContributionCounts;
|
|
355
|
-
}
|
|
356
|
-
interface OffPolicyOptions {
|
|
357
|
-
/**
|
|
358
|
-
* Cap importance weights at this value (Ionides 2008 truncated IS) to
|
|
359
|
-
* trade unbiasedness for variance reduction. Default `Infinity` (no cap).
|
|
360
|
-
* Set e.g. `10` for stable estimates when the policies are close.
|
|
361
|
-
*/
|
|
362
|
-
weightCap?: number;
|
|
363
|
-
/** Reward clipping range. Default `[0, 1]`. */
|
|
364
|
-
rewardClip?: {
|
|
365
|
-
low: number;
|
|
366
|
-
high: number;
|
|
367
|
-
};
|
|
368
|
-
}
|
|
369
|
-
|
|
370
|
-
declare const BELIEF_DECISION_KINDS: readonly ["continue", "verify", "ask", "retry", "stop", "memory-write", "memory-read", "tool-select", "skill-select", "workflow-select", "surface-promote"];
|
|
1
|
+
import { a as RunRecord, s as RunSplitTag } from "../run-record-CnZu_gjl.js";
|
|
2
|
+
import { s as TraceStore } from "../store-CT9YIIve.js";
|
|
3
|
+
import { w as CalibrationReport } from "../index-6N0aYmpW.js";
|
|
4
|
+
import { i as OffPolicyTrajectory, n as OffPolicyEstimate, r as OffPolicyOptions } from "../off-policy-mskQw8Mb.js";
|
|
5
|
+
import { i as CodeAgentSessionMetrics, n as CodeAgentSessionIntakeOptions, t as CodeAgentSessionDiagnostic, v as CodeAgentSessionObservation, y as CodeAgentSessionSource } from "../code-agent-session-DqqgOJaz.js";
|
|
6
|
+
import { a as RuntimeTrajectoryRecord, n as RuntimeTrajectoryEvidenceProjection, t as ProjectRuntimeTrajectoryEvidenceOptions } from "../runtime-trajectory-BvSZcCHD.js";
|
|
7
|
+
//#region src/belief-state/types.d.ts
|
|
8
|
+
declare const BELIEF_DECISION_KINDS: readonly ['continue', 'verify', 'ask', 'retry', 'stop', 'memory-write', 'memory-read', 'tool-select', 'skill-select', 'workflow-select', 'surface-promote'];
|
|
371
9
|
type BeliefDecisionKind = (typeof BELIEF_DECISION_KINDS)[number];
|
|
372
|
-
declare const BELIEF_EVIDENCE_SOURCES: readonly [
|
|
10
|
+
declare const BELIEF_EVIDENCE_SOURCES: readonly ['run', 'span', 'event', 'finding', 'memory', 'knowledge', 'policy'];
|
|
373
11
|
type BeliefEvidenceSource = (typeof BELIEF_EVIDENCE_SOURCES)[number];
|
|
374
|
-
declare const BELIEF_EVIDENCE_QUALITIES: readonly [
|
|
12
|
+
declare const BELIEF_EVIDENCE_QUALITIES: readonly ['direct', 'derived', 'self-reported', 'unverified', 'stale', 'contradicted'];
|
|
375
13
|
type BeliefEvidenceQuality = (typeof BELIEF_EVIDENCE_QUALITIES)[number];
|
|
376
14
|
declare const BELIEF_EVALUATION_CRITERIA: readonly [{
|
|
377
|
-
|
|
378
|
-
|
|
379
|
-
|
|
15
|
+
readonly id: 'capture-integrity';
|
|
16
|
+
readonly label: 'Capture integrity';
|
|
17
|
+
readonly reasonCodes: readonly ['trace-missing', 'run-record-missing', 'backend-integrity-missing'];
|
|
380
18
|
}, {
|
|
381
|
-
|
|
382
|
-
|
|
383
|
-
|
|
19
|
+
readonly id: 'decision-completeness';
|
|
20
|
+
readonly label: 'Decision completeness';
|
|
21
|
+
readonly reasonCodes: readonly ['candidate-actions-missing', 'chosen-action-missing', 'decision-evidence-missing'];
|
|
384
22
|
}, {
|
|
385
|
-
|
|
386
|
-
|
|
387
|
-
|
|
23
|
+
readonly id: 'evidence-quality';
|
|
24
|
+
readonly label: 'Evidence quality';
|
|
25
|
+
readonly reasonCodes: readonly ['evidence-stale', 'evidence-contradictory', 'evidence-unverified', 'evidence-self-reported'];
|
|
388
26
|
}, {
|
|
389
|
-
|
|
390
|
-
|
|
391
|
-
|
|
27
|
+
readonly id: 'outcome-quality';
|
|
28
|
+
readonly label: 'Outcome quality';
|
|
29
|
+
readonly reasonCodes: readonly ['outcome-missing', 'outcome-delayed', 'cost-missing'];
|
|
392
30
|
}, {
|
|
393
|
-
|
|
394
|
-
|
|
395
|
-
|
|
31
|
+
readonly id: 'calibration';
|
|
32
|
+
readonly label: 'Calibration';
|
|
33
|
+
readonly reasonCodes: readonly ['confidence-missing', 'calibration-unsupported', 'calibration-gap-high'];
|
|
396
34
|
}, {
|
|
397
|
-
|
|
398
|
-
|
|
399
|
-
|
|
35
|
+
readonly id: 'accepted-region-risk';
|
|
36
|
+
readonly label: 'Accepted-region risk';
|
|
37
|
+
readonly reasonCodes: readonly ['accepted-error-high', 'coverage-too-low'];
|
|
400
38
|
}, {
|
|
401
|
-
|
|
402
|
-
|
|
403
|
-
|
|
39
|
+
readonly id: 'policy-value';
|
|
40
|
+
readonly label: 'Policy value';
|
|
41
|
+
readonly reasonCodes: readonly ['utility-lift-missing', 'baseline-dominates', 'cost-too-high'];
|
|
404
42
|
}, {
|
|
405
|
-
|
|
406
|
-
|
|
407
|
-
|
|
43
|
+
readonly id: 'ope-support';
|
|
44
|
+
readonly label: 'OPE support';
|
|
45
|
+
readonly reasonCodes: readonly ['behavior-propensity-missing', 'behavior-propensity-invalid', 'target-propensity-missing', 'target-propensity-invalid', 'effective-sample-size-low', 'importance-weight-high'];
|
|
408
46
|
}, {
|
|
409
|
-
|
|
410
|
-
|
|
411
|
-
|
|
47
|
+
readonly id: 'memory-health';
|
|
48
|
+
readonly label: 'Memory health';
|
|
49
|
+
readonly reasonCodes: readonly ['memory-stale', 'memory-poisoning-risk', 'context-bloat', 'memory-write-unverified'];
|
|
412
50
|
}, {
|
|
413
|
-
|
|
414
|
-
|
|
415
|
-
|
|
51
|
+
readonly id: 'surface-attribution';
|
|
52
|
+
readonly label: 'Surface attribution';
|
|
53
|
+
readonly reasonCodes: readonly ['surface-claim-unsupported', 'causal-attribution-missing'];
|
|
416
54
|
}, {
|
|
417
|
-
|
|
418
|
-
|
|
419
|
-
|
|
55
|
+
readonly id: 'generalization';
|
|
56
|
+
readonly label: 'Generalization';
|
|
57
|
+
readonly reasonCodes: readonly ['split-missing', 'holdout-regression', 'task-family-coverage-low', 'leakage-risk'];
|
|
420
58
|
}, {
|
|
421
|
-
|
|
422
|
-
|
|
423
|
-
|
|
59
|
+
readonly id: 'promotion';
|
|
60
|
+
readonly label: 'Promotion';
|
|
61
|
+
readonly reasonCodes: readonly ['negative-control-failed', 'promotion-gate-failed', 'human-review-required'];
|
|
424
62
|
}];
|
|
425
63
|
type BeliefEvaluationCriterionId = (typeof BELIEF_EVALUATION_CRITERIA)[number]['id'];
|
|
426
64
|
type BeliefDecisionReasonCode = (typeof BELIEF_EVALUATION_CRITERIA)[number]['reasonCodes'][number];
|
|
427
65
|
interface BeliefDecisionReason {
|
|
428
|
-
|
|
429
|
-
|
|
430
|
-
|
|
431
|
-
|
|
432
|
-
|
|
66
|
+
code: BeliefDecisionReasonCode;
|
|
67
|
+
criterion?: BeliefEvaluationCriterionId;
|
|
68
|
+
detail?: string;
|
|
69
|
+
evidenceIds?: string[];
|
|
70
|
+
metadata?: Record<string, unknown>;
|
|
433
71
|
}
|
|
434
72
|
declare function isBeliefDecisionKind(value: unknown): value is BeliefDecisionKind;
|
|
435
73
|
declare function isBeliefEvidenceSource(value: unknown): value is BeliefEvidenceSource;
|
|
436
74
|
interface BeliefEvidenceRef {
|
|
437
|
-
|
|
438
|
-
|
|
439
|
-
|
|
440
|
-
|
|
441
|
-
|
|
442
|
-
|
|
443
|
-
|
|
444
|
-
|
|
445
|
-
|
|
75
|
+
source: BeliefEvidenceSource;
|
|
76
|
+
id: string;
|
|
77
|
+
runId?: string;
|
|
78
|
+
spanId?: string;
|
|
79
|
+
eventId?: string;
|
|
80
|
+
detail?: string;
|
|
81
|
+
quality?: BeliefEvidenceQuality;
|
|
82
|
+
observedAt?: string;
|
|
83
|
+
metadata?: Record<string, unknown>;
|
|
446
84
|
}
|
|
447
85
|
interface BeliefDecisionOutcome {
|
|
448
|
-
|
|
449
|
-
|
|
450
|
-
|
|
451
|
-
|
|
452
|
-
|
|
453
|
-
|
|
86
|
+
success?: boolean;
|
|
87
|
+
score?: number;
|
|
88
|
+
reward?: number;
|
|
89
|
+
costUsd?: number;
|
|
90
|
+
observedAt?: string;
|
|
91
|
+
metadata?: Record<string, unknown>;
|
|
454
92
|
}
|
|
455
93
|
interface BeliefDecisionPoint {
|
|
456
|
-
|
|
457
|
-
|
|
458
|
-
|
|
459
|
-
|
|
460
|
-
|
|
461
|
-
|
|
462
|
-
|
|
463
|
-
|
|
464
|
-
|
|
465
|
-
|
|
466
|
-
|
|
467
|
-
|
|
468
|
-
|
|
469
|
-
|
|
470
|
-
|
|
471
|
-
|
|
472
|
-
|
|
94
|
+
id: string;
|
|
95
|
+
runId: string;
|
|
96
|
+
scenarioId?: string;
|
|
97
|
+
stepIndex: number;
|
|
98
|
+
kind: BeliefDecisionKind;
|
|
99
|
+
chosenAction: string;
|
|
100
|
+
candidateActions?: string[];
|
|
101
|
+
confidence?: number;
|
|
102
|
+
behaviorProb?: number;
|
|
103
|
+
targetProb?: number;
|
|
104
|
+
qHatChosen?: number | null;
|
|
105
|
+
vHatTarget?: number | null;
|
|
106
|
+
costUsd?: number;
|
|
107
|
+
evidence: BeliefEvidenceRef[];
|
|
108
|
+
outcome?: BeliefDecisionOutcome;
|
|
109
|
+
reasons?: BeliefDecisionReason[];
|
|
110
|
+
metadata?: Record<string, unknown>;
|
|
473
111
|
}
|
|
474
112
|
interface BeliefDecisionExtractionDiagnostic {
|
|
475
|
-
|
|
476
|
-
|
|
477
|
-
|
|
478
|
-
|
|
113
|
+
runId: string;
|
|
114
|
+
eventId?: string;
|
|
115
|
+
severity: 'info' | 'warning' | 'error';
|
|
116
|
+
reason: string;
|
|
479
117
|
}
|
|
480
118
|
interface BeliefDecisionExtractionReport {
|
|
481
|
-
|
|
482
|
-
|
|
119
|
+
decisions: BeliefDecisionPoint[];
|
|
120
|
+
diagnostics: BeliefDecisionExtractionDiagnostic[];
|
|
483
121
|
}
|
|
484
122
|
type BeliefPolicyAction = 'accept' | 'defer' | 'verify' | 'ask' | 'retry' | 'stop';
|
|
485
123
|
interface BeliefPolicyDecision {
|
|
486
|
-
|
|
487
|
-
|
|
488
|
-
|
|
489
|
-
|
|
490
|
-
|
|
491
|
-
|
|
492
|
-
|
|
124
|
+
action: BeliefPolicyAction;
|
|
125
|
+
confidence?: number;
|
|
126
|
+
targetProb?: number;
|
|
127
|
+
qHatChosen?: number | null;
|
|
128
|
+
vHatTarget?: number | null;
|
|
129
|
+
reason?: string;
|
|
130
|
+
reasons?: BeliefDecisionReason[];
|
|
493
131
|
}
|
|
494
132
|
interface BeliefSelectivePolicy {
|
|
495
|
-
|
|
496
|
-
|
|
133
|
+
id: string;
|
|
134
|
+
decide(point: BeliefDecisionPoint): BeliefPolicyDecision;
|
|
497
135
|
}
|
|
498
136
|
interface BeliefOpeTargetPolicy {
|
|
499
|
-
|
|
500
|
-
|
|
501
|
-
|
|
502
|
-
|
|
137
|
+
id: string;
|
|
138
|
+
targetProbOf(point: BeliefDecisionPoint): number | null | undefined;
|
|
139
|
+
qHatChosenOf?(point: BeliefDecisionPoint): number | null | undefined;
|
|
140
|
+
vHatTargetOf?(point: BeliefDecisionPoint): number | null | undefined;
|
|
503
141
|
}
|
|
504
142
|
interface BeliefUtilityOptions {
|
|
505
|
-
|
|
506
|
-
|
|
507
|
-
|
|
508
|
-
|
|
509
|
-
|
|
510
|
-
|
|
511
|
-
|
|
512
|
-
|
|
143
|
+
successUtility?: number;
|
|
144
|
+
failureUtility?: number;
|
|
145
|
+
deferUtility?: number;
|
|
146
|
+
verifyCost?: number;
|
|
147
|
+
askCost?: number;
|
|
148
|
+
retryCost?: number;
|
|
149
|
+
stopUtility?: number;
|
|
150
|
+
costWeight?: number;
|
|
513
151
|
}
|
|
514
152
|
interface BeliefSelectivePolicyMetrics {
|
|
515
|
-
|
|
516
|
-
|
|
517
|
-
|
|
518
|
-
|
|
519
|
-
|
|
520
|
-
|
|
521
|
-
|
|
522
|
-
|
|
523
|
-
|
|
524
|
-
|
|
525
|
-
|
|
526
|
-
|
|
527
|
-
|
|
528
|
-
|
|
529
|
-
|
|
530
|
-
|
|
531
|
-
|
|
153
|
+
policyId: string;
|
|
154
|
+
n: number;
|
|
155
|
+
accepted: number;
|
|
156
|
+
rejected: number;
|
|
157
|
+
coverage: number;
|
|
158
|
+
acceptedErrorRate: number;
|
|
159
|
+
baselineUtility: number;
|
|
160
|
+
policyUtility: number;
|
|
161
|
+
utilityDelta: number;
|
|
162
|
+
utilityCi95: {
|
|
163
|
+
mean: number;
|
|
164
|
+
lower: number;
|
|
165
|
+
upper: number;
|
|
166
|
+
};
|
|
167
|
+
rejectedMeanReward: number | null;
|
|
168
|
+
recommendation: 'ship' | 'hold' | 'need_more_data';
|
|
169
|
+
reasons: string[];
|
|
532
170
|
}
|
|
533
171
|
interface BeliefOpeSupportDiagnostics {
|
|
534
|
-
|
|
535
|
-
|
|
536
|
-
|
|
537
|
-
|
|
538
|
-
|
|
539
|
-
|
|
540
|
-
|
|
172
|
+
supported: boolean;
|
|
173
|
+
n: number;
|
|
174
|
+
dropped: number;
|
|
175
|
+
effectiveSampleSize: number;
|
|
176
|
+
effectiveSampleRatio: number;
|
|
177
|
+
maxImportanceWeight: number;
|
|
178
|
+
reasons: string[];
|
|
541
179
|
}
|
|
542
180
|
interface BeliefOpeReport {
|
|
543
|
-
|
|
544
|
-
|
|
545
|
-
|
|
546
|
-
|
|
547
|
-
|
|
181
|
+
targetPolicyId: string;
|
|
182
|
+
ips: OffPolicyEstimate;
|
|
183
|
+
snips: OffPolicyEstimate;
|
|
184
|
+
dr: OffPolicyEstimate;
|
|
185
|
+
support: BeliefOpeSupportDiagnostics;
|
|
548
186
|
}
|
|
549
187
|
type BeliefEvaluationStatus = 'ship' | 'hold' | 'need_more_data';
|
|
550
188
|
type BeliefCalibrationStatus = 'supported' | 'unsupported';
|
|
551
189
|
type BeliefOpeStatus = 'supported' | 'unsupported' | 'not_requested';
|
|
552
190
|
interface BeliefPolicyEvaluationReport {
|
|
553
|
-
|
|
554
|
-
|
|
555
|
-
|
|
556
|
-
|
|
557
|
-
|
|
558
|
-
|
|
559
|
-
|
|
560
|
-
|
|
561
|
-
|
|
562
|
-
|
|
563
|
-
|
|
564
|
-
}
|
|
565
|
-
|
|
191
|
+
policyId: string;
|
|
192
|
+
n: number;
|
|
193
|
+
status: BeliefEvaluationStatus;
|
|
194
|
+
selectiveStatus: BeliefEvaluationStatus;
|
|
195
|
+
calibrationStatus: BeliefCalibrationStatus;
|
|
196
|
+
opeStatus: BeliefOpeStatus;
|
|
197
|
+
opeTargetPolicyId?: string;
|
|
198
|
+
selective: BeliefSelectivePolicyMetrics;
|
|
199
|
+
calibration?: CalibrationReport;
|
|
200
|
+
ope?: BeliefOpeReport;
|
|
201
|
+
diagnostics: string[];
|
|
202
|
+
}
|
|
203
|
+
//#endregion
|
|
204
|
+
//#region src/belief-state/calibration.d.ts
|
|
566
205
|
type BeliefCalibrationRegion = 'all' | 'accepted' | 'rejected';
|
|
567
206
|
interface BeliefCalibrationOptions {
|
|
568
|
-
|
|
569
|
-
|
|
570
|
-
|
|
571
|
-
|
|
207
|
+
bins?: number;
|
|
208
|
+
minPairs?: number;
|
|
209
|
+
policy?: BeliefSelectivePolicy;
|
|
210
|
+
region?: BeliefCalibrationRegion;
|
|
572
211
|
}
|
|
573
212
|
declare function calibrateBeliefDecisions(points: BeliefDecisionPoint[], options?: BeliefCalibrationOptions): CalibrationReport | null;
|
|
574
|
-
|
|
575
|
-
|
|
576
|
-
type AgentProfileDimensionValue = string | number | boolean | null;
|
|
577
|
-
interface AgentProfileSource {
|
|
578
|
-
/** Runtime/profile contract being fingerprinted, e.g. `agent-interface-profile`. */
|
|
579
|
-
kind: string;
|
|
580
|
-
/** sha256 over the canonical source profile object. */
|
|
581
|
-
hash: string;
|
|
582
|
-
}
|
|
583
|
-
interface AgentProfileHarness {
|
|
584
|
-
id: string;
|
|
585
|
-
version?: string;
|
|
586
|
-
hash?: string;
|
|
587
|
-
}
|
|
588
|
-
interface AgentProfileCell {
|
|
589
|
-
schemaVersion: AgentProfileCellSchemaVersion;
|
|
590
|
-
cellId: string;
|
|
591
|
-
profileId: string;
|
|
592
|
-
sourceProfile: AgentProfileSource;
|
|
593
|
-
harness?: AgentProfileHarness;
|
|
594
|
-
model?: string;
|
|
595
|
-
promptHash?: string;
|
|
596
|
-
dimensions?: Record<string, AgentProfileDimensionValue>;
|
|
597
|
-
}
|
|
598
|
-
|
|
599
|
-
/**
|
|
600
|
-
* Paper-grade RunRecord schema + runtime validator.
|
|
601
|
-
*
|
|
602
|
-
* Every run that participates in a promotion gate, paper table, or
|
|
603
|
-
* researcher loop SHOULD be recorded as a `RunRecord`. The mandatory
|
|
604
|
-
* fields are exactly those the paper "Two Loops, Three Roles" requires
|
|
605
|
-
* for reproducibility: who/what/when/cost/seed/hash, plus the search vs
|
|
606
|
-
* holdout split tag. A task score is optional because execution-only records
|
|
607
|
-
* must preserve missing labels instead of converting errors into zero quality.
|
|
608
|
-
*
|
|
609
|
-
* This is intentionally NOT a replacement for the rich `Run` /
|
|
610
|
-
* `ProposeReviewReport` / `ScenarioResult` types already in the
|
|
611
|
-
* package. Those are runtime structures with full provenance. A
|
|
612
|
-
* `RunRecord` is the analysis-time projection — the JSON-friendly
|
|
613
|
-
* row you'd put in a parquet file or paste into a notebook.
|
|
614
|
-
*
|
|
615
|
-
* Validate at the boundary:
|
|
616
|
-
*
|
|
617
|
-
* const rec = validateRunRecord(rawJson) // throws on missing
|
|
618
|
-
* const ok = isRunRecord(rawJson) // boolean check
|
|
619
|
-
* const rec = parseRunRecordSafe(rawJson) // { ok, value | error }
|
|
620
|
-
*
|
|
621
|
-
* The validator runs in pure TS — zod is intentionally NOT a
|
|
622
|
-
* dependency. Round-trip tested in `tests/run-record.test.ts`.
|
|
623
|
-
*/
|
|
624
|
-
|
|
625
|
-
/** Search/dev/holdout split tag. 'search' is the paper-grade alias for the
|
|
626
|
-
* combined train+test pool that the optimizer is allowed to read. */
|
|
627
|
-
type RunSplitTag = 'search' | 'dev' | 'holdout';
|
|
628
|
-
/**
|
|
629
|
-
* Explicit execution-lifecycle result for a run.
|
|
630
|
-
*
|
|
631
|
-
* This is separate from task quality (`outcome`) and failure classification.
|
|
632
|
-
* Producers set it only from root-run or process evidence.
|
|
633
|
-
*/
|
|
634
|
-
type RunTerminalOutcome = 'succeeded' | 'failed' | 'cancelled' | 'incomplete' | 'unknown';
|
|
635
|
-
interface RunTokenUsage {
|
|
636
|
-
input: number;
|
|
637
|
-
/** All generated tokens charged as output, including reasoning tokens. */
|
|
638
|
-
output: number;
|
|
639
|
-
/** Reasoning-token subset of `output`, when the provider reports it. */
|
|
640
|
-
reasoning?: number;
|
|
641
|
-
/** Prompt tokens served from a provider cache. */
|
|
642
|
-
cached?: number;
|
|
643
|
-
/** Prompt tokens written into a provider cache. */
|
|
644
|
-
cacheWrite?: number;
|
|
645
|
-
}
|
|
646
|
-
/**
|
|
647
|
-
* How a run's USD amount was obtained.
|
|
648
|
-
*/
|
|
649
|
-
type RunCostProvenance = {
|
|
650
|
-
kind: 'observed';
|
|
651
|
-
usd: number;
|
|
652
|
-
} | {
|
|
653
|
-
kind: 'estimated';
|
|
654
|
-
usd: number;
|
|
655
|
-
} | {
|
|
656
|
-
kind: 'uncaptured';
|
|
657
|
-
usd: null;
|
|
658
|
-
};
|
|
659
|
-
interface RunJudgeMetadata {
|
|
660
|
-
model: string;
|
|
661
|
-
promptVersion: string;
|
|
662
|
-
/** [0,1] confidence the judge declared. Constant judge confidence
|
|
663
|
-
* across many runs is a fallback signal (see `canary.ts`). */
|
|
664
|
-
confidence: number;
|
|
665
|
-
/** True if the judge degraded to a fallback path (rules-only,
|
|
666
|
-
* prior-call cache, etc.). The canary uses this to alert. */
|
|
667
|
-
fallback: boolean;
|
|
668
|
-
}
|
|
669
|
-
/**
|
|
670
|
-
* Per-judge / per-dimension breakdown for runs scored by an ensemble of
|
|
671
|
-
* judges over a multi-dimensional rubric.
|
|
672
|
-
*
|
|
673
|
-
* The collapsed `outcome.searchScore` / `holdoutScore` carries the
|
|
674
|
-
* composite the gate uses. The full breakdown belongs here so consumers
|
|
675
|
-
* can answer "which judge disagreed?", "which dimension dragged the
|
|
676
|
-
* composite down?", and "did half the panel fail?" without re-running.
|
|
677
|
-
*
|
|
678
|
-
* `perJudge[judgeId][dim]` is the canonical source; `perDimMean` and
|
|
679
|
-
* `composite` are convenience projections — derivable but precomputed so
|
|
680
|
-
* downstream IRR primitives (`interRaterReliability`,
|
|
681
|
-
* `corpusInterRaterAgreement`) and reporters don't pay the same
|
|
682
|
-
* aggregation twice.
|
|
683
|
-
*
|
|
684
|
-
* Fail-loud discipline: judges that errored out land in `failedJudges`
|
|
685
|
-
* by id. A missing key in `perJudge` is ambiguous (silent zero vs not
|
|
686
|
-
* run); the explicit list makes a partial-failure recorded as such.
|
|
687
|
-
*/
|
|
688
|
-
interface JudgeScoresRecord {
|
|
689
|
-
/** Per-judge per-dimension scores. `{ "kimi-k2.6": { helpfulness: 0.8, clarity: 0.7 }, ... }`. */
|
|
690
|
-
perJudge: Record<string, Record<string, number>>;
|
|
691
|
-
/** Per-dim mean across judges. Convenience — derivable from `perJudge`. */
|
|
692
|
-
perDimMean: Record<string, number>;
|
|
693
|
-
/** Composite mean across successful judges. Mirrors the task score only
|
|
694
|
-
* when `failedJudges` is empty. */
|
|
695
|
-
composite: number;
|
|
696
|
-
/** Judges that errored or returned an unparseable verdict. Recorded
|
|
697
|
-
* by id (e.g. `['glm-5.1']`) so a partial-failure case is explicit,
|
|
698
|
-
* not inferred from missing keys in `perJudge`. */
|
|
699
|
-
failedJudges?: string[];
|
|
700
|
-
/** Free-form notes the judges emitted (joined across judges or
|
|
701
|
-
* first-judge only — consumer's choice). */
|
|
702
|
-
notes?: string;
|
|
703
|
-
}
|
|
704
|
-
interface RunOutcome {
|
|
705
|
-
/** Score on the search/optimization split. Optional for holdout-only and
|
|
706
|
-
* execution-only records. */
|
|
707
|
-
searchScore?: number;
|
|
708
|
-
/** Score on the held-out split. Optional for search-only and execution-only
|
|
709
|
-
* records. When both scores are absent, the run is explicitly unlabeled. */
|
|
710
|
-
holdoutScore?: number;
|
|
711
|
-
/** Bag of any other metric the run produced — judge dimensions,
|
|
712
|
-
* pass/fail counters, latency stats, etc. Numeric only — keeps
|
|
713
|
-
* reporters honest. */
|
|
714
|
-
raw: Record<string, number>;
|
|
715
|
-
/** Per-judge / per-dim breakdown. Consumers writing ensemble
|
|
716
|
-
* judgements populate this; substrate primitives like
|
|
717
|
-
* `interRaterReliability` and `corpusInterRaterAgreement` accept
|
|
718
|
-
* these records as input. Optional — single-judge or scalar-only
|
|
719
|
-
* runs leave it unset. */
|
|
720
|
-
judgeScores?: JudgeScoresRecord;
|
|
721
|
-
/** Authenticity / realness verdict — did the run build the REAL thing on the
|
|
722
|
-
* intended infra, or fake it (see `./authenticity`)? Optional: only domains
|
|
723
|
-
* with an authenticity config populate it. Carried in the corpus so the
|
|
724
|
-
* flywheel / off-policy learning can optimize for real completion, not gamed
|
|
725
|
-
* pass-rate. `score` is 0-1; `gated` is the anti-Goodhart flag — a gated run
|
|
726
|
-
* must not count as a real success regardless of `score`. */
|
|
727
|
-
realness?: {
|
|
728
|
-
score: number;
|
|
729
|
-
gated: boolean;
|
|
730
|
-
reason?: string;
|
|
731
|
-
};
|
|
732
|
-
}
|
|
733
|
-
/**
|
|
734
|
-
* Mandatory paper-grade fields for a single evaluation run. Optional
|
|
735
|
-
* fields are extension points; mandatory fields throw if missing.
|
|
736
|
-
*
|
|
737
|
-
* Hash discipline:
|
|
738
|
-
* - `promptHash` is the sha256 of the EFFECTIVE prompt sent to the
|
|
739
|
-
* model (after any steering bundle merge).
|
|
740
|
-
* - `configHash` is the sha256 of the effective run config (model,
|
|
741
|
-
* temperature, tools, judges, splits). The pair (promptHash,
|
|
742
|
-
* configHash) uniquely identifies an experiment cell.
|
|
743
|
-
*
|
|
744
|
-
* Model snapshot discipline:
|
|
745
|
-
* - `model` MUST encode a snapshot version. Bare aliases like
|
|
746
|
-
* `claude-sonnet-4` or `gpt-4o` are banned — they remap silently.
|
|
747
|
-
* Use `claude-sonnet-4-6@2025-04-15` or `gpt-4o-2024-11-20`.
|
|
748
|
-
*/
|
|
749
|
-
interface RunRecord {
|
|
750
|
-
/** UUID for the run. */
|
|
751
|
-
runId: string;
|
|
752
|
-
/** Logical experiment grouping (a treatment vs a baseline within
|
|
753
|
-
* the same sweep should share `experimentId`). */
|
|
754
|
-
experimentId: string;
|
|
755
|
-
/** Stable identifier for the candidate (variant) being run. The
|
|
756
|
-
* promotion gate compares two `candidateId`s on matched items. */
|
|
757
|
-
candidateId: string;
|
|
758
|
-
/** RNG seed for the run. Always recorded — silent re-seeding is
|
|
759
|
-
* the most common cause of non-reproducible numbers. */
|
|
760
|
-
seed: number;
|
|
761
|
-
/** Model identifier WITH snapshot version. */
|
|
762
|
-
model: string;
|
|
763
|
-
/** sha256 of the effective prompt (post-steering). */
|
|
764
|
-
promptHash: string;
|
|
765
|
-
/** sha256 of the effective config. */
|
|
766
|
-
configHash: string;
|
|
767
|
-
/** Git SHA the harness was run from. */
|
|
768
|
-
commitSha: string;
|
|
769
|
-
/** End-to-end wall-clock duration in milliseconds. */
|
|
770
|
-
wallMs: number;
|
|
771
|
-
/** Time spent queued before execution started, if known. */
|
|
772
|
-
queueMs?: number;
|
|
773
|
-
/** Total USD cost, or null when the producer could not capture one. */
|
|
774
|
-
costUsd: number | null;
|
|
775
|
-
/** Whether `costUsd` came from billing data, a price calculation, or is unavailable. */
|
|
776
|
-
costProvenance: RunCostProvenance;
|
|
777
|
-
/** Token usage breakdown. */
|
|
778
|
-
tokenUsage: RunTokenUsage;
|
|
779
|
-
/** Root-run or process terminal result. Never inferred from a child span. */
|
|
780
|
-
terminalOutcome: RunTerminalOutcome;
|
|
781
|
-
/** Root-run or process failure reason. Valid only for a failed, cancelled,
|
|
782
|
-
* or incomplete terminal result; never populated from a child span. */
|
|
783
|
-
terminalFailureReason?: string;
|
|
784
|
-
/** Judge-side metadata, if a judge was used. */
|
|
785
|
-
judgeMetadata?: RunJudgeMetadata;
|
|
786
|
-
/** Per-split scores + raw bag. */
|
|
787
|
-
outcome: RunOutcome;
|
|
788
|
-
/** Canonical task-failure class drawn from the shared
|
|
789
|
-
* `FAILURE_CLASSES` taxonomy. Producers set it only from task-result
|
|
790
|
-
* evidence. Execution errors belong in
|
|
791
|
-
* `outcome.raw.execution_error_count`. */
|
|
792
|
-
failureClass?: FailureClass;
|
|
793
|
-
/** Free-form task-failure detail scoped under a non-success
|
|
794
|
-
* `failureClass`. It is invalid without that class. */
|
|
795
|
-
failureMode?: string;
|
|
796
|
-
/** Which split this run was drawn from. */
|
|
797
|
-
splitTag: RunSplitTag;
|
|
798
|
-
/**
|
|
799
|
-
* Stable scenario identifier the run observed or was scored against.
|
|
800
|
-
* Comparison primitives match this identity rather than input order.
|
|
801
|
-
*/
|
|
802
|
-
scenarioId: string;
|
|
803
|
-
/**
|
|
804
|
-
* Canonical identity for the agent profile cell that produced this row:
|
|
805
|
-
* profile artifact hash plus optional harness/model/prompt/reporting
|
|
806
|
-
* dimensions. Use `agentProfile.cellId` to group persona sweeps and
|
|
807
|
-
* longitudinal reports by the complete source profile, not by a loose
|
|
808
|
-
* candidate label or opaque config hash.
|
|
809
|
-
*/
|
|
810
|
-
agentProfile?: AgentProfileCell;
|
|
811
|
-
}
|
|
812
|
-
|
|
813
|
-
type CodeAgentSessionSource = 'codex' | 'claude-code' | 'opencode' | 'kimi-code' | 'pi';
|
|
814
|
-
type CodeAgentSessionTerminalStatus = 'completed' | 'failed' | 'unknown';
|
|
815
|
-
type CodeAgentSessionActionKind = 'tool' | 'patch' | 'terminal' | 'graph-completion';
|
|
816
|
-
type CodeAgentSessionActionSurface = 'tool' | 'mcp' | 'subagent' | 'skill' | 'hook' | 'web' | 'code';
|
|
817
|
-
type CodeAgentSessionActionStatus = 'started' | 'completed' | 'failed' | 'unknown';
|
|
818
|
-
interface CodeAgentSessionExecutionReceipt {
|
|
819
|
-
exitCode: number;
|
|
820
|
-
startedAtMs?: number;
|
|
821
|
-
completedAtMs?: number;
|
|
822
|
-
}
|
|
823
|
-
interface CodeAgentSessionAction {
|
|
824
|
-
id: string;
|
|
825
|
-
stepIndex: number;
|
|
826
|
-
kind: CodeAgentSessionActionKind;
|
|
827
|
-
surface: CodeAgentSessionActionSurface;
|
|
828
|
-
name: string;
|
|
829
|
-
status: CodeAgentSessionActionStatus;
|
|
830
|
-
timestampMs?: number;
|
|
831
|
-
costUsd?: number;
|
|
832
|
-
metadata: Record<string, unknown>;
|
|
833
|
-
}
|
|
834
|
-
interface CodeAgentSessionObservation {
|
|
835
|
-
source: CodeAgentSessionSource;
|
|
836
|
-
sessionId: string;
|
|
837
|
-
finalText: string | null;
|
|
838
|
-
terminal: {
|
|
839
|
-
status: CodeAgentSessionTerminalStatus;
|
|
840
|
-
explicit: boolean;
|
|
841
|
-
};
|
|
842
|
-
actions: CodeAgentSessionAction[];
|
|
843
|
-
}
|
|
844
|
-
|
|
845
|
-
interface CodeAgentSessionMetrics {
|
|
846
|
-
entries: number;
|
|
847
|
-
userMessages: number;
|
|
848
|
-
assistantMessages: number;
|
|
849
|
-
reasoningItems: number;
|
|
850
|
-
toolCalls: number;
|
|
851
|
-
toolOutputs: number;
|
|
852
|
-
toolErrors: number;
|
|
853
|
-
unclassifiedErrors: number;
|
|
854
|
-
patchAttempts: number;
|
|
855
|
-
patchSuccesses: number;
|
|
856
|
-
patchFailures: number;
|
|
857
|
-
turnsStarted: number;
|
|
858
|
-
turnsCompleted: number;
|
|
859
|
-
turnsAborted: number;
|
|
860
|
-
contextCompactions: number;
|
|
861
|
-
mcpCalls: number;
|
|
862
|
-
subagentCalls: number;
|
|
863
|
-
skillCalls: number;
|
|
864
|
-
hookCalls: number;
|
|
865
|
-
webCalls: number;
|
|
866
|
-
codeActions: number;
|
|
867
|
-
prLinks: number;
|
|
868
|
-
fileSnapshots: number;
|
|
869
|
-
graphNodes: number;
|
|
870
|
-
graphEdges: number;
|
|
871
|
-
actionCandidates: number;
|
|
872
|
-
verificationReports: number;
|
|
873
|
-
completionDecisions: number;
|
|
874
|
-
reliabilityRows: number;
|
|
875
|
-
reliabilityLift: number;
|
|
876
|
-
inputTokens: number;
|
|
877
|
-
outputTokens: number;
|
|
878
|
-
reasoningTokens: number;
|
|
879
|
-
cachedTokens: number;
|
|
880
|
-
cacheWriteTokens: number;
|
|
881
|
-
observedCostUsd: number;
|
|
882
|
-
observedCostCaptured?: boolean;
|
|
883
|
-
wallMs: number;
|
|
884
|
-
processScore: number;
|
|
885
|
-
}
|
|
886
|
-
interface CodeAgentSessionDiagnostic {
|
|
887
|
-
source: CodeAgentSessionSource;
|
|
888
|
-
sessionId: string;
|
|
889
|
-
sourcePath?: string;
|
|
890
|
-
entries: number;
|
|
891
|
-
malformedLines: number;
|
|
892
|
-
hasExplicitTerminalSignal: boolean;
|
|
893
|
-
hasFinalOutput: boolean;
|
|
894
|
-
hasQualityLabel: boolean;
|
|
895
|
-
hasTokenUsage: boolean;
|
|
896
|
-
hasCost: boolean;
|
|
897
|
-
costKind?: RunCostProvenance['kind'];
|
|
898
|
-
warnings: string[];
|
|
899
|
-
}
|
|
900
|
-
interface CodeAgentSessionIntakeOptions {
|
|
901
|
-
entries: unknown[];
|
|
902
|
-
malformedLines?: number;
|
|
903
|
-
sourcePath?: string;
|
|
904
|
-
experimentId?: string;
|
|
905
|
-
candidateId?: string;
|
|
906
|
-
seed?: number;
|
|
907
|
-
splitTag?: RunSplitTag;
|
|
908
|
-
scenarioId?: string;
|
|
909
|
-
model?: string;
|
|
910
|
-
promptHash?: string;
|
|
911
|
-
configHash?: string;
|
|
912
|
-
commitSha?: string;
|
|
913
|
-
score?: number;
|
|
914
|
-
/** Explicit cost receipt. When omitted, source-reported cost wins, then a
|
|
915
|
-
* token-priced estimate, then uncaptured. */
|
|
916
|
-
costProvenance?: RunCostProvenance;
|
|
917
|
-
/** Exact executor-owned process result. This is required when a provider's
|
|
918
|
-
* JSON stream has no terminal event, as with `opencode run --format json`. */
|
|
919
|
-
execution?: CodeAgentSessionExecutionReceipt;
|
|
920
|
-
}
|
|
921
|
-
|
|
213
|
+
//#endregion
|
|
214
|
+
//#region src/belief-state/ope.d.ts
|
|
922
215
|
interface BeliefOpeOptions extends OffPolicyOptions {
|
|
923
|
-
|
|
924
|
-
|
|
925
|
-
|
|
216
|
+
minEffectiveSampleSize?: number;
|
|
217
|
+
minEffectiveSampleRatio?: number;
|
|
218
|
+
maxDiagnostics?: number;
|
|
926
219
|
}
|
|
927
220
|
interface BeliefOffPolicyTrajectoryReport {
|
|
928
|
-
|
|
929
|
-
|
|
930
|
-
|
|
931
|
-
|
|
221
|
+
targetPolicyId: string;
|
|
222
|
+
trajectories: OffPolicyTrajectory[];
|
|
223
|
+
dropped: number;
|
|
224
|
+
diagnostics: string[];
|
|
932
225
|
}
|
|
933
226
|
declare function embeddedBeliefOpeTargetPolicy(id?: string): BeliefOpeTargetPolicy;
|
|
934
227
|
declare function beliefDecisionsToOffPolicyTrajectories(points: BeliefDecisionPoint[], targetPolicy: BeliefOpeTargetPolicy, options?: Pick<BeliefOpeOptions, 'maxDiagnostics'>): BeliefOffPolicyTrajectoryReport;
|
|
935
228
|
declare function evaluateBeliefOffPolicy(points: BeliefDecisionPoint[], targetPolicy: BeliefOpeTargetPolicy, options?: BeliefOpeOptions): BeliefOpeReport;
|
|
936
|
-
|
|
229
|
+
//#endregion
|
|
230
|
+
//#region src/belief-state/selective.d.ts
|
|
937
231
|
interface EvaluateBeliefSelectivePolicyOptions {
|
|
938
|
-
|
|
939
|
-
|
|
940
|
-
|
|
941
|
-
|
|
942
|
-
|
|
232
|
+
utility?: BeliefUtilityOptions;
|
|
233
|
+
minN?: number;
|
|
234
|
+
minAccepted?: number;
|
|
235
|
+
minUtilityDelta?: number;
|
|
236
|
+
seed?: number;
|
|
943
237
|
}
|
|
944
238
|
declare function thresholdSelectivePolicy(options: {
|
|
945
|
-
|
|
946
|
-
|
|
947
|
-
|
|
239
|
+
id?: string;
|
|
240
|
+
confidenceThreshold: number;
|
|
241
|
+
belowThresholdAction?: Exclude<BeliefPolicyAction, 'accept'>;
|
|
948
242
|
}): BeliefSelectivePolicy;
|
|
949
243
|
declare function evaluateBeliefSelectivePolicy(points: BeliefDecisionPoint[], policy: BeliefSelectivePolicy, options?: EvaluateBeliefSelectivePolicyOptions): BeliefSelectivePolicyMetrics;
|
|
950
|
-
|
|
244
|
+
//#endregion
|
|
245
|
+
//#region src/belief-state/report.d.ts
|
|
951
246
|
interface AnalyzeBeliefPolicyOpeOptions extends BeliefOpeOptions {
|
|
952
|
-
|
|
247
|
+
targetPolicy?: BeliefOpeTargetPolicy;
|
|
953
248
|
}
|
|
954
249
|
interface AnalyzeBeliefPolicyOptions {
|
|
955
|
-
|
|
956
|
-
|
|
957
|
-
|
|
958
|
-
|
|
959
|
-
|
|
960
|
-
|
|
250
|
+
points: BeliefDecisionPoint[];
|
|
251
|
+
policy: BeliefSelectivePolicy;
|
|
252
|
+
selective?: EvaluateBeliefSelectivePolicyOptions;
|
|
253
|
+
calibration?: BeliefCalibrationOptions;
|
|
254
|
+
ope?: AnalyzeBeliefPolicyOpeOptions;
|
|
255
|
+
requireOpe?: boolean;
|
|
961
256
|
}
|
|
962
257
|
declare function analyzeBeliefPolicy(options: AnalyzeBeliefPolicyOptions): BeliefPolicyEvaluationReport;
|
|
963
|
-
|
|
258
|
+
//#endregion
|
|
259
|
+
//#region src/belief-state/code-agent-corpus.d.ts
|
|
964
260
|
type CodeAgentBeliefDecisionTargetId = 'failure-recovery' | 'tool-selection' | 'graph-completion';
|
|
965
261
|
interface ExtractCodeAgentBeliefDecisionPointsOptions {
|
|
966
|
-
|
|
967
|
-
|
|
968
|
-
|
|
969
|
-
|
|
970
|
-
|
|
971
|
-
|
|
262
|
+
source: CodeAgentSessionSource;
|
|
263
|
+
entries: unknown[];
|
|
264
|
+
/** Reuse intake's provider-neutral projection when available. */
|
|
265
|
+
observation?: CodeAgentSessionObservation;
|
|
266
|
+
run: Pick<RunRecord, 'runId' | 'scenarioId' | 'outcome' | 'costUsd'>;
|
|
267
|
+
sourcePath?: string;
|
|
972
268
|
}
|
|
973
269
|
interface BeliefDecisionInventoryBucket {
|
|
974
|
-
|
|
975
|
-
|
|
976
|
-
|
|
977
|
-
|
|
978
|
-
|
|
979
|
-
|
|
980
|
-
|
|
981
|
-
|
|
982
|
-
|
|
983
|
-
|
|
984
|
-
|
|
985
|
-
|
|
270
|
+
id: string;
|
|
271
|
+
kind?: BeliefDecisionKind;
|
|
272
|
+
targetId?: CodeAgentBeliefDecisionTargetId;
|
|
273
|
+
n: number;
|
|
274
|
+
withOutcome: number;
|
|
275
|
+
withConfidence: number;
|
|
276
|
+
withCandidateActions: number;
|
|
277
|
+
withBehaviorProb: number;
|
|
278
|
+
withTargetProb: number;
|
|
279
|
+
successRate: number | null;
|
|
280
|
+
meanScore: number | null;
|
|
281
|
+
meanConfidence: number | null;
|
|
986
282
|
}
|
|
987
283
|
interface BeliefDecisionInventoryReport {
|
|
988
|
-
|
|
989
|
-
|
|
990
|
-
|
|
991
|
-
|
|
284
|
+
n: number;
|
|
285
|
+
byKind: BeliefDecisionInventoryBucket[];
|
|
286
|
+
byTarget: BeliefDecisionInventoryBucket[];
|
|
287
|
+
diagnostics: string[];
|
|
992
288
|
}
|
|
993
289
|
interface BeliefDecisionTargetSelection {
|
|
994
|
-
|
|
995
|
-
|
|
996
|
-
|
|
997
|
-
|
|
998
|
-
|
|
290
|
+
id: CodeAgentBeliefDecisionTargetId;
|
|
291
|
+
label: string;
|
|
292
|
+
points: BeliefDecisionPoint[];
|
|
293
|
+
support: BeliefDecisionInventoryBucket;
|
|
294
|
+
reasons: string[];
|
|
999
295
|
}
|
|
1000
296
|
interface SelectBeliefDecisionTargetOptions {
|
|
1001
|
-
|
|
1002
|
-
|
|
1003
|
-
|
|
297
|
+
minN?: number;
|
|
298
|
+
minOutcomeCoverage?: number;
|
|
299
|
+
preferredTargets?: CodeAgentBeliefDecisionTargetId[];
|
|
1004
300
|
}
|
|
1005
301
|
interface AnalyzeBeliefDecisionCorpusOptions {
|
|
1006
|
-
|
|
1007
|
-
|
|
1008
|
-
|
|
1009
|
-
|
|
1010
|
-
|
|
1011
|
-
|
|
1012
|
-
|
|
1013
|
-
|
|
1014
|
-
|
|
302
|
+
points: BeliefDecisionPoint[];
|
|
303
|
+
targetId?: CodeAgentBeliefDecisionTargetId;
|
|
304
|
+
minN?: number;
|
|
305
|
+
minOutcomeCoverage?: number;
|
|
306
|
+
minAccepted?: number;
|
|
307
|
+
confidenceThreshold?: number;
|
|
308
|
+
policy?: BeliefSelectivePolicy;
|
|
309
|
+
requireOpe?: boolean;
|
|
310
|
+
policyOptions?: Partial<AnalyzeBeliefPolicyOptions>;
|
|
1015
311
|
}
|
|
1016
312
|
interface BeliefDecisionCorpusEvaluation {
|
|
1017
|
-
|
|
1018
|
-
|
|
1019
|
-
|
|
1020
|
-
|
|
1021
|
-
|
|
313
|
+
inventory: BeliefDecisionInventoryReport;
|
|
314
|
+
target?: BeliefDecisionTargetSelection;
|
|
315
|
+
policy?: BeliefSelectivePolicy;
|
|
316
|
+
evaluation?: BeliefPolicyEvaluationReport;
|
|
317
|
+
diagnostics: string[];
|
|
1022
318
|
}
|
|
1023
319
|
declare function extractCodeAgentBeliefDecisionPoints(options: ExtractCodeAgentBeliefDecisionPointsOptions): BeliefDecisionExtractionReport;
|
|
1024
320
|
declare function inventoryBeliefDecisionPoints(points: BeliefDecisionPoint[]): BeliefDecisionInventoryReport;
|
|
1025
321
|
declare function selectBeliefDecisionTarget(points: BeliefDecisionPoint[], options?: SelectBeliefDecisionTargetOptions): BeliefDecisionTargetSelection | null;
|
|
1026
322
|
declare function analyzeBeliefDecisionCorpus(options: AnalyzeBeliefDecisionCorpusOptions): BeliefDecisionCorpusEvaluation;
|
|
1027
|
-
|
|
323
|
+
//#endregion
|
|
324
|
+
//#region src/belief-state/research-evidence.d.ts
|
|
1028
325
|
type BeliefResearchClaimScope = 'selective' | 'counterfactual';
|
|
1029
326
|
type BeliefResearchEvidenceStatus = 'supported' | 'blocked';
|
|
1030
327
|
type BeliefResearchGateId = 'corpus' | 'selective' | 'calibration' | 'ope';
|
|
1031
328
|
interface BeliefResearchEvidenceGate {
|
|
1032
|
-
|
|
1033
|
-
|
|
1034
|
-
|
|
1035
|
-
|
|
329
|
+
id: BeliefResearchGateId;
|
|
330
|
+
status: BeliefResearchEvidenceStatus;
|
|
331
|
+
blockers: string[];
|
|
332
|
+
caveats: string[];
|
|
1036
333
|
}
|
|
1037
334
|
interface BeliefDecisionResearchEvidencePacket {
|
|
1038
|
-
|
|
1039
|
-
|
|
1040
|
-
|
|
1041
|
-
|
|
1042
|
-
|
|
1043
|
-
|
|
335
|
+
claimScope: BeliefResearchClaimScope;
|
|
336
|
+
status: BeliefResearchEvidenceStatus;
|
|
337
|
+
analysis: BeliefDecisionCorpusEvaluation;
|
|
338
|
+
gates: BeliefResearchEvidenceGate[];
|
|
339
|
+
blockers: string[];
|
|
340
|
+
caveats: string[];
|
|
1044
341
|
}
|
|
1045
342
|
interface BuildBeliefDecisionResearchEvidencePacketOptions extends AnalyzeBeliefDecisionCorpusOptions {
|
|
1046
|
-
|
|
343
|
+
claimScope?: BeliefResearchClaimScope;
|
|
1047
344
|
}
|
|
1048
345
|
declare function buildBeliefDecisionResearchEvidencePacket(options: BuildBeliefDecisionResearchEvidencePacketOptions): BeliefDecisionResearchEvidencePacket;
|
|
1049
|
-
|
|
346
|
+
//#endregion
|
|
347
|
+
//#region src/belief-state/code-agent-evidence.d.ts
|
|
1050
348
|
interface CodeAgentBeliefSession extends CodeAgentSessionIntakeOptions {
|
|
1051
|
-
|
|
349
|
+
source: CodeAgentSessionSource;
|
|
1052
350
|
}
|
|
1053
351
|
interface BuildCodeAgentBeliefEvidenceCorpusOptions extends Omit<BuildBeliefDecisionResearchEvidencePacketOptions, 'points'> {
|
|
1054
|
-
|
|
352
|
+
sessions: CodeAgentBeliefSession[];
|
|
1055
353
|
}
|
|
1056
354
|
interface CodeAgentBeliefEvidenceCorpus {
|
|
1057
|
-
|
|
1058
|
-
|
|
1059
|
-
|
|
1060
|
-
|
|
1061
|
-
|
|
1062
|
-
|
|
1063
|
-
|
|
355
|
+
runs: RunRecord[];
|
|
356
|
+
metrics: CodeAgentSessionMetrics[];
|
|
357
|
+
intakeDiagnostics: CodeAgentSessionDiagnostic[];
|
|
358
|
+
extractionDiagnostics: BeliefDecisionExtractionDiagnostic[];
|
|
359
|
+
decisions: BeliefDecisionPoint[];
|
|
360
|
+
inventory: BeliefDecisionInventoryReport;
|
|
361
|
+
evidence: BeliefDecisionResearchEvidencePacket;
|
|
1064
362
|
}
|
|
1065
363
|
declare function buildCodeAgentBeliefEvidenceCorpus(options: BuildCodeAgentBeliefEvidenceCorpusOptions): CodeAgentBeliefEvidenceCorpus;
|
|
1066
|
-
|
|
364
|
+
//#endregion
|
|
365
|
+
//#region src/belief-state/extract.d.ts
|
|
1067
366
|
interface ExtractBeliefDecisionPointsOptions {
|
|
1068
|
-
|
|
367
|
+
runIds?: string[];
|
|
1069
368
|
}
|
|
1070
369
|
declare function extractBeliefDecisionPoints(store: TraceStore, options?: ExtractBeliefDecisionPointsOptions): Promise<BeliefDecisionExtractionReport>;
|
|
1071
|
-
|
|
370
|
+
//#endregion
|
|
371
|
+
//#region src/belief-state/shadow-probe.d.ts
|
|
1072
372
|
interface BeliefShadowProbeInput {
|
|
1073
|
-
|
|
1074
|
-
|
|
1075
|
-
|
|
1076
|
-
|
|
1077
|
-
|
|
1078
|
-
|
|
1079
|
-
|
|
1080
|
-
|
|
1081
|
-
|
|
1082
|
-
|
|
1083
|
-
|
|
373
|
+
probeId: string;
|
|
374
|
+
decisionId: string;
|
|
375
|
+
runId: string;
|
|
376
|
+
scenarioId?: string;
|
|
377
|
+
stepIndex: number;
|
|
378
|
+
decisionKind: BeliefDecisionKind;
|
|
379
|
+
candidateActions: string[];
|
|
380
|
+
observedAction?: string;
|
|
381
|
+
evidence: BeliefShadowProbeEvidenceRef[];
|
|
382
|
+
context?: string;
|
|
383
|
+
metadata?: Record<string, unknown>;
|
|
1084
384
|
}
|
|
1085
385
|
interface BeliefShadowProbeEvidenceRef {
|
|
1086
|
-
|
|
1087
|
-
|
|
1088
|
-
|
|
1089
|
-
|
|
386
|
+
id: string;
|
|
387
|
+
source: string;
|
|
388
|
+
detail?: string;
|
|
389
|
+
quality?: BeliefEvidenceQuality;
|
|
1090
390
|
}
|
|
1091
391
|
interface BeliefShadowProbeResponse {
|
|
1092
|
-
|
|
1093
|
-
|
|
1094
|
-
|
|
1095
|
-
|
|
1096
|
-
|
|
1097
|
-
|
|
1098
|
-
|
|
1099
|
-
|
|
1100
|
-
|
|
1101
|
-
|
|
392
|
+
predictedAction: string;
|
|
393
|
+
confidence: number;
|
|
394
|
+
beliefSummary?: string;
|
|
395
|
+
uncertainty?: string[];
|
|
396
|
+
evidenceRefs?: string[];
|
|
397
|
+
wouldChangeMindIf?: string[];
|
|
398
|
+
targetProb?: number;
|
|
399
|
+
qHatChosen?: number | null;
|
|
400
|
+
vHatTarget?: number | null;
|
|
401
|
+
metadata?: Record<string, unknown>;
|
|
1102
402
|
}
|
|
1103
403
|
interface BeliefShadowProbeRecord extends BeliefShadowProbeResponse {
|
|
1104
|
-
|
|
1105
|
-
|
|
1106
|
-
|
|
1107
|
-
|
|
1108
|
-
|
|
1109
|
-
|
|
1110
|
-
|
|
1111
|
-
|
|
1112
|
-
|
|
1113
|
-
|
|
404
|
+
probeId: string;
|
|
405
|
+
decisionId: string;
|
|
406
|
+
runId: string;
|
|
407
|
+
scenarioId?: string;
|
|
408
|
+
stepIndex: number;
|
|
409
|
+
decisionKind: BeliefDecisionKind;
|
|
410
|
+
candidateActions: string[];
|
|
411
|
+
observedAction: string;
|
|
412
|
+
agreesWithObservedAction: boolean;
|
|
413
|
+
outcome?: BeliefDecisionOutcome;
|
|
1114
414
|
}
|
|
1115
415
|
interface BeliefShadowProbeDiagnostic {
|
|
1116
|
-
|
|
1117
|
-
|
|
1118
|
-
|
|
416
|
+
decisionId: string;
|
|
417
|
+
severity: 'warning' | 'error';
|
|
418
|
+
reason: string;
|
|
1119
419
|
}
|
|
1120
420
|
interface BeliefShadowProbeSummary {
|
|
1121
|
-
|
|
1122
|
-
|
|
1123
|
-
|
|
1124
|
-
|
|
1125
|
-
|
|
1126
|
-
|
|
1127
|
-
|
|
421
|
+
attempted: number;
|
|
422
|
+
completed: number;
|
|
423
|
+
dropped: number;
|
|
424
|
+
withOutcome: number;
|
|
425
|
+
withTargetProb: number;
|
|
426
|
+
meanConfidence: number | null;
|
|
427
|
+
observedAgreementRate: number | null;
|
|
1128
428
|
}
|
|
1129
429
|
interface BeliefShadowProbeRun {
|
|
1130
|
-
|
|
1131
|
-
|
|
1132
|
-
|
|
1133
|
-
|
|
430
|
+
probeId: string;
|
|
431
|
+
records: BeliefShadowProbeRecord[];
|
|
432
|
+
diagnostics: BeliefShadowProbeDiagnostic[];
|
|
433
|
+
summary: BeliefShadowProbeSummary;
|
|
1134
434
|
}
|
|
1135
435
|
interface RunBeliefShadowProbeOptions {
|
|
1136
|
-
|
|
1137
|
-
|
|
1138
|
-
|
|
1139
|
-
|
|
1140
|
-
|
|
1141
|
-
|
|
1142
|
-
|
|
1143
|
-
|
|
1144
|
-
|
|
1145
|
-
|
|
1146
|
-
|
|
1147
|
-
|
|
436
|
+
probeId: string;
|
|
437
|
+
points: BeliefDecisionPoint[];
|
|
438
|
+
probe: (input: BeliefShadowProbeInput) => BeliefShadowProbeResponse | Promise<BeliefShadowProbeResponse>;
|
|
439
|
+
contextOf?: (point: BeliefDecisionPoint) => string | undefined | Promise<string | undefined>;
|
|
440
|
+
metadataOf?: (point: BeliefDecisionPoint) => Record<string, unknown> | undefined | Promise<Record<string, unknown> | undefined>;
|
|
441
|
+
includeObservedAction?: boolean;
|
|
442
|
+
includeEvidenceDetail?: boolean;
|
|
443
|
+
includeOutcomeInRecord?: boolean;
|
|
444
|
+
requireCandidateActions?: boolean;
|
|
445
|
+
allowOutOfSetActions?: boolean;
|
|
446
|
+
concurrency?: number;
|
|
447
|
+
maxContextChars?: number;
|
|
1148
448
|
}
|
|
1149
449
|
declare function runBeliefShadowProbe(options: RunBeliefShadowProbeOptions): Promise<BeliefShadowProbeRun>;
|
|
1150
450
|
declare function formatBeliefShadowProbePrompt(input: BeliefShadowProbeInput): string;
|
|
1151
|
-
|
|
451
|
+
//#endregion
|
|
452
|
+
//#region src/belief-state/runtime-hooks.d.ts
|
|
1152
453
|
interface RuntimeBeliefDecisionEvidenceRef {
|
|
1153
|
-
|
|
1154
|
-
|
|
1155
|
-
|
|
1156
|
-
|
|
1157
|
-
|
|
454
|
+
source: string;
|
|
455
|
+
id: string;
|
|
456
|
+
detail?: string;
|
|
457
|
+
quality?: BeliefEvidenceQuality;
|
|
458
|
+
metadata?: Record<string, unknown>;
|
|
1158
459
|
}
|
|
1159
460
|
interface RuntimeBeliefDecisionPoint {
|
|
1160
|
-
|
|
1161
|
-
|
|
1162
|
-
|
|
1163
|
-
|
|
1164
|
-
|
|
1165
|
-
|
|
1166
|
-
|
|
1167
|
-
|
|
1168
|
-
|
|
461
|
+
id: string;
|
|
462
|
+
runId: string;
|
|
463
|
+
scenarioId?: string;
|
|
464
|
+
stepIndex: number;
|
|
465
|
+
kind: string;
|
|
466
|
+
candidateActions?: string[];
|
|
467
|
+
context?: string;
|
|
468
|
+
evidence?: RuntimeBeliefDecisionEvidenceRef[];
|
|
469
|
+
metadata?: Record<string, unknown>;
|
|
1169
470
|
}
|
|
1170
471
|
interface RuntimeBeliefHookEvent {
|
|
1171
|
-
|
|
1172
|
-
|
|
1173
|
-
|
|
1174
|
-
|
|
1175
|
-
|
|
1176
|
-
|
|
1177
|
-
|
|
1178
|
-
|
|
1179
|
-
|
|
1180
|
-
|
|
472
|
+
id: string;
|
|
473
|
+
runId: string;
|
|
474
|
+
scenarioId?: string;
|
|
475
|
+
target: string;
|
|
476
|
+
phase: string;
|
|
477
|
+
timestamp: number;
|
|
478
|
+
stepIndex?: number;
|
|
479
|
+
parentId?: string;
|
|
480
|
+
payload?: unknown;
|
|
481
|
+
metadata?: Record<string, unknown>;
|
|
1181
482
|
}
|
|
1182
483
|
interface RuntimeBeliefHookContext {
|
|
1183
|
-
|
|
484
|
+
signal?: AbortSignal;
|
|
1184
485
|
}
|
|
1185
486
|
interface RuntimeBeliefHooks {
|
|
1186
|
-
|
|
1187
|
-
|
|
487
|
+
onEvent?: (event: RuntimeBeliefHookEvent, context: RuntimeBeliefHookContext) => void | Promise<void>;
|
|
488
|
+
onDecisionPoint?: (point: RuntimeBeliefDecisionPoint, context: RuntimeBeliefHookContext) => void | Promise<void>;
|
|
1188
489
|
}
|
|
1189
490
|
interface RuntimeBeliefConversionDiagnostic {
|
|
1190
|
-
|
|
1191
|
-
|
|
1192
|
-
|
|
491
|
+
decisionId: string;
|
|
492
|
+
severity: 'warning' | 'error';
|
|
493
|
+
reason: string;
|
|
1193
494
|
}
|
|
1194
495
|
interface RuntimeBeliefShadowProbeInputOptions {
|
|
1195
|
-
|
|
1196
|
-
|
|
1197
|
-
|
|
1198
|
-
|
|
1199
|
-
|
|
1200
|
-
|
|
496
|
+
probeId: string;
|
|
497
|
+
decisionKind?: BeliefDecisionKind;
|
|
498
|
+
includeEvidenceDetail?: boolean;
|
|
499
|
+
includeLifecycleEvidence?: boolean;
|
|
500
|
+
lifecycleEvents?: RuntimeBeliefHookEvent[];
|
|
501
|
+
maxContextChars?: number;
|
|
1201
502
|
}
|
|
1202
503
|
interface RuntimeBeliefDecisionPointOptions {
|
|
1203
|
-
|
|
1204
|
-
|
|
1205
|
-
|
|
1206
|
-
|
|
1207
|
-
|
|
1208
|
-
|
|
1209
|
-
|
|
1210
|
-
|
|
1211
|
-
|
|
1212
|
-
|
|
1213
|
-
|
|
1214
|
-
|
|
504
|
+
chosenAction?: string;
|
|
505
|
+
decisionKind?: BeliefDecisionKind;
|
|
506
|
+
confidence?: number;
|
|
507
|
+
behaviorProb?: number;
|
|
508
|
+
targetProb?: number;
|
|
509
|
+
qHatChosen?: number | null;
|
|
510
|
+
vHatTarget?: number | null;
|
|
511
|
+
costUsd?: number;
|
|
512
|
+
outcome?: BeliefDecisionOutcome;
|
|
513
|
+
metadata?: Record<string, unknown>;
|
|
514
|
+
includeLifecycleEvidence?: boolean;
|
|
515
|
+
lifecycleEvents?: RuntimeBeliefHookEvent[];
|
|
1215
516
|
}
|
|
1216
517
|
interface RuntimeBeliefShadowProbeInputReport {
|
|
1217
|
-
|
|
1218
|
-
|
|
518
|
+
input?: BeliefShadowProbeInput;
|
|
519
|
+
diagnostics: RuntimeBeliefConversionDiagnostic[];
|
|
1219
520
|
}
|
|
1220
521
|
interface RuntimeBeliefDecisionPointReport {
|
|
1221
|
-
|
|
1222
|
-
|
|
522
|
+
point?: BeliefDecisionPoint;
|
|
523
|
+
diagnostics: RuntimeBeliefConversionDiagnostic[];
|
|
1223
524
|
}
|
|
1224
525
|
interface BeliefRuntimeHookCollector {
|
|
1225
|
-
|
|
1226
|
-
|
|
1227
|
-
|
|
1228
|
-
|
|
1229
|
-
|
|
1230
|
-
|
|
1231
|
-
|
|
1232
|
-
|
|
526
|
+
hooks: RuntimeBeliefHooks;
|
|
527
|
+
decisions: RuntimeBeliefDecisionPoint[];
|
|
528
|
+
events: RuntimeBeliefHookEvent[];
|
|
529
|
+
toShadowProbeInputs(options?: Partial<RuntimeBeliefShadowProbeInputOptions>): {
|
|
530
|
+
inputs: BeliefShadowProbeInput[];
|
|
531
|
+
diagnostics: RuntimeBeliefConversionDiagnostic[];
|
|
532
|
+
};
|
|
533
|
+
clear(): void;
|
|
1233
534
|
}
|
|
1234
535
|
declare function runtimeDecisionPointToBeliefShadowProbeInput(point: RuntimeBeliefDecisionPoint, options: RuntimeBeliefShadowProbeInputOptions): RuntimeBeliefShadowProbeInputReport;
|
|
1235
536
|
declare function runtimeDecisionPointToBeliefDecisionPoint(point: RuntimeBeliefDecisionPoint, options: RuntimeBeliefDecisionPointOptions): RuntimeBeliefDecisionPointReport;
|
|
1236
537
|
declare function createBeliefRuntimeHookCollector(defaults: RuntimeBeliefShadowProbeInputOptions): BeliefRuntimeHookCollector;
|
|
1237
|
-
|
|
538
|
+
//#endregion
|
|
539
|
+
//#region src/belief-state/phase0-measurement.d.ts
|
|
1238
540
|
interface RuntimeBeliefPhase0RunRecord {
|
|
1239
|
-
|
|
1240
|
-
|
|
1241
|
-
|
|
541
|
+
runId: string;
|
|
542
|
+
scenarioId?: string;
|
|
543
|
+
splitTag: RunSplitTag;
|
|
1242
544
|
}
|
|
1243
545
|
interface RuntimeBeliefDecisionLabel {
|
|
1244
|
-
|
|
1245
|
-
|
|
1246
|
-
|
|
1247
|
-
|
|
1248
|
-
|
|
1249
|
-
|
|
1250
|
-
|
|
1251
|
-
|
|
1252
|
-
|
|
1253
|
-
|
|
1254
|
-
|
|
546
|
+
decisionId: string;
|
|
547
|
+
chosenAction: string;
|
|
548
|
+
outcome: BeliefDecisionOutcome;
|
|
549
|
+
confidence?: number;
|
|
550
|
+
behaviorProb?: number;
|
|
551
|
+
targetProb?: number;
|
|
552
|
+
qHatChosen?: number | null;
|
|
553
|
+
vHatTarget?: number | null;
|
|
554
|
+
costUsd?: number;
|
|
555
|
+
splitTag?: RunSplitTag;
|
|
556
|
+
metadata?: Record<string, unknown>;
|
|
1255
557
|
}
|
|
1256
558
|
interface BuildRuntimeBeliefPhase0MeasurementOptions extends Omit<BuildBeliefDecisionResearchEvidencePacketOptions, 'points'> {
|
|
1257
|
-
|
|
1258
|
-
|
|
1259
|
-
|
|
1260
|
-
|
|
1261
|
-
|
|
559
|
+
runs: RuntimeBeliefPhase0RunRecord[];
|
|
560
|
+
decisions: RuntimeBeliefDecisionPoint[];
|
|
561
|
+
events?: RuntimeBeliefHookEvent[];
|
|
562
|
+
labels: RuntimeBeliefDecisionLabel[];
|
|
563
|
+
baselinePolicyId?: string;
|
|
1262
564
|
}
|
|
1263
565
|
interface RuntimeBeliefPhase0MeasurementSummary {
|
|
1264
|
-
|
|
1265
|
-
|
|
1266
|
-
|
|
1267
|
-
|
|
1268
|
-
|
|
1269
|
-
|
|
1270
|
-
|
|
1271
|
-
|
|
1272
|
-
|
|
1273
|
-
|
|
1274
|
-
|
|
1275
|
-
|
|
1276
|
-
|
|
1277
|
-
|
|
1278
|
-
|
|
1279
|
-
|
|
1280
|
-
|
|
566
|
+
runCount: number;
|
|
567
|
+
producerDecisionCount: number;
|
|
568
|
+
lifecycleEventCount: number;
|
|
569
|
+
labelCount: number;
|
|
570
|
+
completedPointCount: number;
|
|
571
|
+
runJoinRate: number;
|
|
572
|
+
labelJoinRate: number;
|
|
573
|
+
missingRunRecordCount: number;
|
|
574
|
+
missingLabelCount: number;
|
|
575
|
+
withEvidence: number;
|
|
576
|
+
withOutcome: number;
|
|
577
|
+
withSplit: number;
|
|
578
|
+
withBehaviorProb: number;
|
|
579
|
+
withTargetProb: number;
|
|
580
|
+
baselinePolicyId: string;
|
|
581
|
+
packetStatus: BeliefDecisionResearchEvidencePacket['status'];
|
|
582
|
+
claimScope: BeliefDecisionResearchEvidencePacket['claimScope'];
|
|
1281
583
|
}
|
|
1282
584
|
interface RuntimeBeliefPhase0Measurement {
|
|
1283
|
-
|
|
1284
|
-
|
|
1285
|
-
|
|
1286
|
-
|
|
585
|
+
points: BeliefDecisionPoint[];
|
|
586
|
+
packet: BeliefDecisionResearchEvidencePacket;
|
|
587
|
+
summary: RuntimeBeliefPhase0MeasurementSummary;
|
|
588
|
+
diagnostics: string[];
|
|
1287
589
|
}
|
|
1288
590
|
declare function buildRuntimeBeliefPhase0Measurement(options: BuildRuntimeBeliefPhase0MeasurementOptions): RuntimeBeliefPhase0Measurement;
|
|
1289
|
-
|
|
1290
|
-
|
|
1291
|
-
id: string;
|
|
1292
|
-
runId: string;
|
|
1293
|
-
scenarioId?: string;
|
|
1294
|
-
target: string;
|
|
1295
|
-
phase: string;
|
|
1296
|
-
timestamp: number;
|
|
1297
|
-
stepIndex?: number;
|
|
1298
|
-
parentId?: string;
|
|
1299
|
-
payload?: unknown;
|
|
1300
|
-
metadata?: Record<string, unknown>;
|
|
1301
|
-
}
|
|
1302
|
-
interface RuntimeTrajectoryRecord {
|
|
1303
|
-
id?: string;
|
|
1304
|
-
scenarioId?: string;
|
|
1305
|
-
splitTag?: RunSplitTag;
|
|
1306
|
-
runtimeEvents?: unknown;
|
|
1307
|
-
[key: string]: unknown;
|
|
1308
|
-
}
|
|
1309
|
-
interface RuntimeTrajectoryRunRecord {
|
|
1310
|
-
runId: string;
|
|
1311
|
-
scenarioId?: string;
|
|
1312
|
-
splitTag: RunSplitTag;
|
|
1313
|
-
}
|
|
1314
|
-
interface RuntimeTrajectoryEvidenceSummary {
|
|
1315
|
-
recordCount: number;
|
|
1316
|
-
recordWithRuntimeEventsCount: number;
|
|
1317
|
-
runtimeRunCount: number;
|
|
1318
|
-
lifecycleEventCount: number;
|
|
1319
|
-
defaultedSplitCount: number;
|
|
1320
|
-
}
|
|
1321
|
-
interface RuntimeTrajectoryEvidenceProjection {
|
|
1322
|
-
runs: RuntimeTrajectoryRunRecord[];
|
|
1323
|
-
events: RuntimeTrajectoryHookEvent[];
|
|
1324
|
-
summary: RuntimeTrajectoryEvidenceSummary;
|
|
1325
|
-
diagnostics: string[];
|
|
1326
|
-
}
|
|
1327
|
-
interface ProjectRuntimeTrajectoryEvidenceOptions<TRecord extends RuntimeTrajectoryRecord = RuntimeTrajectoryRecord> {
|
|
1328
|
-
records: TRecord[];
|
|
1329
|
-
defaultSplitTag?: RunSplitTag;
|
|
1330
|
-
recordIdOf?: (record: TRecord, index: number) => string | undefined;
|
|
1331
|
-
scenarioIdOf?: (record: TRecord, index: number) => string | undefined;
|
|
1332
|
-
}
|
|
1333
|
-
|
|
591
|
+
//#endregion
|
|
592
|
+
//#region src/belief-state/runtime-benchmark-corpus.d.ts
|
|
1334
593
|
type RuntimeBenchmarkTrajectoryRecord = RuntimeTrajectoryRecord & {
|
|
1335
|
-
|
|
1336
|
-
|
|
1337
|
-
|
|
1338
|
-
|
|
594
|
+
benchmark?: unknown;
|
|
595
|
+
condition?: unknown;
|
|
596
|
+
instanceId?: unknown;
|
|
597
|
+
runtimeDecisionPoints?: unknown;
|
|
1339
598
|
};
|
|
1340
599
|
interface BuildRuntimeBenchmarkBeliefPhase0MeasurementOptions extends Omit<BuildRuntimeBeliefPhase0MeasurementOptions, 'runs' | 'events' | 'decisions' | 'labels'> {
|
|
1341
|
-
|
|
1342
|
-
|
|
1343
|
-
|
|
1344
|
-
|
|
600
|
+
records: RuntimeBenchmarkTrajectoryRecord[];
|
|
601
|
+
decisions?: RuntimeBeliefDecisionPoint[];
|
|
602
|
+
defaultSplitTag?: ProjectRuntimeTrajectoryEvidenceOptions['defaultSplitTag'];
|
|
603
|
+
labels?: RuntimeBeliefDecisionLabel[];
|
|
1345
604
|
}
|
|
1346
605
|
interface RuntimeBenchmarkBeliefPhase0Summary {
|
|
1347
|
-
|
|
1348
|
-
|
|
606
|
+
decisionCount: number;
|
|
607
|
+
labelCount: number;
|
|
1349
608
|
}
|
|
1350
609
|
interface RuntimeBenchmarkBeliefPhase0Measurement {
|
|
1351
|
-
|
|
1352
|
-
|
|
1353
|
-
|
|
1354
|
-
|
|
1355
|
-
|
|
1356
|
-
|
|
1357
|
-
|
|
1358
|
-
|
|
610
|
+
runs: RuntimeBeliefPhase0RunRecord[];
|
|
611
|
+
events: RuntimeBeliefHookEvent[];
|
|
612
|
+
decisions: RuntimeBeliefDecisionPoint[];
|
|
613
|
+
labels: RuntimeBeliefDecisionLabel[];
|
|
614
|
+
trajectory: RuntimeTrajectoryEvidenceProjection;
|
|
615
|
+
measurement: RuntimeBeliefPhase0Measurement;
|
|
616
|
+
summary: RuntimeBenchmarkBeliefPhase0Summary;
|
|
617
|
+
diagnostics: string[];
|
|
1359
618
|
}
|
|
1360
619
|
declare function buildRuntimeBenchmarkBeliefPhase0Measurement(options: BuildRuntimeBenchmarkBeliefPhase0MeasurementOptions): RuntimeBenchmarkBeliefPhase0Measurement;
|
|
1361
|
-
|
|
1362
|
-
export {
|
|
620
|
+
//#endregion
|
|
621
|
+
export { AnalyzeBeliefDecisionCorpusOptions, AnalyzeBeliefPolicyOpeOptions, AnalyzeBeliefPolicyOptions, BELIEF_DECISION_KINDS, BELIEF_EVALUATION_CRITERIA, BELIEF_EVIDENCE_QUALITIES, BELIEF_EVIDENCE_SOURCES, BeliefCalibrationOptions, BeliefCalibrationRegion, BeliefCalibrationStatus, BeliefDecisionCorpusEvaluation, BeliefDecisionExtractionDiagnostic, BeliefDecisionExtractionReport, BeliefDecisionInventoryBucket, BeliefDecisionInventoryReport, BeliefDecisionKind, BeliefDecisionOutcome, BeliefDecisionPoint, BeliefDecisionReason, BeliefDecisionReasonCode, BeliefDecisionResearchEvidencePacket, BeliefDecisionTargetSelection, BeliefEvaluationCriterionId, BeliefEvaluationStatus, BeliefEvidenceQuality, BeliefEvidenceRef, BeliefEvidenceSource, BeliefOffPolicyTrajectoryReport, BeliefOpeOptions, BeliefOpeReport, BeliefOpeStatus, BeliefOpeSupportDiagnostics, BeliefOpeTargetPolicy, BeliefPolicyAction, BeliefPolicyDecision, BeliefPolicyEvaluationReport, BeliefResearchClaimScope, BeliefResearchEvidenceGate, BeliefResearchEvidenceStatus, BeliefResearchGateId, BeliefRuntimeHookCollector, BeliefSelectivePolicy, BeliefSelectivePolicyMetrics, BeliefShadowProbeDiagnostic, BeliefShadowProbeEvidenceRef, BeliefShadowProbeInput, BeliefShadowProbeRecord, BeliefShadowProbeResponse, BeliefShadowProbeRun, BeliefShadowProbeSummary, BeliefUtilityOptions, BuildBeliefDecisionResearchEvidencePacketOptions, BuildCodeAgentBeliefEvidenceCorpusOptions, BuildRuntimeBeliefPhase0MeasurementOptions, BuildRuntimeBenchmarkBeliefPhase0MeasurementOptions, CodeAgentBeliefDecisionTargetId, CodeAgentBeliefEvidenceCorpus, CodeAgentBeliefSession, EvaluateBeliefSelectivePolicyOptions, ExtractBeliefDecisionPointsOptions, ExtractCodeAgentBeliefDecisionPointsOptions, RunBeliefShadowProbeOptions, RuntimeBeliefConversionDiagnostic, RuntimeBeliefDecisionEvidenceRef, RuntimeBeliefDecisionLabel, RuntimeBeliefDecisionPoint, RuntimeBeliefDecisionPointOptions, RuntimeBeliefDecisionPointReport, RuntimeBeliefHookContext, RuntimeBeliefHookEvent, RuntimeBeliefHooks, RuntimeBeliefPhase0Measurement, RuntimeBeliefPhase0MeasurementSummary, RuntimeBeliefPhase0RunRecord, RuntimeBeliefShadowProbeInputOptions, RuntimeBeliefShadowProbeInputReport, RuntimeBenchmarkBeliefPhase0Measurement, RuntimeBenchmarkBeliefPhase0Summary, SelectBeliefDecisionTargetOptions, analyzeBeliefDecisionCorpus, analyzeBeliefPolicy, beliefDecisionsToOffPolicyTrajectories, buildBeliefDecisionResearchEvidencePacket, buildCodeAgentBeliefEvidenceCorpus, buildRuntimeBeliefPhase0Measurement, buildRuntimeBenchmarkBeliefPhase0Measurement, calibrateBeliefDecisions, createBeliefRuntimeHookCollector, embeddedBeliefOpeTargetPolicy, evaluateBeliefOffPolicy, evaluateBeliefSelectivePolicy, extractBeliefDecisionPoints, extractCodeAgentBeliefDecisionPoints, formatBeliefShadowProbePrompt, inventoryBeliefDecisionPoints, isBeliefDecisionKind, isBeliefEvidenceSource, runBeliefShadowProbe, runtimeDecisionPointToBeliefDecisionPoint, runtimeDecisionPointToBeliefShadowProbeInput, selectBeliefDecisionTarget, thresholdSelectivePolicy };
|
|
622
|
+
//# sourceMappingURL=index.d.ts.map
|