@tangle-network/agent-eval 0.128.2 → 0.130.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +279 -0
- package/README.md +19 -0
- package/dist/active-curriculum-C4mk67HP.js +214 -0
- package/dist/active-curriculum-C4mk67HP.js.map +1 -0
- package/dist/adversarial-smnADNFS.d.ts +21 -0
- package/dist/adversarial-smnADNFS.d.ts.map +1 -0
- package/dist/analyst/index.d.ts +83 -2932
- package/dist/analyst/index.d.ts.map +1 -0
- package/dist/analyst/index.js +319 -364
- package/dist/analyst/index.js.map +1 -1
- package/dist/analyst-BkTS3C58.d.ts +89 -0
- package/dist/analyst-BkTS3C58.d.ts.map +1 -0
- package/dist/analyst-LsnNpSkm.js +152 -0
- package/dist/analyst-LsnNpSkm.js.map +1 -0
- package/dist/analyze-runs-C1CavBMk.js +1067 -0
- package/dist/analyze-runs-C1CavBMk.js.map +1 -0
- package/dist/analyze-runs-FsgYCinh.d.ts +72 -0
- package/dist/analyze-runs-FsgYCinh.d.ts.map +1 -0
- package/dist/attribute-vocabulary-DLJ6303h.d.ts +54 -0
- package/dist/attribute-vocabulary-DLJ6303h.d.ts.map +1 -0
- package/dist/authenticity/index.d.ts +81 -79
- package/dist/authenticity/index.d.ts.map +1 -0
- package/dist/authenticity/index.js +209 -193
- package/dist/authenticity/index.js.map +1 -1
- package/dist/baseline-HsBvw_dk.js +550 -0
- package/dist/baseline-HsBvw_dk.js.map +1 -0
- package/dist/baseline-hG3K85h4.d.ts +125 -0
- package/dist/baseline-hG3K85h4.d.ts.map +1 -0
- package/dist/belief-state/index.d.ts +448 -1205
- package/dist/belief-state/index.d.ts.map +1 -0
- package/dist/belief-state/index.js +1617 -1710
- package/dist/belief-state/index.js.map +1 -1
- package/dist/benchmarks/index.d.ts +2 -894
- package/dist/benchmarks/index.js +2 -59
- package/dist/benchmarks-DviOvUNr.js +754 -0
- package/dist/benchmarks-DviOvUNr.js.map +1 -0
- package/dist/builder-eval/index.d.ts +150 -662
- package/dist/builder-eval/index.d.ts.map +1 -0
- package/dist/builder-eval/index.js +356 -345
- package/dist/builder-eval/index.js.map +1 -1
- package/dist/calibration-CNWWA6K8.js +94 -0
- package/dist/calibration-CNWWA6K8.js.map +1 -0
- package/dist/campaign/index.d.ts +5 -6390
- package/dist/campaign/index.js +3 -212
- package/dist/campaign-CBKZvQ1H.js +3885 -0
- package/dist/campaign-CBKZvQ1H.js.map +1 -0
- package/dist/cli.d.ts +1 -1
- package/dist/cli.js +164 -174
- package/dist/cli.js.map +1 -1
- package/dist/client-C97NMzqi.d.ts +581 -0
- package/dist/client-C97NMzqi.d.ts.map +1 -0
- package/dist/client-CYzbdJOZ.js +637 -0
- package/dist/client-CYzbdJOZ.js.map +1 -0
- package/dist/code-agent-session-BjkMTQ7H.js +1390 -0
- package/dist/code-agent-session-BjkMTQ7H.js.map +1 -0
- package/dist/code-agent-session-DqqgOJaz.d.ts +143 -0
- package/dist/code-agent-session-DqqgOJaz.d.ts.map +1 -0
- package/dist/concurrency-DIxRZF_J.js +85 -0
- package/dist/concurrency-DIxRZF_J.js.map +1 -0
- package/dist/contract/index.d.ts +645 -5605
- package/dist/contract/index.d.ts.map +1 -0
- package/dist/contract/index.js +1730 -1937
- package/dist/contract/index.js.map +1 -1
- package/dist/control.d.ts +3 -1030
- package/dist/control.js +2 -32
- package/dist/cost-ledger-DIgQUFZZ.js +801 -0
- package/dist/cost-ledger-DIgQUFZZ.js.map +1 -0
- package/dist/cost-ledger-Dye6jCgg.d.ts +236 -0
- package/dist/cost-ledger-Dye6jCgg.d.ts.map +1 -0
- package/dist/dataset-BvtnC8Dc.d.ts +115 -0
- package/dist/dataset-BvtnC8Dc.d.ts.map +1 -0
- package/dist/default-registry-C-vFCSEc.js +2579 -0
- package/dist/default-registry-C-vFCSEc.js.map +1 -0
- package/dist/default-registry-CNPo-Vsb.d.ts +540 -0
- package/dist/default-registry-CNPo-Vsb.d.ts.map +1 -0
- package/dist/emitter-CPBAhxum.js +266 -0
- package/dist/emitter-CPBAhxum.js.map +1 -0
- package/dist/emitter-DGQGoLyj.d.ts +113 -0
- package/dist/emitter-DGQGoLyj.d.ts.map +1 -0
- package/dist/errors-8YnH8WlF.js +64 -0
- package/dist/errors-8YnH8WlF.js.map +1 -0
- package/dist/errors-CEk209JS.d.ts +76 -0
- package/dist/errors-CEk209JS.d.ts.map +1 -0
- package/dist/eval-campaign-DEm6c8ru.js +349 -0
- package/dist/eval-campaign-DEm6c8ru.js.map +1 -0
- package/dist/execution-tracks-CpgFPpS5.js +93 -0
- package/dist/execution-tracks-CpgFPpS5.js.map +1 -0
- package/dist/exporters-q9iL-2Jf.js +148 -0
- package/dist/exporters-q9iL-2Jf.js.map +1 -0
- package/dist/extract-usage-BrQ8mCLX.js +155 -0
- package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
- package/dist/failure-cluster-CqcvCcdR.d.ts +59 -0
- package/dist/failure-cluster-CqcvCcdR.d.ts.map +1 -0
- package/dist/feedback-trajectory-CVaeREXV.d.ts +340 -0
- package/dist/feedback-trajectory-CVaeREXV.d.ts.map +1 -0
- package/dist/fuzz.d.ts +320 -646
- package/dist/fuzz.d.ts.map +1 -0
- package/dist/fuzz.js +670 -617
- package/dist/fuzz.js.map +1 -1
- package/dist/hf-dataset-DBJXXoY1.js +763 -0
- package/dist/hf-dataset-DBJXXoY1.js.map +1 -0
- package/dist/hosted/index.d.ts +11 -831
- package/dist/hosted/index.d.ts.map +1 -0
- package/dist/hosted/index.js +2 -37
- package/dist/index-2JJSA6-r2.d.ts +926 -0
- package/dist/index-2JJSA6-r2.d.ts.map +1 -0
- package/dist/index-6N0aYmpW.d.ts +217 -0
- package/dist/index-6N0aYmpW.d.ts.map +1 -0
- package/dist/index-BAvgST_9.d.ts +131 -0
- package/dist/index-BAvgST_9.d.ts.map +1 -0
- package/dist/index-BvnJuTGD.d.ts +68 -0
- package/dist/index-BvnJuTGD.d.ts.map +1 -0
- package/dist/index-C61Wi7yg.d.ts +547 -0
- package/dist/index-C61Wi7yg.d.ts.map +1 -0
- package/dist/index-CAPUUKaM.d.ts +335 -0
- package/dist/index-CAPUUKaM.d.ts.map +1 -0
- package/dist/index-DE5fb3EC.d.ts +2244 -0
- package/dist/index-DE5fb3EC.d.ts.map +1 -0
- package/dist/index-DSC51roc.d.ts +102 -0
- package/dist/index-DSC51roc.d.ts.map +1 -0
- package/dist/index.d.ts +3776 -15120
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +11185 -11191
- package/dist/index.js.map +1 -1
- package/dist/integrity-BzRbCHzi.js +424 -0
- package/dist/integrity-BzRbCHzi.js.map +1 -0
- package/dist/integrity-rmVhXWA7.d.ts +61 -0
- package/dist/integrity-rmVhXWA7.d.ts.map +1 -0
- package/dist/judge-calibration-DFtEMlde.d.ts +146 -0
- package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
- package/dist/ledger-core/index.d.ts +2 -0
- package/dist/ledger-core/index.js +2 -0
- package/dist/ledger-core-DtZz1RG0.js +388 -0
- package/dist/ledger-core-DtZz1RG0.js.map +1 -0
- package/dist/llm-client--GR4JbZE.js +687 -0
- package/dist/llm-client--GR4JbZE.js.map +1 -0
- package/dist/llm-client-B_nIBlYo.d.ts +290 -0
- package/dist/llm-client-B_nIBlYo.d.ts.map +1 -0
- package/dist/matrix/index.d.ts +3 -155
- package/dist/matrix/index.js +2 -8
- package/dist/matrix-BzQnu2S6.js +270 -0
- package/dist/matrix-BzQnu2S6.js.map +1 -0
- package/dist/meta-eval/index.d.ts +4 -1027
- package/dist/meta-eval/index.js +393 -390
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/metrics-C9YY1OcL.js +239 -0
- package/dist/metrics-C9YY1OcL.js.map +1 -0
- package/dist/mint-yN2M2eh0.js +201 -0
- package/dist/mint-yN2M2eh0.js.map +1 -0
- package/dist/multi-layer-verifier-BHY1gWAc.d.ts +138 -0
- package/dist/multi-layer-verifier-BHY1gWAc.d.ts.map +1 -0
- package/dist/multishot/index.d.ts +273 -481
- package/dist/multishot/index.d.ts.map +1 -0
- package/dist/multishot/index.js +595 -548
- package/dist/multishot/index.js.map +1 -1
- package/dist/off-policy-DvgzvtIx.js +220 -0
- package/dist/off-policy-DvgzvtIx.js.map +1 -0
- package/dist/off-policy-mskQw8Mb.d.ts +153 -0
- package/dist/off-policy-mskQw8Mb.d.ts.map +1 -0
- package/dist/openapi.json +1 -1
- package/dist/opencode-sqlite-BGrHeDu3.js +318 -0
- package/dist/opencode-sqlite-BGrHeDu3.js.map +1 -0
- package/dist/outcome-store-BYHIuO0e.d.ts +65 -0
- package/dist/outcome-store-BYHIuO0e.d.ts.map +1 -0
- package/dist/outcome-store-ChBKlTd_.js +75 -0
- package/dist/outcome-store-ChBKlTd_.js.map +1 -0
- package/dist/paired-arms-D9D0wXj2.js +260 -0
- package/dist/paired-arms-D9D0wXj2.js.map +1 -0
- package/dist/pipelines/index.d.ts +95 -532
- package/dist/pipelines/index.d.ts.map +1 -0
- package/dist/pipelines/index.js +497 -478
- package/dist/pipelines/index.js.map +1 -1
- package/dist/pre-registration-DakwTRXk.js +96 -0
- package/dist/pre-registration-DakwTRXk.js.map +1 -0
- package/dist/propose-review-control-LhIWGzHl.js +1458 -0
- package/dist/propose-review-control-LhIWGzHl.js.map +1 -0
- package/dist/query-CJ_DX8vl.d.ts +28 -0
- package/dist/query-CJ_DX8vl.d.ts.map +1 -0
- package/dist/query-Di7eEQ79.js +83 -0
- package/dist/query-Di7eEQ79.js.map +1 -0
- package/dist/raw-provider-sink-BQd7mzyT.js +201 -0
- package/dist/raw-provider-sink-BQd7mzyT.js.map +1 -0
- package/dist/raw-provider-sink-BU29Sh8h.d.ts +134 -0
- package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +1 -0
- package/dist/redact-7Aq1ukl-.js +107 -0
- package/dist/redact-7Aq1ukl-.js.map +1 -0
- package/dist/release-report-Cz9NKH39.js +603 -0
- package/dist/release-report-Cz9NKH39.js.map +1 -0
- package/dist/release-report-mvB2G4_J.d.ts +244 -0
- package/dist/release-report-mvB2G4_J.d.ts.map +1 -0
- package/dist/replay-GnyotH0J.js +1741 -0
- package/dist/replay-GnyotH0J.js.map +1 -0
- package/dist/replay-OoidtG1E.d.ts +749 -0
- package/dist/replay-OoidtG1E.d.ts.map +1 -0
- package/dist/reporting.d.ts +6 -1298
- package/dist/reporting.js +6 -50
- package/dist/researcher-CwTdwXG1.d.ts +314 -0
- package/dist/researcher-CwTdwXG1.d.ts.map +1 -0
- package/dist/reward-hacking-eAnOsynk.d.ts +224 -0
- package/dist/reward-hacking-eAnOsynk.d.ts.map +1 -0
- package/dist/reward-hacking-qipEpKvY.js +596 -0
- package/dist/reward-hacking-qipEpKvY.js.map +1 -0
- package/dist/reward-nw2xZGZG.js +137 -0
- package/dist/reward-nw2xZGZG.js.map +1 -0
- package/dist/rl.d.ts +916 -3596
- package/dist/rl.d.ts.map +1 -0
- package/dist/rl.js +2362 -1751
- package/dist/rl.js.map +1 -1
- package/dist/rolldown-runtime-8H4AJuhK.js +14 -0
- package/dist/rollout/index.d.ts +3 -1048
- package/dist/rollout/index.js +8 -110
- package/dist/rollout-BOYjemfR.js +624 -0
- package/dist/rollout-BOYjemfR.js.map +1 -0
- package/dist/rubric-predictive-validity-B3xmbmS1.js +141 -0
- package/dist/rubric-predictive-validity-B3xmbmS1.js.map +1 -0
- package/dist/rubric-predictive-validity-Ku_clp_1.d.ts +74 -0
- package/dist/rubric-predictive-validity-Ku_clp_1.d.ts.map +1 -0
- package/dist/run-evidence-ClRX_8A9.d.ts +225 -0
- package/dist/run-evidence-ClRX_8A9.d.ts.map +1 -0
- package/dist/run-record-BuoE80Dq.js +467 -0
- package/dist/run-record-BuoE80Dq.js.map +1 -0
- package/dist/run-record-CnZu_gjl.d.ts +357 -0
- package/dist/run-record-CnZu_gjl.d.ts.map +1 -0
- package/dist/run-score-iEEAWiBY.js +41 -0
- package/dist/run-score-iEEAWiBY.js.map +1 -0
- package/dist/runtime-trajectory-1gyaTOoC.js +93 -0
- package/dist/runtime-trajectory-1gyaTOoC.js.map +1 -0
- package/dist/runtime-trajectory-BvSZcCHD.d.ts +50 -0
- package/dist/runtime-trajectory-BvSZcCHD.d.ts.map +1 -0
- package/dist/schema-BtVldJ3T.d.ts +206 -0
- package/dist/schema-BtVldJ3T.d.ts.map +1 -0
- package/dist/schema-C6DW4ZHR.js +821 -0
- package/dist/schema-C6DW4ZHR.js.map +1 -0
- package/dist/schema-CRhEY1SO.js +69 -0
- package/dist/schema-CRhEY1SO.js.map +1 -0
- package/dist/schema-Cef2cFmb.d.ts +408 -0
- package/dist/schema-Cef2cFmb.d.ts.map +1 -0
- package/dist/semantic-concept-judge-B6cWNJ2K.js +725 -0
- package/dist/semantic-concept-judge-B6cWNJ2K.js.map +1 -0
- package/dist/sequential-Br0mAPHA.js +148 -0
- package/dist/sequential-Br0mAPHA.js.map +1 -0
- package/dist/sequential-CYwq6Ff_.d.ts +141 -0
- package/dist/sequential-CYwq6Ff_.d.ts.map +1 -0
- package/dist/series-convergence-CjO2QdRW.js +43 -0
- package/dist/series-convergence-CjO2QdRW.js.map +1 -0
- package/dist/series-convergence-ofsqPWhs.d.ts +35 -0
- package/dist/series-convergence-ofsqPWhs.d.ts.map +1 -0
- package/dist/server-m5D9cvnG.js +1040 -0
- package/dist/server-m5D9cvnG.js.map +1 -0
- package/dist/skill-usage-C_pXm7lP.d.ts +534 -0
- package/dist/skill-usage-C_pXm7lP.d.ts.map +1 -0
- package/dist/skillopt-optimization-method-B7wX7XkF.d.ts +1739 -0
- package/dist/skillopt-optimization-method-B7wX7XkF.d.ts.map +1 -0
- package/dist/skillopt-optimization-method-D4ODwFVV.js +7712 -0
- package/dist/skillopt-optimization-method-D4ODwFVV.js.map +1 -0
- package/dist/statistics-Cmj6nynr.d.ts +514 -0
- package/dist/statistics-Cmj6nynr.d.ts.map +1 -0
- package/dist/statistics-CnnxdpOg.js +1437 -0
- package/dist/statistics-CnnxdpOg.js.map +1 -0
- package/dist/store-CT9YIIve.d.ts +117 -0
- package/dist/store-CT9YIIve.d.ts.map +1 -0
- package/dist/store-CxJry_cs.d.ts +229 -0
- package/dist/store-CxJry_cs.d.ts.map +1 -0
- package/dist/storyboard/index.d.ts +84 -203
- package/dist/storyboard/index.d.ts.map +1 -0
- package/dist/storyboard/index.js +609 -542
- package/dist/storyboard/index.js.map +1 -1
- package/dist/summary-report-BNs5nmXI.js +862 -0
- package/dist/summary-report-BNs5nmXI.js.map +1 -0
- package/dist/summary-report-Cj9gdw4i.d.ts +407 -0
- package/dist/summary-report-Cj9gdw4i.d.ts.map +1 -0
- package/dist/supervisor-run/index.d.ts +2 -849
- package/dist/supervisor-run/index.js +2 -64
- package/dist/supervisor-run-_lnTLM3z.js +1679 -0
- package/dist/supervisor-run-_lnTLM3z.js.map +1 -0
- package/dist/task-failure-attributes-CQZlB3et.js +311 -0
- package/dist/task-failure-attributes-CQZlB3et.js.map +1 -0
- package/dist/test-graded-scenario-BsqWLmPt.js +318 -0
- package/dist/test-graded-scenario-BsqWLmPt.js.map +1 -0
- package/dist/test-graded-scenario-D1TaI2va.d.ts +141 -0
- package/dist/test-graded-scenario-D1TaI2va.d.ts.map +1 -0
- package/dist/tools-BmuN627J.js +1085 -0
- package/dist/tools-BmuN627J.js.map +1 -0
- package/dist/trace-attributes.d.ts +2 -52
- package/dist/trace-attributes.js +131 -61
- package/dist/trace-attributes.js.map +1 -1
- package/dist/traces.d.ts +12 -2365
- package/dist/traces.js +12 -251
- package/dist/trajectory-D_7rLrvE.js +56 -0
- package/dist/trajectory-D_7rLrvE.js.map +1 -0
- package/dist/types-DGsxbAEd.d.ts +387 -0
- package/dist/types-DGsxbAEd.d.ts.map +1 -0
- package/dist/types-k9tZGKUg.d.ts +640 -0
- package/dist/types-k9tZGKUg.d.ts.map +1 -0
- package/dist/verdict-Dps8_okt.d.ts +37 -0
- package/dist/verdict-Dps8_okt.d.ts.map +1 -0
- package/dist/wire/index.d.ts +702 -1174
- package/dist/wire/index.d.ts.map +1 -0
- package/dist/wire/index.js +2 -81
- package/docs/feature-guide.md +1 -1
- package/docs/rollout.md +116 -2
- package/package.json +18 -10
- package/dist/benchmarks/index.js.map +0 -1
- package/dist/campaign/index.js.map +0 -1
- package/dist/chunk-2JX3CFMB.js +0 -695
- package/dist/chunk-2JX3CFMB.js.map +0 -1
- package/dist/chunk-2MKQIFS4.js +0 -183
- package/dist/chunk-2MKQIFS4.js.map +0 -1
- package/dist/chunk-3RF76KTD.js +0 -84
- package/dist/chunk-3RF76KTD.js.map +0 -1
- package/dist/chunk-5DTSBUL2.js +0 -159
- package/dist/chunk-5DTSBUL2.js.map +0 -1
- package/dist/chunk-7ZZMD7UK.js +0 -386
- package/dist/chunk-7ZZMD7UK.js.map +0 -1
- package/dist/chunk-BOD4O7OF.js +0 -40
- package/dist/chunk-BOD4O7OF.js.map +0 -1
- package/dist/chunk-BYT7ELPS.js +0 -1553
- package/dist/chunk-BYT7ELPS.js.map +0 -1
- package/dist/chunk-DJKY2TSY.js +0 -2428
- package/dist/chunk-DJKY2TSY.js.map +0 -1
- package/dist/chunk-DPUHNQLN.js +0 -232
- package/dist/chunk-DPUHNQLN.js.map +0 -1
- package/dist/chunk-DRYIUNWY.js +0 -622
- package/dist/chunk-DRYIUNWY.js.map +0 -1
- package/dist/chunk-EJGRPCO3.js +0 -617
- package/dist/chunk-EJGRPCO3.js.map +0 -1
- package/dist/chunk-EOSZT7PL.js +0 -2001
- package/dist/chunk-EOSZT7PL.js.map +0 -1
- package/dist/chunk-EZJEIH2R.js +0 -1559
- package/dist/chunk-EZJEIH2R.js.map +0 -1
- package/dist/chunk-GGE4NNQT.js +0 -65
- package/dist/chunk-GGE4NNQT.js.map +0 -1
- package/dist/chunk-HHWE3POT.js +0 -94
- package/dist/chunk-HHWE3POT.js.map +0 -1
- package/dist/chunk-IHQDPH7D.js +0 -171
- package/dist/chunk-IHQDPH7D.js.map +0 -1
- package/dist/chunk-JHCHEVET.js +0 -274
- package/dist/chunk-JHCHEVET.js.map +0 -1
- package/dist/chunk-K4DBDHLK.js +0 -158
- package/dist/chunk-K4DBDHLK.js.map +0 -1
- package/dist/chunk-K6N6XJJX.js +0 -306
- package/dist/chunk-K6N6XJJX.js.map +0 -1
- package/dist/chunk-MA6HLL3S.js +0 -65
- package/dist/chunk-MA6HLL3S.js.map +0 -1
- package/dist/chunk-MAZ26DC7.js +0 -99
- package/dist/chunk-MAZ26DC7.js.map +0 -1
- package/dist/chunk-MHELPNRP.js +0 -1212
- package/dist/chunk-MHELPNRP.js.map +0 -1
- package/dist/chunk-NACAGYSY.js +0 -1040
- package/dist/chunk-NACAGYSY.js.map +0 -1
- package/dist/chunk-NKAGIDE2.js +0 -7633
- package/dist/chunk-NKAGIDE2.js.map +0 -1
- package/dist/chunk-NPCTHQIO.js +0 -91
- package/dist/chunk-NPCTHQIO.js.map +0 -1
- package/dist/chunk-NYLOYM6N.js +0 -332
- package/dist/chunk-NYLOYM6N.js.map +0 -1
- package/dist/chunk-ONWEPEDO.js +0 -57
- package/dist/chunk-ONWEPEDO.js.map +0 -1
- package/dist/chunk-P5W7RQKK.js +0 -576
- package/dist/chunk-P5W7RQKK.js.map +0 -1
- package/dist/chunk-P6FYH6K4.js +0 -1161
- package/dist/chunk-P6FYH6K4.js.map +0 -1
- package/dist/chunk-PBE2LOSS.js +0 -669
- package/dist/chunk-PBE2LOSS.js.map +0 -1
- package/dist/chunk-PC4UYEBM.js +0 -166
- package/dist/chunk-PC4UYEBM.js.map +0 -1
- package/dist/chunk-PXE2VKMX.js +0 -140
- package/dist/chunk-PXE2VKMX.js.map +0 -1
- package/dist/chunk-PZ5AY32C.js +0 -10
- package/dist/chunk-PZ5AY32C.js.map +0 -1
- package/dist/chunk-RZTMDUO7.js +0 -49
- package/dist/chunk-RZTMDUO7.js.map +0 -1
- package/dist/chunk-S5YLIBFX.js +0 -136
- package/dist/chunk-S5YLIBFX.js.map +0 -1
- package/dist/chunk-SZLVEKMJ.js +0 -1446
- package/dist/chunk-SZLVEKMJ.js.map +0 -1
- package/dist/chunk-T4SQEITX.js +0 -95
- package/dist/chunk-T4SQEITX.js.map +0 -1
- package/dist/chunk-TBL77AUT.js +0 -355
- package/dist/chunk-TBL77AUT.js.map +0 -1
- package/dist/chunk-TSN7JT6D.js +0 -1646
- package/dist/chunk-TSN7JT6D.js.map +0 -1
- package/dist/chunk-TT4KNT67.js +0 -124
- package/dist/chunk-TT4KNT67.js.map +0 -1
- package/dist/chunk-UB2LOJ6Q.js +0 -4461
- package/dist/chunk-UB2LOJ6Q.js.map +0 -1
- package/dist/chunk-UWZZKKU7.js +0 -237
- package/dist/chunk-UWZZKKU7.js.map +0 -1
- package/dist/chunk-VBQ3CRKH.js +0 -291
- package/dist/chunk-VBQ3CRKH.js.map +0 -1
- package/dist/chunk-VGRCHJON.js +0 -163
- package/dist/chunk-VGRCHJON.js.map +0 -1
- package/dist/chunk-VI2UW6B6.js +0 -162
- package/dist/chunk-VI2UW6B6.js.map +0 -1
- package/dist/chunk-VLOATJQ2.js +0 -908
- package/dist/chunk-VLOATJQ2.js.map +0 -1
- package/dist/chunk-VQMK5FMP.js +0 -247
- package/dist/chunk-VQMK5FMP.js.map +0 -1
- package/dist/chunk-VZSRQ272.js +0 -149
- package/dist/chunk-VZSRQ272.js.map +0 -1
- package/dist/chunk-WGXIEX7P.js +0 -116
- package/dist/chunk-WGXIEX7P.js.map +0 -1
- package/dist/chunk-WS3NZZQQ.js +0 -929
- package/dist/chunk-WS3NZZQQ.js.map +0 -1
- package/dist/chunk-XDWDC2MP.js +0 -695
- package/dist/chunk-XDWDC2MP.js.map +0 -1
- package/dist/chunk-XPRT64IE.js +0 -766
- package/dist/chunk-XPRT64IE.js.map +0 -1
- package/dist/chunk-YJBNWCAA.js +0 -1056
- package/dist/chunk-YJBNWCAA.js.map +0 -1
- package/dist/chunk-ZET2UAYW.js +0 -89
- package/dist/chunk-ZET2UAYW.js.map +0 -1
- package/dist/chunk-ZUUWPZCV.js +0 -752
- package/dist/chunk-ZUUWPZCV.js.map +0 -1
- package/dist/control.js.map +0 -1
- package/dist/hosted/index.js.map +0 -1
- package/dist/matrix/index.js.map +0 -1
- package/dist/reporting.js.map +0 -1
- package/dist/rollout/index.js.map +0 -1
- package/dist/run-campaign-ISHFZ7FJ.js +0 -17
- package/dist/run-campaign-ISHFZ7FJ.js.map +0 -1
- package/dist/supervisor-run/index.js.map +0 -1
- package/dist/traces.js.map +0 -1
- package/dist/wire/index.js.map +0 -1
|
@@ -1,1379 +1,622 @@
|
|
|
1
|
-
|
|
2
|
-
|
|
3
|
-
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
score?: number;
|
|
10
|
-
pass?: boolean;
|
|
11
|
-
failureClass?: FailureClass;
|
|
12
|
-
notes?: string;
|
|
13
|
-
}
|
|
14
|
-
/**
|
|
15
|
-
* Layer — optional classification in a nested build workflow.
|
|
16
|
-
* `builder`: the meta-agent editing a project (e.g. agent-builder Forge chat).
|
|
17
|
-
* `app-build`: sandbox harness that compiled + tested the generated scaffold.
|
|
18
|
-
* `app-runtime`: a run of the generated agent against a domain scenario.
|
|
19
|
-
* `meta`: any meta-eval (judge replay, correlation analysis).
|
|
20
|
-
*/
|
|
21
|
-
type RunLayer = 'builder' | 'app-build' | 'app-runtime' | 'meta' | 'custom';
|
|
22
|
-
interface Run {
|
|
23
|
-
runId: string;
|
|
24
|
-
/**
|
|
25
|
-
* Stable identifier of the scenario being executed.
|
|
26
|
-
*
|
|
27
|
-
* Always populated on the persisted Run — but `TraceEmitter.startRun` accepts
|
|
28
|
-
* input WITHOUT this field, substituting a sensible default
|
|
29
|
-
* (`run.layer ?? run.tags?.['kind'] ?? 'runtime'`) when the caller has no
|
|
30
|
-
* curated scenario to anchor to (runtime / operator / meta-eval runs). This
|
|
31
|
-
* keeps the persisted shape unambiguous for downstream filters + aggregations
|
|
32
|
-
* while removing the boilerplate of inventing placeholder ids at the call site.
|
|
33
|
-
*/
|
|
34
|
-
scenarioId: string;
|
|
35
|
-
variantId?: string;
|
|
36
|
-
datasetVersion?: string;
|
|
37
|
-
/** Git SHA of agent code at run time. */
|
|
38
|
-
codeSha?: string;
|
|
39
|
-
/** Hash of the prompt template + any system prompt. */
|
|
40
|
-
promptSha?: string;
|
|
41
|
-
/** Model id + date + system-prompt hash, concatenated. */
|
|
42
|
-
modelFingerprint?: string;
|
|
43
|
-
seed?: number;
|
|
44
|
-
/** Arbitrary environment markers (shell, docker version, tz). */
|
|
45
|
-
envFingerprint?: Record<string, string>;
|
|
46
|
-
/** Version of the redaction rules applied to this run. */
|
|
47
|
-
redactionVersion?: string;
|
|
48
|
-
/** Parent run in a nested build workflow. A builder run's children are
|
|
49
|
-
* app-build runs; those children are app-runtime runs. */
|
|
50
|
-
parentRunId?: string;
|
|
51
|
-
/** Stable project identifier — groups runs across chats + sessions. */
|
|
52
|
-
projectId?: string;
|
|
53
|
-
/** Chat/conversation identifier within a project. */
|
|
54
|
-
chatId?: string;
|
|
55
|
-
/** Layer classification — hint for aggregation; not enforced. */
|
|
56
|
-
layer?: RunLayer;
|
|
57
|
-
startedAt: number;
|
|
58
|
-
endedAt?: number;
|
|
59
|
-
status: RunStatus;
|
|
60
|
-
outcome?: RunOutcome$1;
|
|
61
|
-
budget?: BudgetSpec;
|
|
62
|
-
/** Free-form labels for downstream grouping. */
|
|
63
|
-
tags?: Record<string, string>;
|
|
64
|
-
}
|
|
65
|
-
type SpanKind = 'agent' | 'llm' | 'tool' | 'retrieval' | 'judge' | 'sandbox' | 'custom';
|
|
66
|
-
type SpanStatus = 'ok' | 'error';
|
|
67
|
-
interface SpanBase {
|
|
68
|
-
spanId: string;
|
|
69
|
-
parentSpanId?: string;
|
|
70
|
-
runId: string;
|
|
71
|
-
kind: SpanKind;
|
|
72
|
-
name: string;
|
|
73
|
-
startedAt: number;
|
|
74
|
-
endedAt?: number;
|
|
75
|
-
status?: SpanStatus;
|
|
76
|
-
error?: string;
|
|
77
|
-
/** Anything not covered by typed fields. Kept deliberately free-form. */
|
|
78
|
-
attributes?: Record<string, unknown>;
|
|
79
|
-
}
|
|
80
|
-
interface Message {
|
|
81
|
-
role: 'system' | 'user' | 'assistant' | 'tool';
|
|
82
|
-
content: string;
|
|
83
|
-
tokens?: number;
|
|
84
|
-
/** Multi-modal content descriptors; blobs themselves live in Artifacts. */
|
|
85
|
-
images?: Array<{
|
|
86
|
-
artifactId?: string;
|
|
87
|
-
url?: string;
|
|
88
|
-
mime?: string;
|
|
89
|
-
}>;
|
|
90
|
-
}
|
|
91
|
-
interface LlmSpan extends SpanBase {
|
|
92
|
-
kind: 'llm';
|
|
93
|
-
model: string;
|
|
94
|
-
messages: Message[];
|
|
95
|
-
output?: string;
|
|
96
|
-
inputTokens?: number;
|
|
97
|
-
/** All generated tokens, including the reasoning subset when present. */
|
|
98
|
-
outputTokens?: number;
|
|
99
|
-
cachedTokens?: number;
|
|
100
|
-
cacheWriteTokens?: number;
|
|
101
|
-
/** Reasoning-token subset of `outputTokens`. */
|
|
102
|
-
reasoningTokens?: number;
|
|
103
|
-
costUsd?: number;
|
|
104
|
-
finishReason?: string;
|
|
105
|
-
}
|
|
106
|
-
interface ToolSpan extends SpanBase {
|
|
107
|
-
kind: 'tool';
|
|
108
|
-
toolName: string;
|
|
109
|
-
args: unknown;
|
|
110
|
-
/** False when the source observed the call but did not capture its arguments. */
|
|
111
|
-
argsCaptured?: boolean;
|
|
112
|
-
result?: unknown;
|
|
113
|
-
latencyMs?: number;
|
|
114
|
-
}
|
|
115
|
-
interface RetrievalSpan extends SpanBase {
|
|
116
|
-
kind: 'retrieval';
|
|
117
|
-
query: string;
|
|
118
|
-
hits: Array<{
|
|
119
|
-
docId: string;
|
|
120
|
-
score: number;
|
|
121
|
-
content?: string;
|
|
122
|
-
}>;
|
|
123
|
-
}
|
|
124
|
-
interface JudgeSpan extends SpanBase {
|
|
125
|
-
kind: 'judge';
|
|
126
|
-
judgeId: string;
|
|
127
|
-
/** Span this judgment applies to. */
|
|
128
|
-
targetSpanId: string;
|
|
129
|
-
dimension: string;
|
|
130
|
-
/** Numeric score (free-range; interpretation up to the judge). */
|
|
131
|
-
score: number;
|
|
132
|
-
rationale?: string;
|
|
133
|
-
evidence?: string;
|
|
134
|
-
}
|
|
135
|
-
interface SandboxSpan extends SpanBase {
|
|
136
|
-
kind: 'sandbox';
|
|
137
|
-
image?: string;
|
|
138
|
-
command?: string;
|
|
139
|
-
exitCode?: number;
|
|
140
|
-
testsTotal?: number;
|
|
141
|
-
testsPassed?: number;
|
|
142
|
-
stdoutHash?: string;
|
|
143
|
-
stderrHash?: string;
|
|
144
|
-
/** Duration in ms; the harness fills this explicitly (endedAt - startedAt may miss setup). */
|
|
145
|
-
wallMs?: number;
|
|
146
|
-
}
|
|
147
|
-
interface GenericSpan extends SpanBase {
|
|
148
|
-
kind: 'agent' | 'custom';
|
|
149
|
-
}
|
|
150
|
-
type Span = LlmSpan | ToolSpan | RetrievalSpan | JudgeSpan | SandboxSpan | GenericSpan;
|
|
151
|
-
type EventKind = 'log' | 'error' | 'budget_decrement' | 'budget_breach' | 'state_mutation' | 'policy_violation' | 'redaction_applied' | 'custom';
|
|
152
|
-
interface TraceEvent {
|
|
153
|
-
eventId: string;
|
|
154
|
-
runId: string;
|
|
155
|
-
spanId?: string;
|
|
156
|
-
kind: EventKind;
|
|
157
|
-
timestamp: number;
|
|
158
|
-
payload: Record<string, unknown>;
|
|
159
|
-
}
|
|
160
|
-
interface BudgetLedgerEntry {
|
|
161
|
-
runId: string;
|
|
162
|
-
dimension: keyof BudgetSpec;
|
|
163
|
-
limit: number;
|
|
164
|
-
consumed: number;
|
|
165
|
-
remaining: number;
|
|
166
|
-
timestamp: number;
|
|
167
|
-
breached: boolean;
|
|
168
|
-
/** Span that triggered this entry, if any. */
|
|
169
|
-
spanId?: string;
|
|
170
|
-
}
|
|
171
|
-
interface Artifact {
|
|
172
|
-
artifactId: string;
|
|
173
|
-
runId: string;
|
|
174
|
-
spanId?: string;
|
|
175
|
-
contentType: string;
|
|
176
|
-
sizeBytes: number;
|
|
177
|
-
/** sha256 in hex. */
|
|
178
|
-
hash: string;
|
|
179
|
-
/** External storage URL (R2, S3, filesystem path). */
|
|
180
|
-
storageUrl?: string;
|
|
181
|
-
/** Inline content for small blobs — keep under ~64KB. */
|
|
182
|
-
inlineContent?: string;
|
|
183
|
-
}
|
|
184
|
-
type FailureClass = 'success' | 'reasoning_error' | 'tool_selection_error' | 'tool_argument_error' | 'tool_recovery_failure' | 'hallucination' | 'instruction_following' | 'safety_refusal_miss' | 'policy_violation' | 'budget_exceeded' | 'format_drift' | 'permission_escalation' | 'pii_leak' | 'cost_overrun' | 'timeout' | 'sandbox_failure' | 'missing_user_data' | 'missing_domain_data' | 'missing_codebase_context' | 'missing_runtime_context' | 'missing_credentials' | 'missing_integration_connection' | 'missing_integration_scope' | 'integration_approval_required' | 'integration_auth_expired' | 'integration_provider_failure' | 'bad_integration_manifest' | 'unsafe_integration_write_denied' | 'stale_external_data' | 'bad_retrieval' | 'insufficient_evidence' | 'contradictory_evidence' | 'ambiguous_user_intent' | 'knowledge_readiness_blocked' | 'unknown';
|
|
185
|
-
|
|
186
|
-
interface RunFilter {
|
|
187
|
-
scenarioId?: string;
|
|
188
|
-
variantId?: string;
|
|
189
|
-
status?: RunStatus;
|
|
190
|
-
since?: number;
|
|
191
|
-
until?: number;
|
|
192
|
-
tag?: {
|
|
193
|
-
key: string;
|
|
194
|
-
value: string;
|
|
195
|
-
};
|
|
196
|
-
parentRunId?: string;
|
|
197
|
-
projectId?: string;
|
|
198
|
-
chatId?: string;
|
|
199
|
-
layer?: RunLayer;
|
|
200
|
-
}
|
|
201
|
-
interface SpanFilter {
|
|
202
|
-
runId?: string;
|
|
203
|
-
parentSpanId?: string;
|
|
204
|
-
kind?: SpanKind;
|
|
205
|
-
name?: string;
|
|
206
|
-
toolName?: string;
|
|
207
|
-
judgeId?: string;
|
|
208
|
-
since?: number;
|
|
209
|
-
until?: number;
|
|
210
|
-
}
|
|
211
|
-
interface EventFilter {
|
|
212
|
-
runId?: string;
|
|
213
|
-
spanId?: string;
|
|
214
|
-
kind?: EventKind;
|
|
215
|
-
since?: number;
|
|
216
|
-
until?: number;
|
|
217
|
-
}
|
|
218
|
-
interface TraceStore {
|
|
219
|
-
appendRun(run: Run): Promise<void>;
|
|
220
|
-
updateRun(runId: string, patch: Partial<Run>): Promise<void>;
|
|
221
|
-
appendSpan(span: Span): Promise<void>;
|
|
222
|
-
updateSpan(spanId: string, patch: Partial<Span>): Promise<void>;
|
|
223
|
-
appendEvent(event: TraceEvent): Promise<void>;
|
|
224
|
-
appendArtifact(artifact: Artifact): Promise<void>;
|
|
225
|
-
appendBudgetEntry(entry: BudgetLedgerEntry): Promise<void>;
|
|
226
|
-
getRun(runId: string): Promise<Run | undefined>;
|
|
227
|
-
listRuns(filter?: RunFilter): Promise<Run[]>;
|
|
228
|
-
spans(filter?: SpanFilter): Promise<Span[]>;
|
|
229
|
-
events(filter?: EventFilter): Promise<TraceEvent[]>;
|
|
230
|
-
budget(runId: string): Promise<BudgetLedgerEntry[]>;
|
|
231
|
-
artifacts(runId: string): Promise<Artifact[]>;
|
|
232
|
-
}
|
|
233
|
-
|
|
234
|
-
/**
|
|
235
|
-
* Calibration curve — binned "if eval says X, what does reality show?"
|
|
236
|
-
*
|
|
237
|
-
* Companion to correlationStudy. Raw correlation is a single number;
|
|
238
|
-
* the calibration curve shows *where* the eval is well-calibrated vs
|
|
239
|
-
* overconfident / underconfident. Buckets the eval metric, computes
|
|
240
|
-
* mean outcome per bucket, reports expected-calibration-error (ECE).
|
|
241
|
-
*/
|
|
242
|
-
|
|
243
|
-
interface CalibrationBin {
|
|
244
|
-
lower: number;
|
|
245
|
-
upper: number;
|
|
246
|
-
n: number;
|
|
247
|
-
evalMean: number;
|
|
248
|
-
outcomeMean: number;
|
|
249
|
-
/** |outcomeMean − evalMean|; contributes to ECE weighted by n/total. */
|
|
250
|
-
gap: number;
|
|
251
|
-
}
|
|
252
|
-
interface CalibrationReport {
|
|
253
|
-
evalMetric: string;
|
|
254
|
-
outcomeMetric: string;
|
|
255
|
-
n: number;
|
|
256
|
-
bins: CalibrationBin[];
|
|
257
|
-
/** Expected Calibration Error — Σ (n_i/N) × |outcomeMean_i − evalMean_i|. */
|
|
258
|
-
ece: number;
|
|
259
|
-
/** Max bin gap — upper bound on miscalibration. */
|
|
260
|
-
maxGap: number;
|
|
261
|
-
}
|
|
262
|
-
|
|
263
|
-
/**
|
|
264
|
-
* Off-policy evaluation primitives.
|
|
265
|
-
*
|
|
266
|
-
* Standard inverse-probability-weighted (IPS), self-normalized
|
|
267
|
-
* importance-weighted (SNIPS), and doubly-robust (DR) estimators for the
|
|
268
|
-
* value of a *target* policy given trajectories collected under a
|
|
269
|
-
* *behavior* policy. This is the canonical RL eval task: "we have last
|
|
270
|
-
* week's runs, we changed the policy — how would the new one do without
|
|
271
|
-
* re-running?"
|
|
272
|
-
*
|
|
273
|
-
* The math here is textbook (Dudík, Langford, Li 2011 for DR; Swaminathan
|
|
274
|
-
* & Joachims 2015 for SNIPS) but the *application* to LLM-agent
|
|
275
|
-
* evaluation needs care:
|
|
276
|
-
*
|
|
277
|
-
* - The "policy" is the (prompt, tool config, model snapshot) triple.
|
|
278
|
-
* Two policies have the same probability over an action *iff* their
|
|
279
|
-
* LLM call would emit the same token with the same probability —
|
|
280
|
-
* which is generally unknowable without the model log-probs.
|
|
281
|
-
* - For LLM agents, propensity scores must be supplied by the caller
|
|
282
|
-
* (logged in the trace, recovered from token log-probs, or estimated
|
|
283
|
-
* via a learned propensity model). We do NOT estimate propensity here.
|
|
284
|
-
* - Doubly-robust requires two outputs from a Q-function: its prediction
|
|
285
|
-
* for the logged action and its expectation under the target policy.
|
|
286
|
-
* Consumers compute these with a tabular estimate, regression fit, or
|
|
287
|
-
* learned reward model before constructing the trajectories.
|
|
288
|
-
*
|
|
289
|
-
* Bias / variance tradeoffs:
|
|
290
|
-
* - IPS: unbiased; high variance for small overlap, infinite variance
|
|
291
|
-
* when target has support outside behavior.
|
|
292
|
-
* - SNIPS: lower variance, slight bias; usually preferred in practice.
|
|
293
|
-
* - DR: doubly-robust — unbiased if either propensity OR Q-function is
|
|
294
|
-
* correct. Lowest practical variance when Q is decent. Use this.
|
|
295
|
-
*
|
|
296
|
-
* Caveat the panel will land: on the LLM-agent setting, propensity scores
|
|
297
|
-
* recovered from token log-probs are noisy, the action space is enormous,
|
|
298
|
-
* and overlap is often poor. These estimators are useful but not magic;
|
|
299
|
-
* complement with `replayCampaign` (exact replay where the request hashes
|
|
300
|
-
* match) for high-confidence answers and OPE for the gap.
|
|
301
|
-
*/
|
|
302
|
-
interface OffPolicyTrajectory {
|
|
303
|
-
/** Stable id, for traceability through the dataset. */
|
|
304
|
-
runId: string;
|
|
305
|
-
/** Reward observed under the behavior policy (the realized outcome). */
|
|
306
|
-
reward: number;
|
|
307
|
-
/**
|
|
308
|
-
* Behavior-policy probability of the action that was taken. For LLM
|
|
309
|
-
* agents this is typically `exp(sum(token_log_probs))` over the chosen
|
|
310
|
-
* trajectory. Must be in (0, 1].
|
|
311
|
-
*/
|
|
312
|
-
behaviorProb: number;
|
|
313
|
-
/**
|
|
314
|
-
* Target-policy probability of the same action. For replay-style
|
|
315
|
-
* counterfactual evaluation this is what the *new* policy would have
|
|
316
|
-
* assigned to the *old* trajectory. Must be in [0, 1].
|
|
317
|
-
*/
|
|
318
|
-
targetProb: number;
|
|
319
|
-
/**
|
|
320
|
-
* Model-based reward prediction for the action selected by the behavior
|
|
321
|
-
* policy: `Q_hat(context, loggedAction)`. Supply this together with
|
|
322
|
-
* `vHatTarget` for contextual-bandit doubly-robust estimation.
|
|
323
|
-
*/
|
|
324
|
-
qHatChosen?: number | null;
|
|
325
|
-
/**
|
|
326
|
-
* Expected model-based reward under the target policy:
|
|
327
|
-
* `sum_action targetPolicy(action | context) * Q_hat(context, action)`.
|
|
328
|
-
* Supply this together with `qHatChosen`. For an honest evaluation, both
|
|
329
|
-
* values must come from a model cross-fitted or trained outside this row.
|
|
330
|
-
*/
|
|
331
|
-
vHatTarget?: number | null;
|
|
332
|
-
/**
|
|
333
|
-
* @deprecated Use `qHatChosen` and `vHatTarget` together. When the new pair
|
|
334
|
-
* is absent, this scalar is used as both terms to preserve existing results.
|
|
335
|
-
* When the new pair is present, this field is ignored.
|
|
336
|
-
*/
|
|
337
|
-
qHat?: number | null;
|
|
338
|
-
}
|
|
339
|
-
interface OffPolicyContributionCounts {
|
|
340
|
-
/** Contributions using the contextual-bandit doubly-robust formula. */
|
|
341
|
-
dr: number;
|
|
342
|
-
/** Contributions using exact IPS because no reward-model estimate was supplied. */
|
|
343
|
-
ipsFallback: number;
|
|
344
|
-
/** Contributions using the deprecated single-scalar formula. */
|
|
345
|
-
legacyScalar: number;
|
|
346
|
-
}
|
|
347
|
-
interface OffPolicyEstimate {
|
|
348
|
-
/** Estimated value of the target policy. */
|
|
349
|
-
value: number;
|
|
350
|
-
/** Standard error of the estimate. */
|
|
351
|
-
standardError: number;
|
|
352
|
-
/** Effective sample size (Kong 1992). Lower = more reliance on a few high-weight samples. */
|
|
353
|
-
effectiveSampleSize: number;
|
|
354
|
-
/** Number of trajectories used. */
|
|
355
|
-
n: number;
|
|
356
|
-
/**
|
|
357
|
-
* Diagnostic: maximum importance weight observed. Large values (>>10x
|
|
358
|
-
* mean) are a red flag — variance is dominated by a few outliers.
|
|
359
|
-
*/
|
|
360
|
-
maxImportanceWeight: number;
|
|
361
|
-
/** Populated by `doublyRobust` to expose which formula each row used. */
|
|
362
|
-
contributionCounts?: OffPolicyContributionCounts;
|
|
363
|
-
}
|
|
364
|
-
interface OffPolicyOptions {
|
|
365
|
-
/**
|
|
366
|
-
* Cap importance weights at this value (Ionides 2008 truncated IS) to
|
|
367
|
-
* trade unbiasedness for variance reduction. Default `Infinity` (no cap).
|
|
368
|
-
* Set e.g. `10` for stable estimates when the policies are close.
|
|
369
|
-
*/
|
|
370
|
-
weightCap?: number;
|
|
371
|
-
/** Reward clipping range. Default `[0, 1]`. */
|
|
372
|
-
rewardClip?: {
|
|
373
|
-
low: number;
|
|
374
|
-
high: number;
|
|
375
|
-
};
|
|
376
|
-
}
|
|
377
|
-
|
|
378
|
-
declare const BELIEF_DECISION_KINDS: readonly ["continue", "verify", "ask", "retry", "stop", "memory-write", "memory-read", "tool-select", "skill-select", "workflow-select", "surface-promote"];
|
|
1
|
+
import { a as RunRecord, s as RunSplitTag } from "../run-record-CnZu_gjl.js";
|
|
2
|
+
import { s as TraceStore } from "../store-CT9YIIve.js";
|
|
3
|
+
import { w as CalibrationReport } from "../index-6N0aYmpW.js";
|
|
4
|
+
import { i as OffPolicyTrajectory, n as OffPolicyEstimate, r as OffPolicyOptions } from "../off-policy-mskQw8Mb.js";
|
|
5
|
+
import { i as CodeAgentSessionMetrics, n as CodeAgentSessionIntakeOptions, t as CodeAgentSessionDiagnostic, v as CodeAgentSessionObservation, y as CodeAgentSessionSource } from "../code-agent-session-DqqgOJaz.js";
|
|
6
|
+
import { a as RuntimeTrajectoryRecord, n as RuntimeTrajectoryEvidenceProjection, t as ProjectRuntimeTrajectoryEvidenceOptions } from "../runtime-trajectory-BvSZcCHD.js";
|
|
7
|
+
//#region src/belief-state/types.d.ts
|
|
8
|
+
declare const BELIEF_DECISION_KINDS: readonly ['continue', 'verify', 'ask', 'retry', 'stop', 'memory-write', 'memory-read', 'tool-select', 'skill-select', 'workflow-select', 'surface-promote'];
|
|
379
9
|
type BeliefDecisionKind = (typeof BELIEF_DECISION_KINDS)[number];
|
|
380
|
-
declare const BELIEF_EVIDENCE_SOURCES: readonly [
|
|
10
|
+
declare const BELIEF_EVIDENCE_SOURCES: readonly ['run', 'span', 'event', 'finding', 'memory', 'knowledge', 'policy'];
|
|
381
11
|
type BeliefEvidenceSource = (typeof BELIEF_EVIDENCE_SOURCES)[number];
|
|
382
|
-
declare const BELIEF_EVIDENCE_QUALITIES: readonly [
|
|
12
|
+
declare const BELIEF_EVIDENCE_QUALITIES: readonly ['direct', 'derived', 'self-reported', 'unverified', 'stale', 'contradicted'];
|
|
383
13
|
type BeliefEvidenceQuality = (typeof BELIEF_EVIDENCE_QUALITIES)[number];
|
|
384
14
|
declare const BELIEF_EVALUATION_CRITERIA: readonly [{
|
|
385
|
-
|
|
386
|
-
|
|
387
|
-
|
|
15
|
+
readonly id: 'capture-integrity';
|
|
16
|
+
readonly label: 'Capture integrity';
|
|
17
|
+
readonly reasonCodes: readonly ['trace-missing', 'run-record-missing', 'backend-integrity-missing'];
|
|
388
18
|
}, {
|
|
389
|
-
|
|
390
|
-
|
|
391
|
-
|
|
19
|
+
readonly id: 'decision-completeness';
|
|
20
|
+
readonly label: 'Decision completeness';
|
|
21
|
+
readonly reasonCodes: readonly ['candidate-actions-missing', 'chosen-action-missing', 'decision-evidence-missing'];
|
|
392
22
|
}, {
|
|
393
|
-
|
|
394
|
-
|
|
395
|
-
|
|
23
|
+
readonly id: 'evidence-quality';
|
|
24
|
+
readonly label: 'Evidence quality';
|
|
25
|
+
readonly reasonCodes: readonly ['evidence-stale', 'evidence-contradictory', 'evidence-unverified', 'evidence-self-reported'];
|
|
396
26
|
}, {
|
|
397
|
-
|
|
398
|
-
|
|
399
|
-
|
|
27
|
+
readonly id: 'outcome-quality';
|
|
28
|
+
readonly label: 'Outcome quality';
|
|
29
|
+
readonly reasonCodes: readonly ['outcome-missing', 'outcome-delayed', 'cost-missing'];
|
|
400
30
|
}, {
|
|
401
|
-
|
|
402
|
-
|
|
403
|
-
|
|
31
|
+
readonly id: 'calibration';
|
|
32
|
+
readonly label: 'Calibration';
|
|
33
|
+
readonly reasonCodes: readonly ['confidence-missing', 'calibration-unsupported', 'calibration-gap-high'];
|
|
404
34
|
}, {
|
|
405
|
-
|
|
406
|
-
|
|
407
|
-
|
|
35
|
+
readonly id: 'accepted-region-risk';
|
|
36
|
+
readonly label: 'Accepted-region risk';
|
|
37
|
+
readonly reasonCodes: readonly ['accepted-error-high', 'coverage-too-low'];
|
|
408
38
|
}, {
|
|
409
|
-
|
|
410
|
-
|
|
411
|
-
|
|
39
|
+
readonly id: 'policy-value';
|
|
40
|
+
readonly label: 'Policy value';
|
|
41
|
+
readonly reasonCodes: readonly ['utility-lift-missing', 'baseline-dominates', 'cost-too-high'];
|
|
412
42
|
}, {
|
|
413
|
-
|
|
414
|
-
|
|
415
|
-
|
|
43
|
+
readonly id: 'ope-support';
|
|
44
|
+
readonly label: 'OPE support';
|
|
45
|
+
readonly reasonCodes: readonly ['behavior-propensity-missing', 'behavior-propensity-invalid', 'target-propensity-missing', 'target-propensity-invalid', 'effective-sample-size-low', 'importance-weight-high'];
|
|
416
46
|
}, {
|
|
417
|
-
|
|
418
|
-
|
|
419
|
-
|
|
47
|
+
readonly id: 'memory-health';
|
|
48
|
+
readonly label: 'Memory health';
|
|
49
|
+
readonly reasonCodes: readonly ['memory-stale', 'memory-poisoning-risk', 'context-bloat', 'memory-write-unverified'];
|
|
420
50
|
}, {
|
|
421
|
-
|
|
422
|
-
|
|
423
|
-
|
|
51
|
+
readonly id: 'surface-attribution';
|
|
52
|
+
readonly label: 'Surface attribution';
|
|
53
|
+
readonly reasonCodes: readonly ['surface-claim-unsupported', 'causal-attribution-missing'];
|
|
424
54
|
}, {
|
|
425
|
-
|
|
426
|
-
|
|
427
|
-
|
|
55
|
+
readonly id: 'generalization';
|
|
56
|
+
readonly label: 'Generalization';
|
|
57
|
+
readonly reasonCodes: readonly ['split-missing', 'holdout-regression', 'task-family-coverage-low', 'leakage-risk'];
|
|
428
58
|
}, {
|
|
429
|
-
|
|
430
|
-
|
|
431
|
-
|
|
59
|
+
readonly id: 'promotion';
|
|
60
|
+
readonly label: 'Promotion';
|
|
61
|
+
readonly reasonCodes: readonly ['negative-control-failed', 'promotion-gate-failed', 'human-review-required'];
|
|
432
62
|
}];
|
|
433
63
|
type BeliefEvaluationCriterionId = (typeof BELIEF_EVALUATION_CRITERIA)[number]['id'];
|
|
434
64
|
type BeliefDecisionReasonCode = (typeof BELIEF_EVALUATION_CRITERIA)[number]['reasonCodes'][number];
|
|
435
65
|
interface BeliefDecisionReason {
|
|
436
|
-
|
|
437
|
-
|
|
438
|
-
|
|
439
|
-
|
|
440
|
-
|
|
66
|
+
code: BeliefDecisionReasonCode;
|
|
67
|
+
criterion?: BeliefEvaluationCriterionId;
|
|
68
|
+
detail?: string;
|
|
69
|
+
evidenceIds?: string[];
|
|
70
|
+
metadata?: Record<string, unknown>;
|
|
441
71
|
}
|
|
442
72
|
declare function isBeliefDecisionKind(value: unknown): value is BeliefDecisionKind;
|
|
443
73
|
declare function isBeliefEvidenceSource(value: unknown): value is BeliefEvidenceSource;
|
|
444
74
|
interface BeliefEvidenceRef {
|
|
445
|
-
|
|
446
|
-
|
|
447
|
-
|
|
448
|
-
|
|
449
|
-
|
|
450
|
-
|
|
451
|
-
|
|
452
|
-
|
|
453
|
-
|
|
75
|
+
source: BeliefEvidenceSource;
|
|
76
|
+
id: string;
|
|
77
|
+
runId?: string;
|
|
78
|
+
spanId?: string;
|
|
79
|
+
eventId?: string;
|
|
80
|
+
detail?: string;
|
|
81
|
+
quality?: BeliefEvidenceQuality;
|
|
82
|
+
observedAt?: string;
|
|
83
|
+
metadata?: Record<string, unknown>;
|
|
454
84
|
}
|
|
455
85
|
interface BeliefDecisionOutcome {
|
|
456
|
-
|
|
457
|
-
|
|
458
|
-
|
|
459
|
-
|
|
460
|
-
|
|
461
|
-
|
|
86
|
+
success?: boolean;
|
|
87
|
+
score?: number;
|
|
88
|
+
reward?: number;
|
|
89
|
+
costUsd?: number;
|
|
90
|
+
observedAt?: string;
|
|
91
|
+
metadata?: Record<string, unknown>;
|
|
462
92
|
}
|
|
463
93
|
interface BeliefDecisionPoint {
|
|
464
|
-
|
|
465
|
-
|
|
466
|
-
|
|
467
|
-
|
|
468
|
-
|
|
469
|
-
|
|
470
|
-
|
|
471
|
-
|
|
472
|
-
|
|
473
|
-
|
|
474
|
-
|
|
475
|
-
|
|
476
|
-
|
|
477
|
-
|
|
478
|
-
|
|
479
|
-
|
|
480
|
-
|
|
481
|
-
reasons?: BeliefDecisionReason[];
|
|
482
|
-
metadata?: Record<string, unknown>;
|
|
94
|
+
id: string;
|
|
95
|
+
runId: string;
|
|
96
|
+
scenarioId?: string;
|
|
97
|
+
stepIndex: number;
|
|
98
|
+
kind: BeliefDecisionKind;
|
|
99
|
+
chosenAction: string;
|
|
100
|
+
candidateActions?: string[];
|
|
101
|
+
confidence?: number;
|
|
102
|
+
behaviorProb?: number;
|
|
103
|
+
targetProb?: number;
|
|
104
|
+
qHatChosen?: number | null;
|
|
105
|
+
vHatTarget?: number | null;
|
|
106
|
+
costUsd?: number;
|
|
107
|
+
evidence: BeliefEvidenceRef[];
|
|
108
|
+
outcome?: BeliefDecisionOutcome;
|
|
109
|
+
reasons?: BeliefDecisionReason[];
|
|
110
|
+
metadata?: Record<string, unknown>;
|
|
483
111
|
}
|
|
484
112
|
interface BeliefDecisionExtractionDiagnostic {
|
|
485
|
-
|
|
486
|
-
|
|
487
|
-
|
|
488
|
-
|
|
113
|
+
runId: string;
|
|
114
|
+
eventId?: string;
|
|
115
|
+
severity: 'info' | 'warning' | 'error';
|
|
116
|
+
reason: string;
|
|
489
117
|
}
|
|
490
118
|
interface BeliefDecisionExtractionReport {
|
|
491
|
-
|
|
492
|
-
|
|
119
|
+
decisions: BeliefDecisionPoint[];
|
|
120
|
+
diagnostics: BeliefDecisionExtractionDiagnostic[];
|
|
493
121
|
}
|
|
494
122
|
type BeliefPolicyAction = 'accept' | 'defer' | 'verify' | 'ask' | 'retry' | 'stop';
|
|
495
123
|
interface BeliefPolicyDecision {
|
|
496
|
-
|
|
497
|
-
|
|
498
|
-
|
|
499
|
-
|
|
500
|
-
|
|
501
|
-
|
|
502
|
-
|
|
503
|
-
reason?: string;
|
|
504
|
-
reasons?: BeliefDecisionReason[];
|
|
124
|
+
action: BeliefPolicyAction;
|
|
125
|
+
confidence?: number;
|
|
126
|
+
targetProb?: number;
|
|
127
|
+
qHatChosen?: number | null;
|
|
128
|
+
vHatTarget?: number | null;
|
|
129
|
+
reason?: string;
|
|
130
|
+
reasons?: BeliefDecisionReason[];
|
|
505
131
|
}
|
|
506
132
|
interface BeliefSelectivePolicy {
|
|
507
|
-
|
|
508
|
-
|
|
133
|
+
id: string;
|
|
134
|
+
decide(point: BeliefDecisionPoint): BeliefPolicyDecision;
|
|
509
135
|
}
|
|
510
136
|
interface BeliefOpeTargetPolicy {
|
|
511
|
-
|
|
512
|
-
|
|
513
|
-
|
|
514
|
-
|
|
515
|
-
/** @deprecated Use `qHatChosenOf` and `vHatTargetOf` together. */
|
|
516
|
-
qHatOf?(point: BeliefDecisionPoint): number | null | undefined;
|
|
137
|
+
id: string;
|
|
138
|
+
targetProbOf(point: BeliefDecisionPoint): number | null | undefined;
|
|
139
|
+
qHatChosenOf?(point: BeliefDecisionPoint): number | null | undefined;
|
|
140
|
+
vHatTargetOf?(point: BeliefDecisionPoint): number | null | undefined;
|
|
517
141
|
}
|
|
518
142
|
interface BeliefUtilityOptions {
|
|
519
|
-
|
|
520
|
-
|
|
521
|
-
|
|
522
|
-
|
|
523
|
-
|
|
524
|
-
|
|
525
|
-
|
|
526
|
-
|
|
143
|
+
successUtility?: number;
|
|
144
|
+
failureUtility?: number;
|
|
145
|
+
deferUtility?: number;
|
|
146
|
+
verifyCost?: number;
|
|
147
|
+
askCost?: number;
|
|
148
|
+
retryCost?: number;
|
|
149
|
+
stopUtility?: number;
|
|
150
|
+
costWeight?: number;
|
|
527
151
|
}
|
|
528
152
|
interface BeliefSelectivePolicyMetrics {
|
|
529
|
-
|
|
530
|
-
|
|
531
|
-
|
|
532
|
-
|
|
533
|
-
|
|
534
|
-
|
|
535
|
-
|
|
536
|
-
|
|
537
|
-
|
|
538
|
-
|
|
539
|
-
|
|
540
|
-
|
|
541
|
-
|
|
542
|
-
|
|
543
|
-
|
|
544
|
-
|
|
545
|
-
|
|
153
|
+
policyId: string;
|
|
154
|
+
n: number;
|
|
155
|
+
accepted: number;
|
|
156
|
+
rejected: number;
|
|
157
|
+
coverage: number;
|
|
158
|
+
acceptedErrorRate: number;
|
|
159
|
+
baselineUtility: number;
|
|
160
|
+
policyUtility: number;
|
|
161
|
+
utilityDelta: number;
|
|
162
|
+
utilityCi95: {
|
|
163
|
+
mean: number;
|
|
164
|
+
lower: number;
|
|
165
|
+
upper: number;
|
|
166
|
+
};
|
|
167
|
+
rejectedMeanReward: number | null;
|
|
168
|
+
recommendation: 'ship' | 'hold' | 'need_more_data';
|
|
169
|
+
reasons: string[];
|
|
546
170
|
}
|
|
547
171
|
interface BeliefOpeSupportDiagnostics {
|
|
548
|
-
|
|
549
|
-
|
|
550
|
-
|
|
551
|
-
|
|
552
|
-
|
|
553
|
-
|
|
554
|
-
|
|
172
|
+
supported: boolean;
|
|
173
|
+
n: number;
|
|
174
|
+
dropped: number;
|
|
175
|
+
effectiveSampleSize: number;
|
|
176
|
+
effectiveSampleRatio: number;
|
|
177
|
+
maxImportanceWeight: number;
|
|
178
|
+
reasons: string[];
|
|
555
179
|
}
|
|
556
180
|
interface BeliefOpeReport {
|
|
557
|
-
|
|
558
|
-
|
|
559
|
-
|
|
560
|
-
|
|
561
|
-
|
|
181
|
+
targetPolicyId: string;
|
|
182
|
+
ips: OffPolicyEstimate;
|
|
183
|
+
snips: OffPolicyEstimate;
|
|
184
|
+
dr: OffPolicyEstimate;
|
|
185
|
+
support: BeliefOpeSupportDiagnostics;
|
|
562
186
|
}
|
|
563
187
|
type BeliefEvaluationStatus = 'ship' | 'hold' | 'need_more_data';
|
|
564
188
|
type BeliefCalibrationStatus = 'supported' | 'unsupported';
|
|
565
189
|
type BeliefOpeStatus = 'supported' | 'unsupported' | 'not_requested';
|
|
566
190
|
interface BeliefPolicyEvaluationReport {
|
|
567
|
-
|
|
568
|
-
|
|
569
|
-
|
|
570
|
-
|
|
571
|
-
|
|
572
|
-
|
|
573
|
-
|
|
574
|
-
|
|
575
|
-
|
|
576
|
-
|
|
577
|
-
|
|
578
|
-
}
|
|
579
|
-
|
|
191
|
+
policyId: string;
|
|
192
|
+
n: number;
|
|
193
|
+
status: BeliefEvaluationStatus;
|
|
194
|
+
selectiveStatus: BeliefEvaluationStatus;
|
|
195
|
+
calibrationStatus: BeliefCalibrationStatus;
|
|
196
|
+
opeStatus: BeliefOpeStatus;
|
|
197
|
+
opeTargetPolicyId?: string;
|
|
198
|
+
selective: BeliefSelectivePolicyMetrics;
|
|
199
|
+
calibration?: CalibrationReport;
|
|
200
|
+
ope?: BeliefOpeReport;
|
|
201
|
+
diagnostics: string[];
|
|
202
|
+
}
|
|
203
|
+
//#endregion
|
|
204
|
+
//#region src/belief-state/calibration.d.ts
|
|
580
205
|
type BeliefCalibrationRegion = 'all' | 'accepted' | 'rejected';
|
|
581
206
|
interface BeliefCalibrationOptions {
|
|
582
|
-
|
|
583
|
-
|
|
584
|
-
|
|
585
|
-
|
|
207
|
+
bins?: number;
|
|
208
|
+
minPairs?: number;
|
|
209
|
+
policy?: BeliefSelectivePolicy;
|
|
210
|
+
region?: BeliefCalibrationRegion;
|
|
586
211
|
}
|
|
587
212
|
declare function calibrateBeliefDecisions(points: BeliefDecisionPoint[], options?: BeliefCalibrationOptions): CalibrationReport | null;
|
|
588
|
-
|
|
589
|
-
|
|
590
|
-
type AgentProfileDimensionValue = string | number | boolean | null;
|
|
591
|
-
interface AgentProfileSource {
|
|
592
|
-
/** Runtime/profile contract being fingerprinted, e.g. `agent-interface-profile`. */
|
|
593
|
-
kind: string;
|
|
594
|
-
/** sha256 over the canonical source profile object. */
|
|
595
|
-
hash: string;
|
|
596
|
-
}
|
|
597
|
-
interface AgentProfileHarness {
|
|
598
|
-
id: string;
|
|
599
|
-
version?: string;
|
|
600
|
-
hash?: string;
|
|
601
|
-
}
|
|
602
|
-
interface AgentProfileCell {
|
|
603
|
-
schemaVersion: AgentProfileCellSchemaVersion;
|
|
604
|
-
cellId: string;
|
|
605
|
-
profileId: string;
|
|
606
|
-
sourceProfile: AgentProfileSource;
|
|
607
|
-
harness?: AgentProfileHarness;
|
|
608
|
-
model?: string;
|
|
609
|
-
promptHash?: string;
|
|
610
|
-
dimensions?: Record<string, AgentProfileDimensionValue>;
|
|
611
|
-
}
|
|
612
|
-
|
|
613
|
-
/**
|
|
614
|
-
* Paper-grade RunRecord schema + runtime validator.
|
|
615
|
-
*
|
|
616
|
-
* Every run that participates in a promotion gate, paper table, or
|
|
617
|
-
* researcher loop SHOULD be recorded as a `RunRecord`. The mandatory
|
|
618
|
-
* fields are exactly those the paper "Two Loops, Three Roles" requires
|
|
619
|
-
* for reproducibility: who/what/when/cost/seed/hash, plus the search vs
|
|
620
|
-
* holdout split tag. A task score is optional because execution-only records
|
|
621
|
-
* must preserve missing labels instead of converting errors into zero quality.
|
|
622
|
-
*
|
|
623
|
-
* This is intentionally NOT a replacement for the rich `Run` /
|
|
624
|
-
* `ProposeReviewReport` / `ScenarioResult` types already in the
|
|
625
|
-
* package. Those are runtime structures with full provenance. A
|
|
626
|
-
* `RunRecord` is the analysis-time projection — the JSON-friendly
|
|
627
|
-
* row you'd put in a parquet file or paste into a notebook.
|
|
628
|
-
*
|
|
629
|
-
* Validate at the boundary:
|
|
630
|
-
*
|
|
631
|
-
* const rec = validateRunRecord(rawJson) // throws on missing
|
|
632
|
-
* const ok = isRunRecord(rawJson) // boolean check
|
|
633
|
-
* const rec = parseRunRecordSafe(rawJson) // { ok, value | error }
|
|
634
|
-
*
|
|
635
|
-
* The validator runs in pure TS — zod is intentionally NOT a
|
|
636
|
-
* dependency. Round-trip tested in `tests/run-record.test.ts`.
|
|
637
|
-
*/
|
|
638
|
-
|
|
639
|
-
/** Search/dev/holdout split tag. 'search' is the paper-grade alias for the
|
|
640
|
-
* combined train+test pool that the optimizer is allowed to read. */
|
|
641
|
-
type RunSplitTag = 'search' | 'dev' | 'holdout';
|
|
642
|
-
/**
|
|
643
|
-
* Explicit execution-lifecycle result for a run.
|
|
644
|
-
*
|
|
645
|
-
* This is separate from task quality (`outcome`) and failure classification.
|
|
646
|
-
* Producers set it only from root-run or process evidence.
|
|
647
|
-
*/
|
|
648
|
-
type RunTerminalOutcome = 'succeeded' | 'failed' | 'cancelled' | 'incomplete' | 'unknown';
|
|
649
|
-
interface RunTokenUsage {
|
|
650
|
-
input: number;
|
|
651
|
-
/** All generated tokens charged as output, including reasoning tokens. */
|
|
652
|
-
output: number;
|
|
653
|
-
/** Reasoning-token subset of `output`, when the provider reports it. */
|
|
654
|
-
reasoning?: number;
|
|
655
|
-
/** Prompt tokens served from a provider cache. */
|
|
656
|
-
cached?: number;
|
|
657
|
-
/** Prompt tokens written into a provider cache. */
|
|
658
|
-
cacheWrite?: number;
|
|
659
|
-
}
|
|
660
|
-
/**
|
|
661
|
-
* How a run's USD amount was obtained.
|
|
662
|
-
*/
|
|
663
|
-
type RunCostProvenance = {
|
|
664
|
-
kind: 'observed';
|
|
665
|
-
usd: number;
|
|
666
|
-
} | {
|
|
667
|
-
kind: 'estimated';
|
|
668
|
-
usd: number;
|
|
669
|
-
} | {
|
|
670
|
-
kind: 'uncaptured';
|
|
671
|
-
usd: null;
|
|
672
|
-
};
|
|
673
|
-
interface RunJudgeMetadata {
|
|
674
|
-
model: string;
|
|
675
|
-
promptVersion: string;
|
|
676
|
-
/** [0,1] confidence the judge declared. Constant judge confidence
|
|
677
|
-
* across many runs is a fallback signal (see `canary.ts`). */
|
|
678
|
-
confidence: number;
|
|
679
|
-
/** True if the judge degraded to a fallback path (rules-only,
|
|
680
|
-
* prior-call cache, etc.). The canary uses this to alert. */
|
|
681
|
-
fallback: boolean;
|
|
682
|
-
}
|
|
683
|
-
/**
|
|
684
|
-
* Per-judge / per-dimension breakdown for runs scored by an ensemble of
|
|
685
|
-
* judges over a multi-dimensional rubric.
|
|
686
|
-
*
|
|
687
|
-
* The collapsed `outcome.searchScore` / `holdoutScore` carries the
|
|
688
|
-
* composite the gate uses. The full breakdown belongs here so consumers
|
|
689
|
-
* can answer "which judge disagreed?", "which dimension dragged the
|
|
690
|
-
* composite down?", and "did half the panel fail?" without re-running.
|
|
691
|
-
*
|
|
692
|
-
* `perJudge[judgeId][dim]` is the canonical source; `perDimMean` and
|
|
693
|
-
* `composite` are convenience projections — derivable but precomputed so
|
|
694
|
-
* downstream IRR primitives (`interRaterReliability`,
|
|
695
|
-
* `corpusInterRaterAgreement`) and reporters don't pay the same
|
|
696
|
-
* aggregation twice.
|
|
697
|
-
*
|
|
698
|
-
* Fail-loud discipline: judges that errored out land in `failedJudges`
|
|
699
|
-
* by id. A missing key in `perJudge` is ambiguous (silent zero vs not
|
|
700
|
-
* run); the explicit list makes a partial-failure recorded as such.
|
|
701
|
-
*/
|
|
702
|
-
interface JudgeScoresRecord {
|
|
703
|
-
/** Per-judge per-dimension scores. `{ "kimi-k2.6": { helpfulness: 0.8, clarity: 0.7 }, ... }`. */
|
|
704
|
-
perJudge: Record<string, Record<string, number>>;
|
|
705
|
-
/** Per-dim mean across judges. Convenience — derivable from `perJudge`. */
|
|
706
|
-
perDimMean: Record<string, number>;
|
|
707
|
-
/** Composite mean across successful judges. Mirrors the task score only
|
|
708
|
-
* when `failedJudges` is empty. */
|
|
709
|
-
composite: number;
|
|
710
|
-
/** Judges that errored or returned an unparseable verdict. Recorded
|
|
711
|
-
* by id (e.g. `['glm-5.1']`) so a partial-failure case is explicit,
|
|
712
|
-
* not inferred from missing keys in `perJudge`. */
|
|
713
|
-
failedJudges?: string[];
|
|
714
|
-
/** Free-form notes the judges emitted (joined across judges or
|
|
715
|
-
* first-judge only — consumer's choice). */
|
|
716
|
-
notes?: string;
|
|
717
|
-
}
|
|
718
|
-
interface RunOutcome {
|
|
719
|
-
/** Score on the search/optimization split. Optional for holdout-only and
|
|
720
|
-
* execution-only records. */
|
|
721
|
-
searchScore?: number;
|
|
722
|
-
/** Score on the held-out split. Optional for search-only and execution-only
|
|
723
|
-
* records. When both scores are absent, the run is explicitly unlabeled. */
|
|
724
|
-
holdoutScore?: number;
|
|
725
|
-
/** Bag of any other metric the run produced — judge dimensions,
|
|
726
|
-
* pass/fail counters, latency stats, etc. Numeric only — keeps
|
|
727
|
-
* reporters honest. */
|
|
728
|
-
raw: Record<string, number>;
|
|
729
|
-
/** Per-judge / per-dim breakdown. Consumers writing ensemble
|
|
730
|
-
* judgements populate this; substrate primitives like
|
|
731
|
-
* `interRaterReliability` and `corpusInterRaterAgreement` accept
|
|
732
|
-
* these records as input. Optional — single-judge or scalar-only
|
|
733
|
-
* runs leave it unset. */
|
|
734
|
-
judgeScores?: JudgeScoresRecord;
|
|
735
|
-
/** Authenticity / realness verdict — did the run build the REAL thing on the
|
|
736
|
-
* intended infra, or fake it (see `./authenticity`)? Optional: only domains
|
|
737
|
-
* with an authenticity config populate it. Carried in the corpus so the
|
|
738
|
-
* flywheel / off-policy learning can optimize for real completion, not gamed
|
|
739
|
-
* pass-rate. `score` is 0-1; `gated` is the anti-Goodhart flag — a gated run
|
|
740
|
-
* must not count as a real success regardless of `score`. */
|
|
741
|
-
realness?: {
|
|
742
|
-
score: number;
|
|
743
|
-
gated: boolean;
|
|
744
|
-
reason?: string;
|
|
745
|
-
};
|
|
746
|
-
}
|
|
747
|
-
/**
|
|
748
|
-
* Mandatory paper-grade fields for a single evaluation run. Optional
|
|
749
|
-
* fields are extension points; mandatory fields throw if missing.
|
|
750
|
-
*
|
|
751
|
-
* Hash discipline:
|
|
752
|
-
* - `promptHash` is the sha256 of the EFFECTIVE prompt sent to the
|
|
753
|
-
* model (after any steering bundle merge).
|
|
754
|
-
* - `configHash` is the sha256 of the effective run config (model,
|
|
755
|
-
* temperature, tools, judges, splits). The pair (promptHash,
|
|
756
|
-
* configHash) uniquely identifies an experiment cell.
|
|
757
|
-
*
|
|
758
|
-
* Model snapshot discipline:
|
|
759
|
-
* - `model` MUST encode a snapshot version. Bare aliases like
|
|
760
|
-
* `claude-sonnet-4` or `gpt-4o` are banned — they remap silently.
|
|
761
|
-
* Use `claude-sonnet-4-6@2025-04-15` or `gpt-4o-2024-11-20`.
|
|
762
|
-
*/
|
|
763
|
-
interface RunRecord {
|
|
764
|
-
/** UUID for the run. */
|
|
765
|
-
runId: string;
|
|
766
|
-
/** Logical experiment grouping (a treatment vs a baseline within
|
|
767
|
-
* the same sweep should share `experimentId`). */
|
|
768
|
-
experimentId: string;
|
|
769
|
-
/** Stable identifier for the candidate (variant) being run. The
|
|
770
|
-
* promotion gate compares two `candidateId`s on matched items. */
|
|
771
|
-
candidateId: string;
|
|
772
|
-
/** RNG seed for the run. Always recorded — silent re-seeding is
|
|
773
|
-
* the most common cause of non-reproducible numbers. */
|
|
774
|
-
seed: number;
|
|
775
|
-
/** Model identifier WITH snapshot version. */
|
|
776
|
-
model: string;
|
|
777
|
-
/** sha256 of the effective prompt (post-steering). */
|
|
778
|
-
promptHash: string;
|
|
779
|
-
/** sha256 of the effective config. */
|
|
780
|
-
configHash: string;
|
|
781
|
-
/** Git SHA the harness was run from. */
|
|
782
|
-
commitSha: string;
|
|
783
|
-
/** End-to-end wall-clock duration in milliseconds. */
|
|
784
|
-
wallMs: number;
|
|
785
|
-
/** Time spent queued before execution started, if known. */
|
|
786
|
-
queueMs?: number;
|
|
787
|
-
/** Total USD cost, or null when the producer could not capture one. */
|
|
788
|
-
costUsd: number | null;
|
|
789
|
-
/** Whether `costUsd` came from billing data, a price calculation, or is unavailable. */
|
|
790
|
-
costProvenance: RunCostProvenance;
|
|
791
|
-
/** Token usage breakdown. */
|
|
792
|
-
tokenUsage: RunTokenUsage;
|
|
793
|
-
/** Root-run or process terminal result. Never inferred from a child span. */
|
|
794
|
-
terminalOutcome: RunTerminalOutcome;
|
|
795
|
-
/** Root-run or process failure reason. Valid only for a failed, cancelled,
|
|
796
|
-
* or incomplete terminal result; never populated from a child span. */
|
|
797
|
-
terminalFailureReason?: string;
|
|
798
|
-
/** Judge-side metadata, if a judge was used. */
|
|
799
|
-
judgeMetadata?: RunJudgeMetadata;
|
|
800
|
-
/** Per-split scores + raw bag. */
|
|
801
|
-
outcome: RunOutcome;
|
|
802
|
-
/** Canonical task-failure class drawn from the shared
|
|
803
|
-
* `FAILURE_CLASSES` taxonomy. Producers set it only from task-result
|
|
804
|
-
* evidence. Execution errors belong in
|
|
805
|
-
* `outcome.raw.execution_error_count`. */
|
|
806
|
-
failureClass?: FailureClass;
|
|
807
|
-
/** Free-form task-failure detail scoped under a non-success
|
|
808
|
-
* `failureClass`. It is invalid without that class. */
|
|
809
|
-
failureMode?: string;
|
|
810
|
-
/** Which split this run was drawn from. */
|
|
811
|
-
splitTag: RunSplitTag;
|
|
812
|
-
/**
|
|
813
|
-
* Stable scenario identifier the run observed or was scored against.
|
|
814
|
-
* Comparison primitives match this identity rather than input order.
|
|
815
|
-
*/
|
|
816
|
-
scenarioId: string;
|
|
817
|
-
/**
|
|
818
|
-
* Canonical identity for the agent profile cell that produced this row:
|
|
819
|
-
* profile artifact hash plus optional harness/model/prompt/reporting
|
|
820
|
-
* dimensions. Use `agentProfile.cellId` to group persona sweeps and
|
|
821
|
-
* longitudinal reports by the complete source profile, not by a loose
|
|
822
|
-
* candidate label or opaque config hash.
|
|
823
|
-
*/
|
|
824
|
-
agentProfile?: AgentProfileCell;
|
|
825
|
-
}
|
|
826
|
-
|
|
827
|
-
type CodeAgentSessionSource = 'codex' | 'claude-code' | 'opencode' | 'kimi-code' | 'pi';
|
|
828
|
-
type CodeAgentSessionTerminalStatus = 'completed' | 'failed' | 'unknown';
|
|
829
|
-
type CodeAgentSessionActionKind = 'tool' | 'patch' | 'terminal' | 'graph-completion';
|
|
830
|
-
type CodeAgentSessionActionSurface = 'tool' | 'mcp' | 'subagent' | 'skill' | 'hook' | 'web' | 'code';
|
|
831
|
-
type CodeAgentSessionActionStatus = 'started' | 'completed' | 'failed' | 'unknown';
|
|
832
|
-
interface CodeAgentSessionExecutionReceipt {
|
|
833
|
-
exitCode: number;
|
|
834
|
-
startedAtMs?: number;
|
|
835
|
-
completedAtMs?: number;
|
|
836
|
-
}
|
|
837
|
-
interface CodeAgentSessionAction {
|
|
838
|
-
id: string;
|
|
839
|
-
stepIndex: number;
|
|
840
|
-
kind: CodeAgentSessionActionKind;
|
|
841
|
-
surface: CodeAgentSessionActionSurface;
|
|
842
|
-
name: string;
|
|
843
|
-
status: CodeAgentSessionActionStatus;
|
|
844
|
-
timestampMs?: number;
|
|
845
|
-
costUsd?: number;
|
|
846
|
-
metadata: Record<string, unknown>;
|
|
847
|
-
}
|
|
848
|
-
interface CodeAgentSessionObservation {
|
|
849
|
-
source: CodeAgentSessionSource;
|
|
850
|
-
sessionId: string;
|
|
851
|
-
finalText: string | null;
|
|
852
|
-
terminal: {
|
|
853
|
-
status: CodeAgentSessionTerminalStatus;
|
|
854
|
-
explicit: boolean;
|
|
855
|
-
};
|
|
856
|
-
actions: CodeAgentSessionAction[];
|
|
857
|
-
}
|
|
858
|
-
|
|
859
|
-
interface CodeAgentSessionMetrics {
|
|
860
|
-
entries: number;
|
|
861
|
-
userMessages: number;
|
|
862
|
-
assistantMessages: number;
|
|
863
|
-
reasoningItems: number;
|
|
864
|
-
toolCalls: number;
|
|
865
|
-
toolOutputs: number;
|
|
866
|
-
toolErrors: number;
|
|
867
|
-
unclassifiedErrors: number;
|
|
868
|
-
patchAttempts: number;
|
|
869
|
-
patchSuccesses: number;
|
|
870
|
-
patchFailures: number;
|
|
871
|
-
turnsStarted: number;
|
|
872
|
-
turnsCompleted: number;
|
|
873
|
-
turnsAborted: number;
|
|
874
|
-
contextCompactions: number;
|
|
875
|
-
mcpCalls: number;
|
|
876
|
-
subagentCalls: number;
|
|
877
|
-
skillCalls: number;
|
|
878
|
-
hookCalls: number;
|
|
879
|
-
webCalls: number;
|
|
880
|
-
codeActions: number;
|
|
881
|
-
prLinks: number;
|
|
882
|
-
fileSnapshots: number;
|
|
883
|
-
graphNodes: number;
|
|
884
|
-
graphEdges: number;
|
|
885
|
-
actionCandidates: number;
|
|
886
|
-
verificationReports: number;
|
|
887
|
-
completionDecisions: number;
|
|
888
|
-
reliabilityRows: number;
|
|
889
|
-
reliabilityLift: number;
|
|
890
|
-
inputTokens: number;
|
|
891
|
-
outputTokens: number;
|
|
892
|
-
reasoningTokens: number;
|
|
893
|
-
cachedTokens: number;
|
|
894
|
-
cacheWriteTokens: number;
|
|
895
|
-
observedCostUsd: number;
|
|
896
|
-
observedCostCaptured?: boolean;
|
|
897
|
-
wallMs: number;
|
|
898
|
-
processScore: number;
|
|
899
|
-
}
|
|
900
|
-
interface CodeAgentSessionDiagnostic {
|
|
901
|
-
source: CodeAgentSessionSource;
|
|
902
|
-
sessionId: string;
|
|
903
|
-
sourcePath?: string;
|
|
904
|
-
entries: number;
|
|
905
|
-
malformedLines: number;
|
|
906
|
-
hasExplicitTerminalSignal: boolean;
|
|
907
|
-
hasFinalOutput: boolean;
|
|
908
|
-
hasQualityLabel: boolean;
|
|
909
|
-
hasTokenUsage: boolean;
|
|
910
|
-
hasCost: boolean;
|
|
911
|
-
costKind?: RunCostProvenance['kind'];
|
|
912
|
-
warnings: string[];
|
|
913
|
-
}
|
|
914
|
-
interface CodeAgentSessionIntakeOptions {
|
|
915
|
-
entries: unknown[];
|
|
916
|
-
malformedLines?: number;
|
|
917
|
-
sourcePath?: string;
|
|
918
|
-
experimentId?: string;
|
|
919
|
-
candidateId?: string;
|
|
920
|
-
seed?: number;
|
|
921
|
-
splitTag?: RunSplitTag;
|
|
922
|
-
scenarioId?: string;
|
|
923
|
-
model?: string;
|
|
924
|
-
promptHash?: string;
|
|
925
|
-
configHash?: string;
|
|
926
|
-
commitSha?: string;
|
|
927
|
-
score?: number;
|
|
928
|
-
/** Explicit cost receipt. When omitted, source-reported cost wins, then a
|
|
929
|
-
* token-priced estimate, then uncaptured. */
|
|
930
|
-
costProvenance?: RunCostProvenance;
|
|
931
|
-
/** Exact executor-owned process result. This is required when a provider's
|
|
932
|
-
* JSON stream has no terminal event, as with `opencode run --format json`. */
|
|
933
|
-
execution?: CodeAgentSessionExecutionReceipt;
|
|
934
|
-
}
|
|
935
|
-
|
|
213
|
+
//#endregion
|
|
214
|
+
//#region src/belief-state/ope.d.ts
|
|
936
215
|
interface BeliefOpeOptions extends OffPolicyOptions {
|
|
937
|
-
|
|
938
|
-
|
|
939
|
-
|
|
216
|
+
minEffectiveSampleSize?: number;
|
|
217
|
+
minEffectiveSampleRatio?: number;
|
|
218
|
+
maxDiagnostics?: number;
|
|
940
219
|
}
|
|
941
220
|
interface BeliefOffPolicyTrajectoryReport {
|
|
942
|
-
|
|
943
|
-
|
|
944
|
-
|
|
945
|
-
|
|
221
|
+
targetPolicyId: string;
|
|
222
|
+
trajectories: OffPolicyTrajectory[];
|
|
223
|
+
dropped: number;
|
|
224
|
+
diagnostics: string[];
|
|
946
225
|
}
|
|
947
226
|
declare function embeddedBeliefOpeTargetPolicy(id?: string): BeliefOpeTargetPolicy;
|
|
948
227
|
declare function beliefDecisionsToOffPolicyTrajectories(points: BeliefDecisionPoint[], targetPolicy: BeliefOpeTargetPolicy, options?: Pick<BeliefOpeOptions, 'maxDiagnostics'>): BeliefOffPolicyTrajectoryReport;
|
|
949
228
|
declare function evaluateBeliefOffPolicy(points: BeliefDecisionPoint[], targetPolicy: BeliefOpeTargetPolicy, options?: BeliefOpeOptions): BeliefOpeReport;
|
|
950
|
-
|
|
229
|
+
//#endregion
|
|
230
|
+
//#region src/belief-state/selective.d.ts
|
|
951
231
|
interface EvaluateBeliefSelectivePolicyOptions {
|
|
952
|
-
|
|
953
|
-
|
|
954
|
-
|
|
955
|
-
|
|
956
|
-
|
|
232
|
+
utility?: BeliefUtilityOptions;
|
|
233
|
+
minN?: number;
|
|
234
|
+
minAccepted?: number;
|
|
235
|
+
minUtilityDelta?: number;
|
|
236
|
+
seed?: number;
|
|
957
237
|
}
|
|
958
238
|
declare function thresholdSelectivePolicy(options: {
|
|
959
|
-
|
|
960
|
-
|
|
961
|
-
|
|
239
|
+
id?: string;
|
|
240
|
+
confidenceThreshold: number;
|
|
241
|
+
belowThresholdAction?: Exclude<BeliefPolicyAction, 'accept'>;
|
|
962
242
|
}): BeliefSelectivePolicy;
|
|
963
243
|
declare function evaluateBeliefSelectivePolicy(points: BeliefDecisionPoint[], policy: BeliefSelectivePolicy, options?: EvaluateBeliefSelectivePolicyOptions): BeliefSelectivePolicyMetrics;
|
|
964
|
-
|
|
244
|
+
//#endregion
|
|
245
|
+
//#region src/belief-state/report.d.ts
|
|
965
246
|
interface AnalyzeBeliefPolicyOpeOptions extends BeliefOpeOptions {
|
|
966
|
-
|
|
247
|
+
targetPolicy?: BeliefOpeTargetPolicy;
|
|
967
248
|
}
|
|
968
249
|
interface AnalyzeBeliefPolicyOptions {
|
|
969
|
-
|
|
970
|
-
|
|
971
|
-
|
|
972
|
-
|
|
973
|
-
|
|
974
|
-
|
|
250
|
+
points: BeliefDecisionPoint[];
|
|
251
|
+
policy: BeliefSelectivePolicy;
|
|
252
|
+
selective?: EvaluateBeliefSelectivePolicyOptions;
|
|
253
|
+
calibration?: BeliefCalibrationOptions;
|
|
254
|
+
ope?: AnalyzeBeliefPolicyOpeOptions;
|
|
255
|
+
requireOpe?: boolean;
|
|
975
256
|
}
|
|
976
257
|
declare function analyzeBeliefPolicy(options: AnalyzeBeliefPolicyOptions): BeliefPolicyEvaluationReport;
|
|
977
|
-
|
|
258
|
+
//#endregion
|
|
259
|
+
//#region src/belief-state/code-agent-corpus.d.ts
|
|
978
260
|
type CodeAgentBeliefDecisionTargetId = 'failure-recovery' | 'tool-selection' | 'graph-completion';
|
|
979
261
|
interface ExtractCodeAgentBeliefDecisionPointsOptions {
|
|
980
|
-
|
|
981
|
-
|
|
982
|
-
|
|
983
|
-
|
|
984
|
-
|
|
985
|
-
|
|
262
|
+
source: CodeAgentSessionSource;
|
|
263
|
+
entries: unknown[];
|
|
264
|
+
/** Reuse intake's provider-neutral projection when available. */
|
|
265
|
+
observation?: CodeAgentSessionObservation;
|
|
266
|
+
run: Pick<RunRecord, 'runId' | 'scenarioId' | 'outcome' | 'costUsd'>;
|
|
267
|
+
sourcePath?: string;
|
|
986
268
|
}
|
|
987
269
|
interface BeliefDecisionInventoryBucket {
|
|
988
|
-
|
|
989
|
-
|
|
990
|
-
|
|
991
|
-
|
|
992
|
-
|
|
993
|
-
|
|
994
|
-
|
|
995
|
-
|
|
996
|
-
|
|
997
|
-
|
|
998
|
-
|
|
999
|
-
|
|
270
|
+
id: string;
|
|
271
|
+
kind?: BeliefDecisionKind;
|
|
272
|
+
targetId?: CodeAgentBeliefDecisionTargetId;
|
|
273
|
+
n: number;
|
|
274
|
+
withOutcome: number;
|
|
275
|
+
withConfidence: number;
|
|
276
|
+
withCandidateActions: number;
|
|
277
|
+
withBehaviorProb: number;
|
|
278
|
+
withTargetProb: number;
|
|
279
|
+
successRate: number | null;
|
|
280
|
+
meanScore: number | null;
|
|
281
|
+
meanConfidence: number | null;
|
|
1000
282
|
}
|
|
1001
283
|
interface BeliefDecisionInventoryReport {
|
|
1002
|
-
|
|
1003
|
-
|
|
1004
|
-
|
|
1005
|
-
|
|
284
|
+
n: number;
|
|
285
|
+
byKind: BeliefDecisionInventoryBucket[];
|
|
286
|
+
byTarget: BeliefDecisionInventoryBucket[];
|
|
287
|
+
diagnostics: string[];
|
|
1006
288
|
}
|
|
1007
289
|
interface BeliefDecisionTargetSelection {
|
|
1008
|
-
|
|
1009
|
-
|
|
1010
|
-
|
|
1011
|
-
|
|
1012
|
-
|
|
290
|
+
id: CodeAgentBeliefDecisionTargetId;
|
|
291
|
+
label: string;
|
|
292
|
+
points: BeliefDecisionPoint[];
|
|
293
|
+
support: BeliefDecisionInventoryBucket;
|
|
294
|
+
reasons: string[];
|
|
1013
295
|
}
|
|
1014
296
|
interface SelectBeliefDecisionTargetOptions {
|
|
1015
|
-
|
|
1016
|
-
|
|
1017
|
-
|
|
297
|
+
minN?: number;
|
|
298
|
+
minOutcomeCoverage?: number;
|
|
299
|
+
preferredTargets?: CodeAgentBeliefDecisionTargetId[];
|
|
1018
300
|
}
|
|
1019
301
|
interface AnalyzeBeliefDecisionCorpusOptions {
|
|
1020
|
-
|
|
1021
|
-
|
|
1022
|
-
|
|
1023
|
-
|
|
1024
|
-
|
|
1025
|
-
|
|
1026
|
-
|
|
1027
|
-
|
|
1028
|
-
|
|
302
|
+
points: BeliefDecisionPoint[];
|
|
303
|
+
targetId?: CodeAgentBeliefDecisionTargetId;
|
|
304
|
+
minN?: number;
|
|
305
|
+
minOutcomeCoverage?: number;
|
|
306
|
+
minAccepted?: number;
|
|
307
|
+
confidenceThreshold?: number;
|
|
308
|
+
policy?: BeliefSelectivePolicy;
|
|
309
|
+
requireOpe?: boolean;
|
|
310
|
+
policyOptions?: Partial<AnalyzeBeliefPolicyOptions>;
|
|
1029
311
|
}
|
|
1030
312
|
interface BeliefDecisionCorpusEvaluation {
|
|
1031
|
-
|
|
1032
|
-
|
|
1033
|
-
|
|
1034
|
-
|
|
1035
|
-
|
|
313
|
+
inventory: BeliefDecisionInventoryReport;
|
|
314
|
+
target?: BeliefDecisionTargetSelection;
|
|
315
|
+
policy?: BeliefSelectivePolicy;
|
|
316
|
+
evaluation?: BeliefPolicyEvaluationReport;
|
|
317
|
+
diagnostics: string[];
|
|
1036
318
|
}
|
|
1037
319
|
declare function extractCodeAgentBeliefDecisionPoints(options: ExtractCodeAgentBeliefDecisionPointsOptions): BeliefDecisionExtractionReport;
|
|
1038
320
|
declare function inventoryBeliefDecisionPoints(points: BeliefDecisionPoint[]): BeliefDecisionInventoryReport;
|
|
1039
321
|
declare function selectBeliefDecisionTarget(points: BeliefDecisionPoint[], options?: SelectBeliefDecisionTargetOptions): BeliefDecisionTargetSelection | null;
|
|
1040
322
|
declare function analyzeBeliefDecisionCorpus(options: AnalyzeBeliefDecisionCorpusOptions): BeliefDecisionCorpusEvaluation;
|
|
1041
|
-
|
|
323
|
+
//#endregion
|
|
324
|
+
//#region src/belief-state/research-evidence.d.ts
|
|
1042
325
|
type BeliefResearchClaimScope = 'selective' | 'counterfactual';
|
|
1043
326
|
type BeliefResearchEvidenceStatus = 'supported' | 'blocked';
|
|
1044
327
|
type BeliefResearchGateId = 'corpus' | 'selective' | 'calibration' | 'ope';
|
|
1045
328
|
interface BeliefResearchEvidenceGate {
|
|
1046
|
-
|
|
1047
|
-
|
|
1048
|
-
|
|
1049
|
-
|
|
329
|
+
id: BeliefResearchGateId;
|
|
330
|
+
status: BeliefResearchEvidenceStatus;
|
|
331
|
+
blockers: string[];
|
|
332
|
+
caveats: string[];
|
|
1050
333
|
}
|
|
1051
334
|
interface BeliefDecisionResearchEvidencePacket {
|
|
1052
|
-
|
|
1053
|
-
|
|
1054
|
-
|
|
1055
|
-
|
|
1056
|
-
|
|
1057
|
-
|
|
335
|
+
claimScope: BeliefResearchClaimScope;
|
|
336
|
+
status: BeliefResearchEvidenceStatus;
|
|
337
|
+
analysis: BeliefDecisionCorpusEvaluation;
|
|
338
|
+
gates: BeliefResearchEvidenceGate[];
|
|
339
|
+
blockers: string[];
|
|
340
|
+
caveats: string[];
|
|
1058
341
|
}
|
|
1059
342
|
interface BuildBeliefDecisionResearchEvidencePacketOptions extends AnalyzeBeliefDecisionCorpusOptions {
|
|
1060
|
-
|
|
343
|
+
claimScope?: BeliefResearchClaimScope;
|
|
1061
344
|
}
|
|
1062
345
|
declare function buildBeliefDecisionResearchEvidencePacket(options: BuildBeliefDecisionResearchEvidencePacketOptions): BeliefDecisionResearchEvidencePacket;
|
|
1063
|
-
|
|
346
|
+
//#endregion
|
|
347
|
+
//#region src/belief-state/code-agent-evidence.d.ts
|
|
1064
348
|
interface CodeAgentBeliefSession extends CodeAgentSessionIntakeOptions {
|
|
1065
|
-
|
|
349
|
+
source: CodeAgentSessionSource;
|
|
1066
350
|
}
|
|
1067
351
|
interface BuildCodeAgentBeliefEvidenceCorpusOptions extends Omit<BuildBeliefDecisionResearchEvidencePacketOptions, 'points'> {
|
|
1068
|
-
|
|
352
|
+
sessions: CodeAgentBeliefSession[];
|
|
1069
353
|
}
|
|
1070
354
|
interface CodeAgentBeliefEvidenceCorpus {
|
|
1071
|
-
|
|
1072
|
-
|
|
1073
|
-
|
|
1074
|
-
|
|
1075
|
-
|
|
1076
|
-
|
|
1077
|
-
|
|
355
|
+
runs: RunRecord[];
|
|
356
|
+
metrics: CodeAgentSessionMetrics[];
|
|
357
|
+
intakeDiagnostics: CodeAgentSessionDiagnostic[];
|
|
358
|
+
extractionDiagnostics: BeliefDecisionExtractionDiagnostic[];
|
|
359
|
+
decisions: BeliefDecisionPoint[];
|
|
360
|
+
inventory: BeliefDecisionInventoryReport;
|
|
361
|
+
evidence: BeliefDecisionResearchEvidencePacket;
|
|
1078
362
|
}
|
|
1079
363
|
declare function buildCodeAgentBeliefEvidenceCorpus(options: BuildCodeAgentBeliefEvidenceCorpusOptions): CodeAgentBeliefEvidenceCorpus;
|
|
1080
|
-
|
|
364
|
+
//#endregion
|
|
365
|
+
//#region src/belief-state/extract.d.ts
|
|
1081
366
|
interface ExtractBeliefDecisionPointsOptions {
|
|
1082
|
-
|
|
367
|
+
runIds?: string[];
|
|
1083
368
|
}
|
|
1084
369
|
declare function extractBeliefDecisionPoints(store: TraceStore, options?: ExtractBeliefDecisionPointsOptions): Promise<BeliefDecisionExtractionReport>;
|
|
1085
|
-
|
|
370
|
+
//#endregion
|
|
371
|
+
//#region src/belief-state/shadow-probe.d.ts
|
|
1086
372
|
interface BeliefShadowProbeInput {
|
|
1087
|
-
|
|
1088
|
-
|
|
1089
|
-
|
|
1090
|
-
|
|
1091
|
-
|
|
1092
|
-
|
|
1093
|
-
|
|
1094
|
-
|
|
1095
|
-
|
|
1096
|
-
|
|
1097
|
-
|
|
373
|
+
probeId: string;
|
|
374
|
+
decisionId: string;
|
|
375
|
+
runId: string;
|
|
376
|
+
scenarioId?: string;
|
|
377
|
+
stepIndex: number;
|
|
378
|
+
decisionKind: BeliefDecisionKind;
|
|
379
|
+
candidateActions: string[];
|
|
380
|
+
observedAction?: string;
|
|
381
|
+
evidence: BeliefShadowProbeEvidenceRef[];
|
|
382
|
+
context?: string;
|
|
383
|
+
metadata?: Record<string, unknown>;
|
|
1098
384
|
}
|
|
1099
385
|
interface BeliefShadowProbeEvidenceRef {
|
|
1100
|
-
|
|
1101
|
-
|
|
1102
|
-
|
|
1103
|
-
|
|
386
|
+
id: string;
|
|
387
|
+
source: string;
|
|
388
|
+
detail?: string;
|
|
389
|
+
quality?: BeliefEvidenceQuality;
|
|
1104
390
|
}
|
|
1105
391
|
interface BeliefShadowProbeResponse {
|
|
1106
|
-
|
|
1107
|
-
|
|
1108
|
-
|
|
1109
|
-
|
|
1110
|
-
|
|
1111
|
-
|
|
1112
|
-
|
|
1113
|
-
|
|
1114
|
-
|
|
392
|
+
predictedAction: string;
|
|
393
|
+
confidence: number;
|
|
394
|
+
beliefSummary?: string;
|
|
395
|
+
uncertainty?: string[];
|
|
396
|
+
evidenceRefs?: string[];
|
|
397
|
+
wouldChangeMindIf?: string[];
|
|
398
|
+
targetProb?: number;
|
|
399
|
+
qHatChosen?: number | null;
|
|
400
|
+
vHatTarget?: number | null;
|
|
401
|
+
metadata?: Record<string, unknown>;
|
|
1115
402
|
}
|
|
1116
403
|
interface BeliefShadowProbeRecord extends BeliefShadowProbeResponse {
|
|
1117
|
-
|
|
1118
|
-
|
|
1119
|
-
|
|
1120
|
-
|
|
1121
|
-
|
|
1122
|
-
|
|
1123
|
-
|
|
1124
|
-
|
|
1125
|
-
|
|
1126
|
-
|
|
404
|
+
probeId: string;
|
|
405
|
+
decisionId: string;
|
|
406
|
+
runId: string;
|
|
407
|
+
scenarioId?: string;
|
|
408
|
+
stepIndex: number;
|
|
409
|
+
decisionKind: BeliefDecisionKind;
|
|
410
|
+
candidateActions: string[];
|
|
411
|
+
observedAction: string;
|
|
412
|
+
agreesWithObservedAction: boolean;
|
|
413
|
+
outcome?: BeliefDecisionOutcome;
|
|
1127
414
|
}
|
|
1128
415
|
interface BeliefShadowProbeDiagnostic {
|
|
1129
|
-
|
|
1130
|
-
|
|
1131
|
-
|
|
416
|
+
decisionId: string;
|
|
417
|
+
severity: 'warning' | 'error';
|
|
418
|
+
reason: string;
|
|
1132
419
|
}
|
|
1133
420
|
interface BeliefShadowProbeSummary {
|
|
1134
|
-
|
|
1135
|
-
|
|
1136
|
-
|
|
1137
|
-
|
|
1138
|
-
|
|
1139
|
-
|
|
1140
|
-
|
|
421
|
+
attempted: number;
|
|
422
|
+
completed: number;
|
|
423
|
+
dropped: number;
|
|
424
|
+
withOutcome: number;
|
|
425
|
+
withTargetProb: number;
|
|
426
|
+
meanConfidence: number | null;
|
|
427
|
+
observedAgreementRate: number | null;
|
|
1141
428
|
}
|
|
1142
429
|
interface BeliefShadowProbeRun {
|
|
1143
|
-
|
|
1144
|
-
|
|
1145
|
-
|
|
1146
|
-
|
|
430
|
+
probeId: string;
|
|
431
|
+
records: BeliefShadowProbeRecord[];
|
|
432
|
+
diagnostics: BeliefShadowProbeDiagnostic[];
|
|
433
|
+
summary: BeliefShadowProbeSummary;
|
|
1147
434
|
}
|
|
1148
435
|
interface RunBeliefShadowProbeOptions {
|
|
1149
|
-
|
|
1150
|
-
|
|
1151
|
-
|
|
1152
|
-
|
|
1153
|
-
|
|
1154
|
-
|
|
1155
|
-
|
|
1156
|
-
|
|
1157
|
-
|
|
1158
|
-
|
|
1159
|
-
|
|
1160
|
-
|
|
436
|
+
probeId: string;
|
|
437
|
+
points: BeliefDecisionPoint[];
|
|
438
|
+
probe: (input: BeliefShadowProbeInput) => BeliefShadowProbeResponse | Promise<BeliefShadowProbeResponse>;
|
|
439
|
+
contextOf?: (point: BeliefDecisionPoint) => string | undefined | Promise<string | undefined>;
|
|
440
|
+
metadataOf?: (point: BeliefDecisionPoint) => Record<string, unknown> | undefined | Promise<Record<string, unknown> | undefined>;
|
|
441
|
+
includeObservedAction?: boolean;
|
|
442
|
+
includeEvidenceDetail?: boolean;
|
|
443
|
+
includeOutcomeInRecord?: boolean;
|
|
444
|
+
requireCandidateActions?: boolean;
|
|
445
|
+
allowOutOfSetActions?: boolean;
|
|
446
|
+
concurrency?: number;
|
|
447
|
+
maxContextChars?: number;
|
|
1161
448
|
}
|
|
1162
449
|
declare function runBeliefShadowProbe(options: RunBeliefShadowProbeOptions): Promise<BeliefShadowProbeRun>;
|
|
1163
450
|
declare function formatBeliefShadowProbePrompt(input: BeliefShadowProbeInput): string;
|
|
1164
|
-
|
|
451
|
+
//#endregion
|
|
452
|
+
//#region src/belief-state/runtime-hooks.d.ts
|
|
1165
453
|
interface RuntimeBeliefDecisionEvidenceRef {
|
|
1166
|
-
|
|
1167
|
-
|
|
1168
|
-
|
|
1169
|
-
|
|
1170
|
-
|
|
454
|
+
source: string;
|
|
455
|
+
id: string;
|
|
456
|
+
detail?: string;
|
|
457
|
+
quality?: BeliefEvidenceQuality;
|
|
458
|
+
metadata?: Record<string, unknown>;
|
|
1171
459
|
}
|
|
1172
460
|
interface RuntimeBeliefDecisionPoint {
|
|
1173
|
-
|
|
1174
|
-
|
|
1175
|
-
|
|
1176
|
-
|
|
1177
|
-
|
|
1178
|
-
|
|
1179
|
-
|
|
1180
|
-
|
|
1181
|
-
|
|
461
|
+
id: string;
|
|
462
|
+
runId: string;
|
|
463
|
+
scenarioId?: string;
|
|
464
|
+
stepIndex: number;
|
|
465
|
+
kind: string;
|
|
466
|
+
candidateActions?: string[];
|
|
467
|
+
context?: string;
|
|
468
|
+
evidence?: RuntimeBeliefDecisionEvidenceRef[];
|
|
469
|
+
metadata?: Record<string, unknown>;
|
|
1182
470
|
}
|
|
1183
471
|
interface RuntimeBeliefHookEvent {
|
|
1184
|
-
|
|
1185
|
-
|
|
1186
|
-
|
|
1187
|
-
|
|
1188
|
-
|
|
1189
|
-
|
|
1190
|
-
|
|
1191
|
-
|
|
1192
|
-
|
|
1193
|
-
|
|
472
|
+
id: string;
|
|
473
|
+
runId: string;
|
|
474
|
+
scenarioId?: string;
|
|
475
|
+
target: string;
|
|
476
|
+
phase: string;
|
|
477
|
+
timestamp: number;
|
|
478
|
+
stepIndex?: number;
|
|
479
|
+
parentId?: string;
|
|
480
|
+
payload?: unknown;
|
|
481
|
+
metadata?: Record<string, unknown>;
|
|
1194
482
|
}
|
|
1195
483
|
interface RuntimeBeliefHookContext {
|
|
1196
|
-
|
|
484
|
+
signal?: AbortSignal;
|
|
1197
485
|
}
|
|
1198
486
|
interface RuntimeBeliefHooks {
|
|
1199
|
-
|
|
1200
|
-
|
|
487
|
+
onEvent?: (event: RuntimeBeliefHookEvent, context: RuntimeBeliefHookContext) => void | Promise<void>;
|
|
488
|
+
onDecisionPoint?: (point: RuntimeBeliefDecisionPoint, context: RuntimeBeliefHookContext) => void | Promise<void>;
|
|
1201
489
|
}
|
|
1202
490
|
interface RuntimeBeliefConversionDiagnostic {
|
|
1203
|
-
|
|
1204
|
-
|
|
1205
|
-
|
|
491
|
+
decisionId: string;
|
|
492
|
+
severity: 'warning' | 'error';
|
|
493
|
+
reason: string;
|
|
1206
494
|
}
|
|
1207
495
|
interface RuntimeBeliefShadowProbeInputOptions {
|
|
1208
|
-
|
|
1209
|
-
|
|
1210
|
-
|
|
1211
|
-
|
|
1212
|
-
|
|
1213
|
-
|
|
496
|
+
probeId: string;
|
|
497
|
+
decisionKind?: BeliefDecisionKind;
|
|
498
|
+
includeEvidenceDetail?: boolean;
|
|
499
|
+
includeLifecycleEvidence?: boolean;
|
|
500
|
+
lifecycleEvents?: RuntimeBeliefHookEvent[];
|
|
501
|
+
maxContextChars?: number;
|
|
1214
502
|
}
|
|
1215
503
|
interface RuntimeBeliefDecisionPointOptions {
|
|
1216
|
-
|
|
1217
|
-
|
|
1218
|
-
|
|
1219
|
-
|
|
1220
|
-
|
|
1221
|
-
|
|
1222
|
-
|
|
1223
|
-
|
|
1224
|
-
|
|
1225
|
-
|
|
1226
|
-
|
|
1227
|
-
|
|
1228
|
-
includeLifecycleEvidence?: boolean;
|
|
1229
|
-
lifecycleEvents?: RuntimeBeliefHookEvent[];
|
|
504
|
+
chosenAction?: string;
|
|
505
|
+
decisionKind?: BeliefDecisionKind;
|
|
506
|
+
confidence?: number;
|
|
507
|
+
behaviorProb?: number;
|
|
508
|
+
targetProb?: number;
|
|
509
|
+
qHatChosen?: number | null;
|
|
510
|
+
vHatTarget?: number | null;
|
|
511
|
+
costUsd?: number;
|
|
512
|
+
outcome?: BeliefDecisionOutcome;
|
|
513
|
+
metadata?: Record<string, unknown>;
|
|
514
|
+
includeLifecycleEvidence?: boolean;
|
|
515
|
+
lifecycleEvents?: RuntimeBeliefHookEvent[];
|
|
1230
516
|
}
|
|
1231
517
|
interface RuntimeBeliefShadowProbeInputReport {
|
|
1232
|
-
|
|
1233
|
-
|
|
518
|
+
input?: BeliefShadowProbeInput;
|
|
519
|
+
diagnostics: RuntimeBeliefConversionDiagnostic[];
|
|
1234
520
|
}
|
|
1235
521
|
interface RuntimeBeliefDecisionPointReport {
|
|
1236
|
-
|
|
1237
|
-
|
|
522
|
+
point?: BeliefDecisionPoint;
|
|
523
|
+
diagnostics: RuntimeBeliefConversionDiagnostic[];
|
|
1238
524
|
}
|
|
1239
525
|
interface BeliefRuntimeHookCollector {
|
|
1240
|
-
|
|
1241
|
-
|
|
1242
|
-
|
|
1243
|
-
|
|
1244
|
-
|
|
1245
|
-
|
|
1246
|
-
|
|
1247
|
-
|
|
526
|
+
hooks: RuntimeBeliefHooks;
|
|
527
|
+
decisions: RuntimeBeliefDecisionPoint[];
|
|
528
|
+
events: RuntimeBeliefHookEvent[];
|
|
529
|
+
toShadowProbeInputs(options?: Partial<RuntimeBeliefShadowProbeInputOptions>): {
|
|
530
|
+
inputs: BeliefShadowProbeInput[];
|
|
531
|
+
diagnostics: RuntimeBeliefConversionDiagnostic[];
|
|
532
|
+
};
|
|
533
|
+
clear(): void;
|
|
1248
534
|
}
|
|
1249
535
|
declare function runtimeDecisionPointToBeliefShadowProbeInput(point: RuntimeBeliefDecisionPoint, options: RuntimeBeliefShadowProbeInputOptions): RuntimeBeliefShadowProbeInputReport;
|
|
1250
536
|
declare function runtimeDecisionPointToBeliefDecisionPoint(point: RuntimeBeliefDecisionPoint, options: RuntimeBeliefDecisionPointOptions): RuntimeBeliefDecisionPointReport;
|
|
1251
537
|
declare function createBeliefRuntimeHookCollector(defaults: RuntimeBeliefShadowProbeInputOptions): BeliefRuntimeHookCollector;
|
|
1252
|
-
|
|
538
|
+
//#endregion
|
|
539
|
+
//#region src/belief-state/phase0-measurement.d.ts
|
|
1253
540
|
interface RuntimeBeliefPhase0RunRecord {
|
|
1254
|
-
|
|
1255
|
-
|
|
1256
|
-
|
|
541
|
+
runId: string;
|
|
542
|
+
scenarioId?: string;
|
|
543
|
+
splitTag: RunSplitTag;
|
|
1257
544
|
}
|
|
1258
545
|
interface RuntimeBeliefDecisionLabel {
|
|
1259
|
-
|
|
1260
|
-
|
|
1261
|
-
|
|
1262
|
-
|
|
1263
|
-
|
|
1264
|
-
|
|
1265
|
-
|
|
1266
|
-
|
|
1267
|
-
|
|
1268
|
-
|
|
1269
|
-
|
|
1270
|
-
splitTag?: RunSplitTag;
|
|
1271
|
-
metadata?: Record<string, unknown>;
|
|
546
|
+
decisionId: string;
|
|
547
|
+
chosenAction: string;
|
|
548
|
+
outcome: BeliefDecisionOutcome;
|
|
549
|
+
confidence?: number;
|
|
550
|
+
behaviorProb?: number;
|
|
551
|
+
targetProb?: number;
|
|
552
|
+
qHatChosen?: number | null;
|
|
553
|
+
vHatTarget?: number | null;
|
|
554
|
+
costUsd?: number;
|
|
555
|
+
splitTag?: RunSplitTag;
|
|
556
|
+
metadata?: Record<string, unknown>;
|
|
1272
557
|
}
|
|
1273
558
|
interface BuildRuntimeBeliefPhase0MeasurementOptions extends Omit<BuildBeliefDecisionResearchEvidencePacketOptions, 'points'> {
|
|
1274
|
-
|
|
1275
|
-
|
|
1276
|
-
|
|
1277
|
-
|
|
1278
|
-
|
|
559
|
+
runs: RuntimeBeliefPhase0RunRecord[];
|
|
560
|
+
decisions: RuntimeBeliefDecisionPoint[];
|
|
561
|
+
events?: RuntimeBeliefHookEvent[];
|
|
562
|
+
labels: RuntimeBeliefDecisionLabel[];
|
|
563
|
+
baselinePolicyId?: string;
|
|
1279
564
|
}
|
|
1280
565
|
interface RuntimeBeliefPhase0MeasurementSummary {
|
|
1281
|
-
|
|
1282
|
-
|
|
1283
|
-
|
|
1284
|
-
|
|
1285
|
-
|
|
1286
|
-
|
|
1287
|
-
|
|
1288
|
-
|
|
1289
|
-
|
|
1290
|
-
|
|
1291
|
-
|
|
1292
|
-
|
|
1293
|
-
|
|
1294
|
-
|
|
1295
|
-
|
|
1296
|
-
|
|
1297
|
-
|
|
566
|
+
runCount: number;
|
|
567
|
+
producerDecisionCount: number;
|
|
568
|
+
lifecycleEventCount: number;
|
|
569
|
+
labelCount: number;
|
|
570
|
+
completedPointCount: number;
|
|
571
|
+
runJoinRate: number;
|
|
572
|
+
labelJoinRate: number;
|
|
573
|
+
missingRunRecordCount: number;
|
|
574
|
+
missingLabelCount: number;
|
|
575
|
+
withEvidence: number;
|
|
576
|
+
withOutcome: number;
|
|
577
|
+
withSplit: number;
|
|
578
|
+
withBehaviorProb: number;
|
|
579
|
+
withTargetProb: number;
|
|
580
|
+
baselinePolicyId: string;
|
|
581
|
+
packetStatus: BeliefDecisionResearchEvidencePacket['status'];
|
|
582
|
+
claimScope: BeliefDecisionResearchEvidencePacket['claimScope'];
|
|
1298
583
|
}
|
|
1299
584
|
interface RuntimeBeliefPhase0Measurement {
|
|
1300
|
-
|
|
1301
|
-
|
|
1302
|
-
|
|
1303
|
-
|
|
585
|
+
points: BeliefDecisionPoint[];
|
|
586
|
+
packet: BeliefDecisionResearchEvidencePacket;
|
|
587
|
+
summary: RuntimeBeliefPhase0MeasurementSummary;
|
|
588
|
+
diagnostics: string[];
|
|
1304
589
|
}
|
|
1305
590
|
declare function buildRuntimeBeliefPhase0Measurement(options: BuildRuntimeBeliefPhase0MeasurementOptions): RuntimeBeliefPhase0Measurement;
|
|
1306
|
-
|
|
1307
|
-
|
|
1308
|
-
id: string;
|
|
1309
|
-
runId: string;
|
|
1310
|
-
scenarioId?: string;
|
|
1311
|
-
target: string;
|
|
1312
|
-
phase: string;
|
|
1313
|
-
timestamp: number;
|
|
1314
|
-
stepIndex?: number;
|
|
1315
|
-
parentId?: string;
|
|
1316
|
-
payload?: unknown;
|
|
1317
|
-
metadata?: Record<string, unknown>;
|
|
1318
|
-
}
|
|
1319
|
-
interface RuntimeTrajectoryRecord {
|
|
1320
|
-
id?: string;
|
|
1321
|
-
scenarioId?: string;
|
|
1322
|
-
splitTag?: RunSplitTag;
|
|
1323
|
-
runtimeEvents?: unknown;
|
|
1324
|
-
[key: string]: unknown;
|
|
1325
|
-
}
|
|
1326
|
-
interface RuntimeTrajectoryRunRecord {
|
|
1327
|
-
runId: string;
|
|
1328
|
-
scenarioId?: string;
|
|
1329
|
-
splitTag: RunSplitTag;
|
|
1330
|
-
}
|
|
1331
|
-
interface RuntimeTrajectoryEvidenceSummary {
|
|
1332
|
-
recordCount: number;
|
|
1333
|
-
recordWithRuntimeEventsCount: number;
|
|
1334
|
-
runtimeRunCount: number;
|
|
1335
|
-
lifecycleEventCount: number;
|
|
1336
|
-
defaultedSplitCount: number;
|
|
1337
|
-
}
|
|
1338
|
-
interface RuntimeTrajectoryEvidenceProjection {
|
|
1339
|
-
runs: RuntimeTrajectoryRunRecord[];
|
|
1340
|
-
events: RuntimeTrajectoryHookEvent[];
|
|
1341
|
-
summary: RuntimeTrajectoryEvidenceSummary;
|
|
1342
|
-
diagnostics: string[];
|
|
1343
|
-
}
|
|
1344
|
-
interface ProjectRuntimeTrajectoryEvidenceOptions<TRecord extends RuntimeTrajectoryRecord = RuntimeTrajectoryRecord> {
|
|
1345
|
-
records: TRecord[];
|
|
1346
|
-
defaultSplitTag?: RunSplitTag;
|
|
1347
|
-
recordIdOf?: (record: TRecord, index: number) => string | undefined;
|
|
1348
|
-
scenarioIdOf?: (record: TRecord, index: number) => string | undefined;
|
|
1349
|
-
}
|
|
1350
|
-
|
|
591
|
+
//#endregion
|
|
592
|
+
//#region src/belief-state/runtime-benchmark-corpus.d.ts
|
|
1351
593
|
type RuntimeBenchmarkTrajectoryRecord = RuntimeTrajectoryRecord & {
|
|
1352
|
-
|
|
1353
|
-
|
|
1354
|
-
|
|
1355
|
-
|
|
594
|
+
benchmark?: unknown;
|
|
595
|
+
condition?: unknown;
|
|
596
|
+
instanceId?: unknown;
|
|
597
|
+
runtimeDecisionPoints?: unknown;
|
|
1356
598
|
};
|
|
1357
599
|
interface BuildRuntimeBenchmarkBeliefPhase0MeasurementOptions extends Omit<BuildRuntimeBeliefPhase0MeasurementOptions, 'runs' | 'events' | 'decisions' | 'labels'> {
|
|
1358
|
-
|
|
1359
|
-
|
|
1360
|
-
|
|
1361
|
-
|
|
600
|
+
records: RuntimeBenchmarkTrajectoryRecord[];
|
|
601
|
+
decisions?: RuntimeBeliefDecisionPoint[];
|
|
602
|
+
defaultSplitTag?: ProjectRuntimeTrajectoryEvidenceOptions['defaultSplitTag'];
|
|
603
|
+
labels?: RuntimeBeliefDecisionLabel[];
|
|
1362
604
|
}
|
|
1363
605
|
interface RuntimeBenchmarkBeliefPhase0Summary {
|
|
1364
|
-
|
|
1365
|
-
|
|
606
|
+
decisionCount: number;
|
|
607
|
+
labelCount: number;
|
|
1366
608
|
}
|
|
1367
609
|
interface RuntimeBenchmarkBeliefPhase0Measurement {
|
|
1368
|
-
|
|
1369
|
-
|
|
1370
|
-
|
|
1371
|
-
|
|
1372
|
-
|
|
1373
|
-
|
|
1374
|
-
|
|
1375
|
-
|
|
610
|
+
runs: RuntimeBeliefPhase0RunRecord[];
|
|
611
|
+
events: RuntimeBeliefHookEvent[];
|
|
612
|
+
decisions: RuntimeBeliefDecisionPoint[];
|
|
613
|
+
labels: RuntimeBeliefDecisionLabel[];
|
|
614
|
+
trajectory: RuntimeTrajectoryEvidenceProjection;
|
|
615
|
+
measurement: RuntimeBeliefPhase0Measurement;
|
|
616
|
+
summary: RuntimeBenchmarkBeliefPhase0Summary;
|
|
617
|
+
diagnostics: string[];
|
|
1376
618
|
}
|
|
1377
619
|
declare function buildRuntimeBenchmarkBeliefPhase0Measurement(options: BuildRuntimeBenchmarkBeliefPhase0MeasurementOptions): RuntimeBenchmarkBeliefPhase0Measurement;
|
|
1378
|
-
|
|
1379
|
-
export {
|
|
620
|
+
//#endregion
|
|
621
|
+
export { AnalyzeBeliefDecisionCorpusOptions, AnalyzeBeliefPolicyOpeOptions, AnalyzeBeliefPolicyOptions, BELIEF_DECISION_KINDS, BELIEF_EVALUATION_CRITERIA, BELIEF_EVIDENCE_QUALITIES, BELIEF_EVIDENCE_SOURCES, BeliefCalibrationOptions, BeliefCalibrationRegion, BeliefCalibrationStatus, BeliefDecisionCorpusEvaluation, BeliefDecisionExtractionDiagnostic, BeliefDecisionExtractionReport, BeliefDecisionInventoryBucket, BeliefDecisionInventoryReport, BeliefDecisionKind, BeliefDecisionOutcome, BeliefDecisionPoint, BeliefDecisionReason, BeliefDecisionReasonCode, BeliefDecisionResearchEvidencePacket, BeliefDecisionTargetSelection, BeliefEvaluationCriterionId, BeliefEvaluationStatus, BeliefEvidenceQuality, BeliefEvidenceRef, BeliefEvidenceSource, BeliefOffPolicyTrajectoryReport, BeliefOpeOptions, BeliefOpeReport, BeliefOpeStatus, BeliefOpeSupportDiagnostics, BeliefOpeTargetPolicy, BeliefPolicyAction, BeliefPolicyDecision, BeliefPolicyEvaluationReport, BeliefResearchClaimScope, BeliefResearchEvidenceGate, BeliefResearchEvidenceStatus, BeliefResearchGateId, BeliefRuntimeHookCollector, BeliefSelectivePolicy, BeliefSelectivePolicyMetrics, BeliefShadowProbeDiagnostic, BeliefShadowProbeEvidenceRef, BeliefShadowProbeInput, BeliefShadowProbeRecord, BeliefShadowProbeResponse, BeliefShadowProbeRun, BeliefShadowProbeSummary, BeliefUtilityOptions, BuildBeliefDecisionResearchEvidencePacketOptions, BuildCodeAgentBeliefEvidenceCorpusOptions, BuildRuntimeBeliefPhase0MeasurementOptions, BuildRuntimeBenchmarkBeliefPhase0MeasurementOptions, CodeAgentBeliefDecisionTargetId, CodeAgentBeliefEvidenceCorpus, CodeAgentBeliefSession, EvaluateBeliefSelectivePolicyOptions, ExtractBeliefDecisionPointsOptions, ExtractCodeAgentBeliefDecisionPointsOptions, RunBeliefShadowProbeOptions, RuntimeBeliefConversionDiagnostic, RuntimeBeliefDecisionEvidenceRef, RuntimeBeliefDecisionLabel, RuntimeBeliefDecisionPoint, RuntimeBeliefDecisionPointOptions, RuntimeBeliefDecisionPointReport, RuntimeBeliefHookContext, RuntimeBeliefHookEvent, RuntimeBeliefHooks, RuntimeBeliefPhase0Measurement, RuntimeBeliefPhase0MeasurementSummary, RuntimeBeliefPhase0RunRecord, RuntimeBeliefShadowProbeInputOptions, RuntimeBeliefShadowProbeInputReport, RuntimeBenchmarkBeliefPhase0Measurement, RuntimeBenchmarkBeliefPhase0Summary, SelectBeliefDecisionTargetOptions, analyzeBeliefDecisionCorpus, analyzeBeliefPolicy, beliefDecisionsToOffPolicyTrajectories, buildBeliefDecisionResearchEvidencePacket, buildCodeAgentBeliefEvidenceCorpus, buildRuntimeBeliefPhase0Measurement, buildRuntimeBenchmarkBeliefPhase0Measurement, calibrateBeliefDecisions, createBeliefRuntimeHookCollector, embeddedBeliefOpeTargetPolicy, evaluateBeliefOffPolicy, evaluateBeliefSelectivePolicy, extractBeliefDecisionPoints, extractCodeAgentBeliefDecisionPoints, formatBeliefShadowProbePrompt, inventoryBeliefDecisionPoints, isBeliefDecisionKind, isBeliefEvidenceSource, runBeliefShadowProbe, runtimeDecisionPointToBeliefDecisionPoint, runtimeDecisionPointToBeliefShadowProbeInput, selectBeliefDecisionTarget, thresholdSelectivePolicy };
|
|
622
|
+
//# sourceMappingURL=index.d.ts.map
|