@tangle-network/agent-eval 0.128.2 → 0.130.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +279 -0
- package/README.md +19 -0
- package/dist/active-curriculum-C4mk67HP.js +214 -0
- package/dist/active-curriculum-C4mk67HP.js.map +1 -0
- package/dist/adversarial-smnADNFS.d.ts +21 -0
- package/dist/adversarial-smnADNFS.d.ts.map +1 -0
- package/dist/analyst/index.d.ts +83 -2932
- package/dist/analyst/index.d.ts.map +1 -0
- package/dist/analyst/index.js +319 -364
- package/dist/analyst/index.js.map +1 -1
- package/dist/analyst-BkTS3C58.d.ts +89 -0
- package/dist/analyst-BkTS3C58.d.ts.map +1 -0
- package/dist/analyst-LsnNpSkm.js +152 -0
- package/dist/analyst-LsnNpSkm.js.map +1 -0
- package/dist/analyze-runs-C1CavBMk.js +1067 -0
- package/dist/analyze-runs-C1CavBMk.js.map +1 -0
- package/dist/analyze-runs-FsgYCinh.d.ts +72 -0
- package/dist/analyze-runs-FsgYCinh.d.ts.map +1 -0
- package/dist/attribute-vocabulary-DLJ6303h.d.ts +54 -0
- package/dist/attribute-vocabulary-DLJ6303h.d.ts.map +1 -0
- package/dist/authenticity/index.d.ts +81 -79
- package/dist/authenticity/index.d.ts.map +1 -0
- package/dist/authenticity/index.js +209 -193
- package/dist/authenticity/index.js.map +1 -1
- package/dist/baseline-HsBvw_dk.js +550 -0
- package/dist/baseline-HsBvw_dk.js.map +1 -0
- package/dist/baseline-hG3K85h4.d.ts +125 -0
- package/dist/baseline-hG3K85h4.d.ts.map +1 -0
- package/dist/belief-state/index.d.ts +448 -1205
- package/dist/belief-state/index.d.ts.map +1 -0
- package/dist/belief-state/index.js +1617 -1710
- package/dist/belief-state/index.js.map +1 -1
- package/dist/benchmarks/index.d.ts +2 -894
- package/dist/benchmarks/index.js +2 -59
- package/dist/benchmarks-DviOvUNr.js +754 -0
- package/dist/benchmarks-DviOvUNr.js.map +1 -0
- package/dist/builder-eval/index.d.ts +150 -662
- package/dist/builder-eval/index.d.ts.map +1 -0
- package/dist/builder-eval/index.js +356 -345
- package/dist/builder-eval/index.js.map +1 -1
- package/dist/calibration-CNWWA6K8.js +94 -0
- package/dist/calibration-CNWWA6K8.js.map +1 -0
- package/dist/campaign/index.d.ts +5 -6390
- package/dist/campaign/index.js +3 -212
- package/dist/campaign-CBKZvQ1H.js +3885 -0
- package/dist/campaign-CBKZvQ1H.js.map +1 -0
- package/dist/cli.d.ts +1 -1
- package/dist/cli.js +164 -174
- package/dist/cli.js.map +1 -1
- package/dist/client-C97NMzqi.d.ts +581 -0
- package/dist/client-C97NMzqi.d.ts.map +1 -0
- package/dist/client-CYzbdJOZ.js +637 -0
- package/dist/client-CYzbdJOZ.js.map +1 -0
- package/dist/code-agent-session-BjkMTQ7H.js +1390 -0
- package/dist/code-agent-session-BjkMTQ7H.js.map +1 -0
- package/dist/code-agent-session-DqqgOJaz.d.ts +143 -0
- package/dist/code-agent-session-DqqgOJaz.d.ts.map +1 -0
- package/dist/concurrency-DIxRZF_J.js +85 -0
- package/dist/concurrency-DIxRZF_J.js.map +1 -0
- package/dist/contract/index.d.ts +645 -5605
- package/dist/contract/index.d.ts.map +1 -0
- package/dist/contract/index.js +1730 -1937
- package/dist/contract/index.js.map +1 -1
- package/dist/control.d.ts +3 -1030
- package/dist/control.js +2 -32
- package/dist/cost-ledger-DIgQUFZZ.js +801 -0
- package/dist/cost-ledger-DIgQUFZZ.js.map +1 -0
- package/dist/cost-ledger-Dye6jCgg.d.ts +236 -0
- package/dist/cost-ledger-Dye6jCgg.d.ts.map +1 -0
- package/dist/dataset-BvtnC8Dc.d.ts +115 -0
- package/dist/dataset-BvtnC8Dc.d.ts.map +1 -0
- package/dist/default-registry-C-vFCSEc.js +2579 -0
- package/dist/default-registry-C-vFCSEc.js.map +1 -0
- package/dist/default-registry-CNPo-Vsb.d.ts +540 -0
- package/dist/default-registry-CNPo-Vsb.d.ts.map +1 -0
- package/dist/emitter-CPBAhxum.js +266 -0
- package/dist/emitter-CPBAhxum.js.map +1 -0
- package/dist/emitter-DGQGoLyj.d.ts +113 -0
- package/dist/emitter-DGQGoLyj.d.ts.map +1 -0
- package/dist/errors-8YnH8WlF.js +64 -0
- package/dist/errors-8YnH8WlF.js.map +1 -0
- package/dist/errors-CEk209JS.d.ts +76 -0
- package/dist/errors-CEk209JS.d.ts.map +1 -0
- package/dist/eval-campaign-DEm6c8ru.js +349 -0
- package/dist/eval-campaign-DEm6c8ru.js.map +1 -0
- package/dist/execution-tracks-CpgFPpS5.js +93 -0
- package/dist/execution-tracks-CpgFPpS5.js.map +1 -0
- package/dist/exporters-q9iL-2Jf.js +148 -0
- package/dist/exporters-q9iL-2Jf.js.map +1 -0
- package/dist/extract-usage-BrQ8mCLX.js +155 -0
- package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
- package/dist/failure-cluster-CqcvCcdR.d.ts +59 -0
- package/dist/failure-cluster-CqcvCcdR.d.ts.map +1 -0
- package/dist/feedback-trajectory-CVaeREXV.d.ts +340 -0
- package/dist/feedback-trajectory-CVaeREXV.d.ts.map +1 -0
- package/dist/fuzz.d.ts +320 -646
- package/dist/fuzz.d.ts.map +1 -0
- package/dist/fuzz.js +670 -617
- package/dist/fuzz.js.map +1 -1
- package/dist/hf-dataset-DBJXXoY1.js +763 -0
- package/dist/hf-dataset-DBJXXoY1.js.map +1 -0
- package/dist/hosted/index.d.ts +11 -831
- package/dist/hosted/index.d.ts.map +1 -0
- package/dist/hosted/index.js +2 -37
- package/dist/index-2JJSA6-r2.d.ts +926 -0
- package/dist/index-2JJSA6-r2.d.ts.map +1 -0
- package/dist/index-6N0aYmpW.d.ts +217 -0
- package/dist/index-6N0aYmpW.d.ts.map +1 -0
- package/dist/index-BAvgST_9.d.ts +131 -0
- package/dist/index-BAvgST_9.d.ts.map +1 -0
- package/dist/index-BvnJuTGD.d.ts +68 -0
- package/dist/index-BvnJuTGD.d.ts.map +1 -0
- package/dist/index-C61Wi7yg.d.ts +547 -0
- package/dist/index-C61Wi7yg.d.ts.map +1 -0
- package/dist/index-CAPUUKaM.d.ts +335 -0
- package/dist/index-CAPUUKaM.d.ts.map +1 -0
- package/dist/index-DE5fb3EC.d.ts +2244 -0
- package/dist/index-DE5fb3EC.d.ts.map +1 -0
- package/dist/index-DSC51roc.d.ts +102 -0
- package/dist/index-DSC51roc.d.ts.map +1 -0
- package/dist/index.d.ts +3776 -15120
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +11185 -11191
- package/dist/index.js.map +1 -1
- package/dist/integrity-BzRbCHzi.js +424 -0
- package/dist/integrity-BzRbCHzi.js.map +1 -0
- package/dist/integrity-rmVhXWA7.d.ts +61 -0
- package/dist/integrity-rmVhXWA7.d.ts.map +1 -0
- package/dist/judge-calibration-DFtEMlde.d.ts +146 -0
- package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
- package/dist/ledger-core/index.d.ts +2 -0
- package/dist/ledger-core/index.js +2 -0
- package/dist/ledger-core-DtZz1RG0.js +388 -0
- package/dist/ledger-core-DtZz1RG0.js.map +1 -0
- package/dist/llm-client--GR4JbZE.js +687 -0
- package/dist/llm-client--GR4JbZE.js.map +1 -0
- package/dist/llm-client-B_nIBlYo.d.ts +290 -0
- package/dist/llm-client-B_nIBlYo.d.ts.map +1 -0
- package/dist/matrix/index.d.ts +3 -155
- package/dist/matrix/index.js +2 -8
- package/dist/matrix-BzQnu2S6.js +270 -0
- package/dist/matrix-BzQnu2S6.js.map +1 -0
- package/dist/meta-eval/index.d.ts +4 -1027
- package/dist/meta-eval/index.js +393 -390
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/metrics-C9YY1OcL.js +239 -0
- package/dist/metrics-C9YY1OcL.js.map +1 -0
- package/dist/mint-yN2M2eh0.js +201 -0
- package/dist/mint-yN2M2eh0.js.map +1 -0
- package/dist/multi-layer-verifier-BHY1gWAc.d.ts +138 -0
- package/dist/multi-layer-verifier-BHY1gWAc.d.ts.map +1 -0
- package/dist/multishot/index.d.ts +273 -481
- package/dist/multishot/index.d.ts.map +1 -0
- package/dist/multishot/index.js +595 -548
- package/dist/multishot/index.js.map +1 -1
- package/dist/off-policy-DvgzvtIx.js +220 -0
- package/dist/off-policy-DvgzvtIx.js.map +1 -0
- package/dist/off-policy-mskQw8Mb.d.ts +153 -0
- package/dist/off-policy-mskQw8Mb.d.ts.map +1 -0
- package/dist/openapi.json +1 -1
- package/dist/opencode-sqlite-BGrHeDu3.js +318 -0
- package/dist/opencode-sqlite-BGrHeDu3.js.map +1 -0
- package/dist/outcome-store-BYHIuO0e.d.ts +65 -0
- package/dist/outcome-store-BYHIuO0e.d.ts.map +1 -0
- package/dist/outcome-store-ChBKlTd_.js +75 -0
- package/dist/outcome-store-ChBKlTd_.js.map +1 -0
- package/dist/paired-arms-D9D0wXj2.js +260 -0
- package/dist/paired-arms-D9D0wXj2.js.map +1 -0
- package/dist/pipelines/index.d.ts +95 -532
- package/dist/pipelines/index.d.ts.map +1 -0
- package/dist/pipelines/index.js +497 -478
- package/dist/pipelines/index.js.map +1 -1
- package/dist/pre-registration-DakwTRXk.js +96 -0
- package/dist/pre-registration-DakwTRXk.js.map +1 -0
- package/dist/propose-review-control-LhIWGzHl.js +1458 -0
- package/dist/propose-review-control-LhIWGzHl.js.map +1 -0
- package/dist/query-CJ_DX8vl.d.ts +28 -0
- package/dist/query-CJ_DX8vl.d.ts.map +1 -0
- package/dist/query-Di7eEQ79.js +83 -0
- package/dist/query-Di7eEQ79.js.map +1 -0
- package/dist/raw-provider-sink-BQd7mzyT.js +201 -0
- package/dist/raw-provider-sink-BQd7mzyT.js.map +1 -0
- package/dist/raw-provider-sink-BU29Sh8h.d.ts +134 -0
- package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +1 -0
- package/dist/redact-7Aq1ukl-.js +107 -0
- package/dist/redact-7Aq1ukl-.js.map +1 -0
- package/dist/release-report-Cz9NKH39.js +603 -0
- package/dist/release-report-Cz9NKH39.js.map +1 -0
- package/dist/release-report-mvB2G4_J.d.ts +244 -0
- package/dist/release-report-mvB2G4_J.d.ts.map +1 -0
- package/dist/replay-GnyotH0J.js +1741 -0
- package/dist/replay-GnyotH0J.js.map +1 -0
- package/dist/replay-OoidtG1E.d.ts +749 -0
- package/dist/replay-OoidtG1E.d.ts.map +1 -0
- package/dist/reporting.d.ts +6 -1298
- package/dist/reporting.js +6 -50
- package/dist/researcher-CwTdwXG1.d.ts +314 -0
- package/dist/researcher-CwTdwXG1.d.ts.map +1 -0
- package/dist/reward-hacking-eAnOsynk.d.ts +224 -0
- package/dist/reward-hacking-eAnOsynk.d.ts.map +1 -0
- package/dist/reward-hacking-qipEpKvY.js +596 -0
- package/dist/reward-hacking-qipEpKvY.js.map +1 -0
- package/dist/reward-nw2xZGZG.js +137 -0
- package/dist/reward-nw2xZGZG.js.map +1 -0
- package/dist/rl.d.ts +916 -3596
- package/dist/rl.d.ts.map +1 -0
- package/dist/rl.js +2362 -1751
- package/dist/rl.js.map +1 -1
- package/dist/rolldown-runtime-8H4AJuhK.js +14 -0
- package/dist/rollout/index.d.ts +3 -1048
- package/dist/rollout/index.js +8 -110
- package/dist/rollout-BOYjemfR.js +624 -0
- package/dist/rollout-BOYjemfR.js.map +1 -0
- package/dist/rubric-predictive-validity-B3xmbmS1.js +141 -0
- package/dist/rubric-predictive-validity-B3xmbmS1.js.map +1 -0
- package/dist/rubric-predictive-validity-Ku_clp_1.d.ts +74 -0
- package/dist/rubric-predictive-validity-Ku_clp_1.d.ts.map +1 -0
- package/dist/run-evidence-ClRX_8A9.d.ts +225 -0
- package/dist/run-evidence-ClRX_8A9.d.ts.map +1 -0
- package/dist/run-record-BuoE80Dq.js +467 -0
- package/dist/run-record-BuoE80Dq.js.map +1 -0
- package/dist/run-record-CnZu_gjl.d.ts +357 -0
- package/dist/run-record-CnZu_gjl.d.ts.map +1 -0
- package/dist/run-score-iEEAWiBY.js +41 -0
- package/dist/run-score-iEEAWiBY.js.map +1 -0
- package/dist/runtime-trajectory-1gyaTOoC.js +93 -0
- package/dist/runtime-trajectory-1gyaTOoC.js.map +1 -0
- package/dist/runtime-trajectory-BvSZcCHD.d.ts +50 -0
- package/dist/runtime-trajectory-BvSZcCHD.d.ts.map +1 -0
- package/dist/schema-BtVldJ3T.d.ts +206 -0
- package/dist/schema-BtVldJ3T.d.ts.map +1 -0
- package/dist/schema-C6DW4ZHR.js +821 -0
- package/dist/schema-C6DW4ZHR.js.map +1 -0
- package/dist/schema-CRhEY1SO.js +69 -0
- package/dist/schema-CRhEY1SO.js.map +1 -0
- package/dist/schema-Cef2cFmb.d.ts +408 -0
- package/dist/schema-Cef2cFmb.d.ts.map +1 -0
- package/dist/semantic-concept-judge-B6cWNJ2K.js +725 -0
- package/dist/semantic-concept-judge-B6cWNJ2K.js.map +1 -0
- package/dist/sequential-Br0mAPHA.js +148 -0
- package/dist/sequential-Br0mAPHA.js.map +1 -0
- package/dist/sequential-CYwq6Ff_.d.ts +141 -0
- package/dist/sequential-CYwq6Ff_.d.ts.map +1 -0
- package/dist/series-convergence-CjO2QdRW.js +43 -0
- package/dist/series-convergence-CjO2QdRW.js.map +1 -0
- package/dist/series-convergence-ofsqPWhs.d.ts +35 -0
- package/dist/series-convergence-ofsqPWhs.d.ts.map +1 -0
- package/dist/server-m5D9cvnG.js +1040 -0
- package/dist/server-m5D9cvnG.js.map +1 -0
- package/dist/skill-usage-C_pXm7lP.d.ts +534 -0
- package/dist/skill-usage-C_pXm7lP.d.ts.map +1 -0
- package/dist/skillopt-optimization-method-B7wX7XkF.d.ts +1739 -0
- package/dist/skillopt-optimization-method-B7wX7XkF.d.ts.map +1 -0
- package/dist/skillopt-optimization-method-D4ODwFVV.js +7712 -0
- package/dist/skillopt-optimization-method-D4ODwFVV.js.map +1 -0
- package/dist/statistics-Cmj6nynr.d.ts +514 -0
- package/dist/statistics-Cmj6nynr.d.ts.map +1 -0
- package/dist/statistics-CnnxdpOg.js +1437 -0
- package/dist/statistics-CnnxdpOg.js.map +1 -0
- package/dist/store-CT9YIIve.d.ts +117 -0
- package/dist/store-CT9YIIve.d.ts.map +1 -0
- package/dist/store-CxJry_cs.d.ts +229 -0
- package/dist/store-CxJry_cs.d.ts.map +1 -0
- package/dist/storyboard/index.d.ts +84 -203
- package/dist/storyboard/index.d.ts.map +1 -0
- package/dist/storyboard/index.js +609 -542
- package/dist/storyboard/index.js.map +1 -1
- package/dist/summary-report-BNs5nmXI.js +862 -0
- package/dist/summary-report-BNs5nmXI.js.map +1 -0
- package/dist/summary-report-Cj9gdw4i.d.ts +407 -0
- package/dist/summary-report-Cj9gdw4i.d.ts.map +1 -0
- package/dist/supervisor-run/index.d.ts +2 -849
- package/dist/supervisor-run/index.js +2 -64
- package/dist/supervisor-run-_lnTLM3z.js +1679 -0
- package/dist/supervisor-run-_lnTLM3z.js.map +1 -0
- package/dist/task-failure-attributes-CQZlB3et.js +311 -0
- package/dist/task-failure-attributes-CQZlB3et.js.map +1 -0
- package/dist/test-graded-scenario-BsqWLmPt.js +318 -0
- package/dist/test-graded-scenario-BsqWLmPt.js.map +1 -0
- package/dist/test-graded-scenario-D1TaI2va.d.ts +141 -0
- package/dist/test-graded-scenario-D1TaI2va.d.ts.map +1 -0
- package/dist/tools-BmuN627J.js +1085 -0
- package/dist/tools-BmuN627J.js.map +1 -0
- package/dist/trace-attributes.d.ts +2 -52
- package/dist/trace-attributes.js +131 -61
- package/dist/trace-attributes.js.map +1 -1
- package/dist/traces.d.ts +12 -2365
- package/dist/traces.js +12 -251
- package/dist/trajectory-D_7rLrvE.js +56 -0
- package/dist/trajectory-D_7rLrvE.js.map +1 -0
- package/dist/types-DGsxbAEd.d.ts +387 -0
- package/dist/types-DGsxbAEd.d.ts.map +1 -0
- package/dist/types-k9tZGKUg.d.ts +640 -0
- package/dist/types-k9tZGKUg.d.ts.map +1 -0
- package/dist/verdict-Dps8_okt.d.ts +37 -0
- package/dist/verdict-Dps8_okt.d.ts.map +1 -0
- package/dist/wire/index.d.ts +702 -1174
- package/dist/wire/index.d.ts.map +1 -0
- package/dist/wire/index.js +2 -81
- package/docs/feature-guide.md +1 -1
- package/docs/rollout.md +116 -2
- package/package.json +18 -10
- package/dist/benchmarks/index.js.map +0 -1
- package/dist/campaign/index.js.map +0 -1
- package/dist/chunk-2JX3CFMB.js +0 -695
- package/dist/chunk-2JX3CFMB.js.map +0 -1
- package/dist/chunk-2MKQIFS4.js +0 -183
- package/dist/chunk-2MKQIFS4.js.map +0 -1
- package/dist/chunk-3RF76KTD.js +0 -84
- package/dist/chunk-3RF76KTD.js.map +0 -1
- package/dist/chunk-5DTSBUL2.js +0 -159
- package/dist/chunk-5DTSBUL2.js.map +0 -1
- package/dist/chunk-7ZZMD7UK.js +0 -386
- package/dist/chunk-7ZZMD7UK.js.map +0 -1
- package/dist/chunk-BOD4O7OF.js +0 -40
- package/dist/chunk-BOD4O7OF.js.map +0 -1
- package/dist/chunk-BYT7ELPS.js +0 -1553
- package/dist/chunk-BYT7ELPS.js.map +0 -1
- package/dist/chunk-DJKY2TSY.js +0 -2428
- package/dist/chunk-DJKY2TSY.js.map +0 -1
- package/dist/chunk-DPUHNQLN.js +0 -232
- package/dist/chunk-DPUHNQLN.js.map +0 -1
- package/dist/chunk-DRYIUNWY.js +0 -622
- package/dist/chunk-DRYIUNWY.js.map +0 -1
- package/dist/chunk-EJGRPCO3.js +0 -617
- package/dist/chunk-EJGRPCO3.js.map +0 -1
- package/dist/chunk-EOSZT7PL.js +0 -2001
- package/dist/chunk-EOSZT7PL.js.map +0 -1
- package/dist/chunk-EZJEIH2R.js +0 -1559
- package/dist/chunk-EZJEIH2R.js.map +0 -1
- package/dist/chunk-GGE4NNQT.js +0 -65
- package/dist/chunk-GGE4NNQT.js.map +0 -1
- package/dist/chunk-HHWE3POT.js +0 -94
- package/dist/chunk-HHWE3POT.js.map +0 -1
- package/dist/chunk-IHQDPH7D.js +0 -171
- package/dist/chunk-IHQDPH7D.js.map +0 -1
- package/dist/chunk-JHCHEVET.js +0 -274
- package/dist/chunk-JHCHEVET.js.map +0 -1
- package/dist/chunk-K4DBDHLK.js +0 -158
- package/dist/chunk-K4DBDHLK.js.map +0 -1
- package/dist/chunk-K6N6XJJX.js +0 -306
- package/dist/chunk-K6N6XJJX.js.map +0 -1
- package/dist/chunk-MA6HLL3S.js +0 -65
- package/dist/chunk-MA6HLL3S.js.map +0 -1
- package/dist/chunk-MAZ26DC7.js +0 -99
- package/dist/chunk-MAZ26DC7.js.map +0 -1
- package/dist/chunk-MHELPNRP.js +0 -1212
- package/dist/chunk-MHELPNRP.js.map +0 -1
- package/dist/chunk-NACAGYSY.js +0 -1040
- package/dist/chunk-NACAGYSY.js.map +0 -1
- package/dist/chunk-NKAGIDE2.js +0 -7633
- package/dist/chunk-NKAGIDE2.js.map +0 -1
- package/dist/chunk-NPCTHQIO.js +0 -91
- package/dist/chunk-NPCTHQIO.js.map +0 -1
- package/dist/chunk-NYLOYM6N.js +0 -332
- package/dist/chunk-NYLOYM6N.js.map +0 -1
- package/dist/chunk-ONWEPEDO.js +0 -57
- package/dist/chunk-ONWEPEDO.js.map +0 -1
- package/dist/chunk-P5W7RQKK.js +0 -576
- package/dist/chunk-P5W7RQKK.js.map +0 -1
- package/dist/chunk-P6FYH6K4.js +0 -1161
- package/dist/chunk-P6FYH6K4.js.map +0 -1
- package/dist/chunk-PBE2LOSS.js +0 -669
- package/dist/chunk-PBE2LOSS.js.map +0 -1
- package/dist/chunk-PC4UYEBM.js +0 -166
- package/dist/chunk-PC4UYEBM.js.map +0 -1
- package/dist/chunk-PXE2VKMX.js +0 -140
- package/dist/chunk-PXE2VKMX.js.map +0 -1
- package/dist/chunk-PZ5AY32C.js +0 -10
- package/dist/chunk-PZ5AY32C.js.map +0 -1
- package/dist/chunk-RZTMDUO7.js +0 -49
- package/dist/chunk-RZTMDUO7.js.map +0 -1
- package/dist/chunk-S5YLIBFX.js +0 -136
- package/dist/chunk-S5YLIBFX.js.map +0 -1
- package/dist/chunk-SZLVEKMJ.js +0 -1446
- package/dist/chunk-SZLVEKMJ.js.map +0 -1
- package/dist/chunk-T4SQEITX.js +0 -95
- package/dist/chunk-T4SQEITX.js.map +0 -1
- package/dist/chunk-TBL77AUT.js +0 -355
- package/dist/chunk-TBL77AUT.js.map +0 -1
- package/dist/chunk-TSN7JT6D.js +0 -1646
- package/dist/chunk-TSN7JT6D.js.map +0 -1
- package/dist/chunk-TT4KNT67.js +0 -124
- package/dist/chunk-TT4KNT67.js.map +0 -1
- package/dist/chunk-UB2LOJ6Q.js +0 -4461
- package/dist/chunk-UB2LOJ6Q.js.map +0 -1
- package/dist/chunk-UWZZKKU7.js +0 -237
- package/dist/chunk-UWZZKKU7.js.map +0 -1
- package/dist/chunk-VBQ3CRKH.js +0 -291
- package/dist/chunk-VBQ3CRKH.js.map +0 -1
- package/dist/chunk-VGRCHJON.js +0 -163
- package/dist/chunk-VGRCHJON.js.map +0 -1
- package/dist/chunk-VI2UW6B6.js +0 -162
- package/dist/chunk-VI2UW6B6.js.map +0 -1
- package/dist/chunk-VLOATJQ2.js +0 -908
- package/dist/chunk-VLOATJQ2.js.map +0 -1
- package/dist/chunk-VQMK5FMP.js +0 -247
- package/dist/chunk-VQMK5FMP.js.map +0 -1
- package/dist/chunk-VZSRQ272.js +0 -149
- package/dist/chunk-VZSRQ272.js.map +0 -1
- package/dist/chunk-WGXIEX7P.js +0 -116
- package/dist/chunk-WGXIEX7P.js.map +0 -1
- package/dist/chunk-WS3NZZQQ.js +0 -929
- package/dist/chunk-WS3NZZQQ.js.map +0 -1
- package/dist/chunk-XDWDC2MP.js +0 -695
- package/dist/chunk-XDWDC2MP.js.map +0 -1
- package/dist/chunk-XPRT64IE.js +0 -766
- package/dist/chunk-XPRT64IE.js.map +0 -1
- package/dist/chunk-YJBNWCAA.js +0 -1056
- package/dist/chunk-YJBNWCAA.js.map +0 -1
- package/dist/chunk-ZET2UAYW.js +0 -89
- package/dist/chunk-ZET2UAYW.js.map +0 -1
- package/dist/chunk-ZUUWPZCV.js +0 -752
- package/dist/chunk-ZUUWPZCV.js.map +0 -1
- package/dist/control.js.map +0 -1
- package/dist/hosted/index.js.map +0 -1
- package/dist/matrix/index.js.map +0 -1
- package/dist/reporting.js.map +0 -1
- package/dist/rollout/index.js.map +0 -1
- package/dist/run-campaign-ISHFZ7FJ.js +0 -17
- package/dist/run-campaign-ISHFZ7FJ.js.map +0 -1
- package/dist/supervisor-run/index.js.map +0 -1
- package/dist/traces.js.map +0 -1
- package/dist/wire/index.js.map +0 -1
|
@@ -0,0 +1,540 @@
|
|
|
1
|
+
import { a as RunRecord, n as RunCostProvenance, u as RunTokenUsage } from "./run-record-CnZu_gjl.js";
|
|
2
|
+
import { c as CostLedgerHandle } from "./cost-ledger-Dye6jCgg.js";
|
|
3
|
+
import { A as ChatClient, m as JudgeInput } from "./types-DGsxbAEd.js";
|
|
4
|
+
import { t as TraceAnalysisStore } from "./store-CxJry_cs.js";
|
|
5
|
+
import { AxAIService, AxFunction } from "@ax-llm/ax";
|
|
6
|
+
import { z } from "zod";
|
|
7
|
+
//#region src/analyst/types.d.ts
|
|
8
|
+
/**
|
|
9
|
+
* Unified envelope every analyst emits. Schema-versioned so renderers
|
|
10
|
+
* and time-series diffs survive future field additions.
|
|
11
|
+
*/
|
|
12
|
+
interface AnalystFinding {
|
|
13
|
+
schema_version: '1.0.0';
|
|
14
|
+
/**
|
|
15
|
+
* Stable hash over identity-defining fields (analyst_id + canonical
|
|
16
|
+
* claim + area + optional subject). Two findings from two runs that
|
|
17
|
+
* "are the same finding" share this id — that's what `diffFindings`
|
|
18
|
+
* uses to compute appeared/disappeared sets across runs.
|
|
19
|
+
*/
|
|
20
|
+
finding_id: string;
|
|
21
|
+
analyst_id: string;
|
|
22
|
+
produced_at: string;
|
|
23
|
+
severity: AnalystSeverity;
|
|
24
|
+
/**
|
|
25
|
+
* Coarse classification. Renderers group by this. Free-form so
|
|
26
|
+
* domain-specific analysts can introduce categories without a
|
|
27
|
+
* schema change ('agent-reasoning', 'verification', 'cost',
|
|
28
|
+
* 'tool-use', 'safety', 'latency', 'data-quality', ...).
|
|
29
|
+
*/
|
|
30
|
+
area: string;
|
|
31
|
+
claim: string;
|
|
32
|
+
rationale?: string;
|
|
33
|
+
evidence_refs: EvidenceRef[];
|
|
34
|
+
recommended_action?: string;
|
|
35
|
+
validation_plan?: string;
|
|
36
|
+
/** 0..1 — the analyst's own confidence. Not calibrated across analysts. */
|
|
37
|
+
confidence: number;
|
|
38
|
+
/**
|
|
39
|
+
* Optional subject the finding is about — leaf id, agent id, request
|
|
40
|
+
* id. Included in finding_id when present so per-subject findings
|
|
41
|
+
* diff cleanly across runs.
|
|
42
|
+
*/
|
|
43
|
+
subject?: string;
|
|
44
|
+
/** FIREWALL provenance (docs/learning-flywheel.md): true iff this finding was
|
|
45
|
+
* lifted from a JUDGE verdict (an acceptance score), not OBSERVED from the
|
|
46
|
+
* agent's behavior. A judge-derived finding must NEVER be admitted as a
|
|
47
|
+
* steering input — that is the held-out judge leaking into the loop. Set at
|
|
48
|
+
* the lift site (createJudgeAdapter); checked by `assertNoJudgeVerdict`.
|
|
49
|
+
* Provenance, not evidence presence, is the correct discriminator: an
|
|
50
|
+
* evidence-less trace-analyst observation legitimately steers, while a judge
|
|
51
|
+
* verdict that happens to cite an artifact must not. */
|
|
52
|
+
derived_from_judge?: boolean;
|
|
53
|
+
/** Analyst-private extras; renderers ignore unless they know the analyst. */
|
|
54
|
+
metadata?: Record<string, unknown>;
|
|
55
|
+
}
|
|
56
|
+
type AnalystSeverity = 'critical' | 'high' | 'medium' | 'low' | 'info';
|
|
57
|
+
interface EvidenceRef {
|
|
58
|
+
/**
|
|
59
|
+
* Where the evidence lives. `span` and `event` refer to OTLP trace
|
|
60
|
+
* elements; `artifact` to a file inside the run's artifact tree;
|
|
61
|
+
* `finding` to another AnalystFinding (cross-analyst chaining);
|
|
62
|
+
* `metric` to a named scalar reading the renderer knows how to read.
|
|
63
|
+
*/
|
|
64
|
+
kind: 'span' | 'event' | 'artifact' | 'finding' | 'metric';
|
|
65
|
+
uri: string;
|
|
66
|
+
excerpt?: string;
|
|
67
|
+
}
|
|
68
|
+
/**
|
|
69
|
+
* The discriminator the registry uses to pass the right input.
|
|
70
|
+
* `custom` is the escape hatch — analysts that need something else
|
|
71
|
+
* (e.g. an embedding cache, a partner SDK handle) read it from
|
|
72
|
+
* `AnalystRunInputs.custom[<analyst id>]`.
|
|
73
|
+
*/
|
|
74
|
+
type AnalystInputKind = 'trace-store' | 'artifact-dir' | 'run-record' | 'judge-input' | 'custom';
|
|
75
|
+
interface AnalystCost {
|
|
76
|
+
/** `deterministic` analysts MUST NOT call the LLM. */
|
|
77
|
+
kind: 'deterministic' | 'llm';
|
|
78
|
+
/** Optional declared upper bound; the registry can enforce a budget. */
|
|
79
|
+
est_usd_per_run?: number;
|
|
80
|
+
/** Models the analyst expects to use (informational). */
|
|
81
|
+
models?: string[];
|
|
82
|
+
/** Maximum post-cancellation wait for provider usage. Model analysts default to 5 seconds. */
|
|
83
|
+
settlement_timeout_ms?: number;
|
|
84
|
+
}
|
|
85
|
+
interface AnalystRequirements {
|
|
86
|
+
/** Min number of shots / samples the analyst needs to produce signal. */
|
|
87
|
+
min_shots?: number;
|
|
88
|
+
/** Capabilities the runtime must supply (e.g. ['network', 'gpu']). */
|
|
89
|
+
capabilities?: string[];
|
|
90
|
+
}
|
|
91
|
+
/**
|
|
92
|
+
* What's passed to every analyst call. The registry resolves which
|
|
93
|
+
* field the analyst's `inputKind` selects and asserts it's present.
|
|
94
|
+
*/
|
|
95
|
+
interface AnalystRunInputs {
|
|
96
|
+
traceStore?: TraceAnalysisStore;
|
|
97
|
+
artifactDir?: string;
|
|
98
|
+
runRecord?: RunRecord;
|
|
99
|
+
judgeInput?: JudgeInput;
|
|
100
|
+
/** Keyed by analyst id; populated by callers that registered custom analysts. */
|
|
101
|
+
custom?: Record<string, unknown>;
|
|
102
|
+
}
|
|
103
|
+
interface AnalystContext {
|
|
104
|
+
runId: string;
|
|
105
|
+
/** Stable correlation id so logs from a single registry.run() share a tag. */
|
|
106
|
+
correlationId: string;
|
|
107
|
+
/** Enforced wall-clock deadline (epoch ms). */
|
|
108
|
+
deadlineMs?: number;
|
|
109
|
+
/** Per-analyst USD budget. Analysts MAY check before issuing LLM calls. */
|
|
110
|
+
budgetUsd?: number;
|
|
111
|
+
/** Shared paid-call account when the analyst runs inside a larger campaign. */
|
|
112
|
+
costLedger?: CostLedgerHandle;
|
|
113
|
+
/** Attribution phase used when writing to the shared paid-call account. */
|
|
114
|
+
costPhase?: string;
|
|
115
|
+
/**
|
|
116
|
+
* Shared chat client. Analysts that call an LLM go through this so
|
|
117
|
+
* the operator picks transport (sandbox-sdk | router | cli-bridge |
|
|
118
|
+
* direct-provider | mock) at the registry boundary without touching
|
|
119
|
+
* analyst code.
|
|
120
|
+
*/
|
|
121
|
+
chat?: ChatClient;
|
|
122
|
+
/**
|
|
123
|
+
* Findings from a prior run the operator wants the analyst to see as
|
|
124
|
+
* retrieval context. Kinds that take advantage of cross-run memory
|
|
125
|
+
* (failure-mode "I saw this cluster last run", knowledge-gap "the wiki
|
|
126
|
+
* page I asked for is still missing") render these into the actor's
|
|
127
|
+
* working set. Filtering is the operator's job: pass the slice that
|
|
128
|
+
* matches the analyst's id, or pass everything and let the kind
|
|
129
|
+
* filter. Empty / absent means no cross-run context.
|
|
130
|
+
*/
|
|
131
|
+
priorFindings?: ReadonlyArray<AnalystFinding>;
|
|
132
|
+
/**
|
|
133
|
+
* Findings emitted by analysts that completed earlier in this registry run.
|
|
134
|
+
* This is separate from `priorFindings`: upstream findings are dependency
|
|
135
|
+
* context for the current pass, while prior findings are cross-run memory.
|
|
136
|
+
* The registry populates this only when `RegistryRunOpts.chainFindings` is on.
|
|
137
|
+
*/
|
|
138
|
+
upstreamFindings?: ReadonlyArray<AnalystFinding>;
|
|
139
|
+
/**
|
|
140
|
+
* Report metered work independently of findings. This keeps an empty finding
|
|
141
|
+
* set from erasing token/cost telemetry. Multiple receipts are accumulated.
|
|
142
|
+
*/
|
|
143
|
+
recordUsage?: (receipt: AnalystUsageReceipt) => void;
|
|
144
|
+
/** Free-form runtime tags (env, host, op). Findings can echo these into metadata. */
|
|
145
|
+
tags?: Record<string, string>;
|
|
146
|
+
/** Logger callback — analysts SHOULD prefer this over console.* for testability. */
|
|
147
|
+
log?: (msg: string, fields?: Record<string, unknown>) => void;
|
|
148
|
+
/** Optional abort signal. Analysts SHOULD pass it through to LLM calls. */
|
|
149
|
+
signal?: AbortSignal;
|
|
150
|
+
}
|
|
151
|
+
/**
|
|
152
|
+
* The minimal contract. Concrete analysts can refine `TInput` so
|
|
153
|
+
* implementations stay type-safe (e.g. a trace analyst's `TInput` is
|
|
154
|
+
* `TraceAnalysisStore`); the registry passes the right field from
|
|
155
|
+
* `AnalystRunInputs` based on `inputKind`.
|
|
156
|
+
*/
|
|
157
|
+
interface Analyst<TInput = unknown> {
|
|
158
|
+
/** Stable identifier — appears in finding_id, telemetry, and registry exclusion lists. */
|
|
159
|
+
readonly id: string;
|
|
160
|
+
/** Human-readable. One sentence. */
|
|
161
|
+
readonly description: string;
|
|
162
|
+
readonly inputKind: AnalystInputKind;
|
|
163
|
+
readonly cost: AnalystCost;
|
|
164
|
+
readonly requires?: AnalystRequirements;
|
|
165
|
+
/** Bump on breaking changes to claim wording or area so old finding_ids don't collide. */
|
|
166
|
+
readonly version: string;
|
|
167
|
+
analyze(input: TInput, ctx: AnalystContext): Promise<AnalystFinding[]>;
|
|
168
|
+
}
|
|
169
|
+
/** Metered work performed by one analyst call. */
|
|
170
|
+
interface AnalystUsageReceipt {
|
|
171
|
+
/** Number of model-usage records observed at the provider boundary. */
|
|
172
|
+
calls: number | null;
|
|
173
|
+
/** Null when the provider did not return token accounting. */
|
|
174
|
+
tokens: RunTokenUsage | null;
|
|
175
|
+
/** Observed, estimated, or explicitly uncaptured dollar cost. */
|
|
176
|
+
cost: RunCostProvenance;
|
|
177
|
+
/** Known lower bound when one or more calls have uncaptured cost. */
|
|
178
|
+
knownCostUsd?: number;
|
|
179
|
+
}
|
|
180
|
+
/**
|
|
181
|
+
* Compute the stable finding_id from the identity-defining fields.
|
|
182
|
+
* Default implementation hashes {analyst_id, area, subject, normalized claim}.
|
|
183
|
+
* Analysts that emit findings whose claim text varies per run (timestamps,
|
|
184
|
+
* counts) SHOULD either: (a) pass an explicit `id_basis` to fix the hash,
|
|
185
|
+
* or (b) move the variable part into `rationale`/`metadata` and keep the
|
|
186
|
+
* `claim` static.
|
|
187
|
+
*/
|
|
188
|
+
declare function computeFindingId(input: {
|
|
189
|
+
analyst_id: string;
|
|
190
|
+
area: string;
|
|
191
|
+
subject?: string;
|
|
192
|
+
claim: string;
|
|
193
|
+
/** Override the claim for hashing — use when the displayed claim has run-specific bits. */
|
|
194
|
+
id_basis?: string;
|
|
195
|
+
}): string;
|
|
196
|
+
/**
|
|
197
|
+
* Convenience factory: produce a fully-formed AnalystFinding with the
|
|
198
|
+
* id computed automatically. Analyst code stays terse.
|
|
199
|
+
*/
|
|
200
|
+
declare function makeFinding(init: Omit<AnalystFinding, 'schema_version' | 'finding_id' | 'produced_at'> & {
|
|
201
|
+
id_basis?: string;
|
|
202
|
+
produced_at?: string;
|
|
203
|
+
}): AnalystFinding;
|
|
204
|
+
interface AnalystRunSummary {
|
|
205
|
+
analyst_id: string;
|
|
206
|
+
status: 'ok' | 'skipped' | 'failed';
|
|
207
|
+
/** Why skipped — missing input, budget exceeded, capability unmet. */
|
|
208
|
+
reason?: string;
|
|
209
|
+
findings_count: number;
|
|
210
|
+
latency_ms: number;
|
|
211
|
+
/** Additive model usage and cost provenance for this analyst. */
|
|
212
|
+
usage: AnalystUsageReceipt;
|
|
213
|
+
/** When `status='failed'`: the error class + message, never the full stack. */
|
|
214
|
+
error?: {
|
|
215
|
+
class: string;
|
|
216
|
+
message: string;
|
|
217
|
+
};
|
|
218
|
+
}
|
|
219
|
+
interface AnalystRunResult {
|
|
220
|
+
run_id: string;
|
|
221
|
+
correlation_id: string;
|
|
222
|
+
started_at: string;
|
|
223
|
+
ended_at: string;
|
|
224
|
+
findings: AnalystFinding[];
|
|
225
|
+
per_analyst: AnalystRunSummary[];
|
|
226
|
+
/** Total LLM cost in USD across all analysts in this registry.run(). */
|
|
227
|
+
total_cost_usd: number;
|
|
228
|
+
/**
|
|
229
|
+
* Provenance for `total_cost_usd`. When uncaptured, the numeric field is only
|
|
230
|
+
* the known subtotal and must not be treated as the run's total spend.
|
|
231
|
+
*/
|
|
232
|
+
total_cost_provenance?: RunCostProvenance;
|
|
233
|
+
}
|
|
234
|
+
/**
|
|
235
|
+
* Events emitted by `AnalystRegistry.runStream(...)` in real time as
|
|
236
|
+
* the registry executes. UIs subscribe via `for await (const ev of
|
|
237
|
+
* registry.runStream(...))`; `registry.run(...)` is a thin collector
|
|
238
|
+
* over the same stream, so the two surfaces share their invariants.
|
|
239
|
+
*
|
|
240
|
+
* Per-finding events are intentionally omitted — analyzers are batch
|
|
241
|
+
* operations (an Ax actor returns the full `findings:json[]` at the
|
|
242
|
+
* end of the responder), so streaming inside one analyst would only
|
|
243
|
+
* emit partial JSON consumers can't render. The kind-completion event
|
|
244
|
+
* is the right granularity; subscribers wanting per-finding rendering
|
|
245
|
+
* iterate `event.findings` themselves.
|
|
246
|
+
*/
|
|
247
|
+
type AnalystRunEvent = {
|
|
248
|
+
type: 'run-started';
|
|
249
|
+
run_id: string;
|
|
250
|
+
correlation_id: string;
|
|
251
|
+
started_at: string;
|
|
252
|
+
/** The ordered list of analyst ids the registry will run. */
|
|
253
|
+
analyst_ids: ReadonlyArray<string>;
|
|
254
|
+
} | {
|
|
255
|
+
type: 'analyst-skipped';
|
|
256
|
+
summary: AnalystRunSummary;
|
|
257
|
+
} | {
|
|
258
|
+
type: 'analyst-started';
|
|
259
|
+
analyst_id: string;
|
|
260
|
+
started_at: string;
|
|
261
|
+
} | {
|
|
262
|
+
type: 'analyst-completed';
|
|
263
|
+
/** `summary.status` is `'ok'` for clean completion or `'failed'` for thrown analysts. */
|
|
264
|
+
summary: AnalystRunSummary;
|
|
265
|
+
findings: ReadonlyArray<AnalystFinding>;
|
|
266
|
+
} | {
|
|
267
|
+
type: 'run-completed';
|
|
268
|
+
result: AnalystRunResult;
|
|
269
|
+
};
|
|
270
|
+
//#endregion
|
|
271
|
+
//#region src/analyst/finding-signature.d.ts
|
|
272
|
+
declare const ANALYST_SEVERITIES: readonly ['critical', 'high', 'medium', 'low', 'info'];
|
|
273
|
+
declare const RawAnalystEvidenceSchema: z.ZodObject<{
|
|
274
|
+
uri: z.ZodString;
|
|
275
|
+
excerpt: z.ZodOptional<z.ZodString>;
|
|
276
|
+
}, z.core.$strict>;
|
|
277
|
+
type RawAnalystEvidence = z.infer<typeof RawAnalystEvidenceSchema>;
|
|
278
|
+
declare const RawAnalystFindingSchema: z.ZodObject<{
|
|
279
|
+
severity: z.ZodEnum<{
|
|
280
|
+
critical: "critical";
|
|
281
|
+
high: "high";
|
|
282
|
+
info: "info";
|
|
283
|
+
low: "low";
|
|
284
|
+
medium: "medium";
|
|
285
|
+
}>;
|
|
286
|
+
claim: z.ZodString;
|
|
287
|
+
subject: z.ZodOptional<z.ZodString>;
|
|
288
|
+
confidence: z.ZodNumber;
|
|
289
|
+
rationale: z.ZodOptional<z.ZodString>;
|
|
290
|
+
recommended_action: z.ZodOptional<z.ZodString>;
|
|
291
|
+
evidence: z.ZodArray<z.ZodObject<{
|
|
292
|
+
uri: z.ZodString;
|
|
293
|
+
excerpt: z.ZodOptional<z.ZodString>;
|
|
294
|
+
}, z.core.$strict>>;
|
|
295
|
+
}, z.core.$strict>;
|
|
296
|
+
type RawAnalystFinding = z.infer<typeof RawAnalystFindingSchema>;
|
|
297
|
+
/**
|
|
298
|
+
* Description embedded into the actor prompt so the LLM knows what
|
|
299
|
+
* shape to emit. Kept here so kinds share one source of truth rather
|
|
300
|
+
* than restating the schema in every prompt.
|
|
301
|
+
*/
|
|
302
|
+
declare const RAW_FINDING_SCHEMA_PROMPT = "Each finding MUST be a strict JSON object with:\n - severity: \"critical\" | \"high\" | \"medium\" | \"low\" | \"info\"\n - claim: one-sentence statement (max 2000 chars)\n - subject?: one exact subject form listed by this kind; omit rather than guess\n - evidence: REQUIRED non-empty array of {\"uri\": string, \"excerpt\"?: string}. Use real identifiers with span://, event://, artifact://, metric://, or finding://. Include a short exact quote in excerpt when available. If nothing is citable, do not emit the finding.\n - confidence: number 0..1 (0.9+ exact evidence; 0.6-0.8 inferred pattern; <0.5 speculative)\n - rationale?: one or two reasoning sentences\n - recommended_action?: concrete imperative change; omit for descriptive findings\n\nUnknown fields are rejected. Do not emit area; the factory assigns it. Emit [] when there are no findings. Never fabricate evidence.";
|
|
303
|
+
/** Convert raw citations into the public finding evidence envelope. */
|
|
304
|
+
declare function evidenceRefsFromRawFinding(finding: RawAnalystFinding): EvidenceRef[];
|
|
305
|
+
declare function parseRawFinding(row: unknown, log?: (msg: string, fields?: Record<string, unknown>) => void): RawAnalystFinding | null;
|
|
306
|
+
//#endregion
|
|
307
|
+
//#region src/analyst/kind-factory.d.ts
|
|
308
|
+
/**
|
|
309
|
+
* Per-kind specification. The factory turns this into a regular
|
|
310
|
+
* `Analyst<TraceAnalysisStore>` ready for `AnalystRegistry.register()`.
|
|
311
|
+
*/
|
|
312
|
+
interface TraceAnalystKindSpec {
|
|
313
|
+
/** Stable id. Appears in finding_id, telemetry, and registry exclusions. */
|
|
314
|
+
id: string;
|
|
315
|
+
/** One-sentence description shown in `registry.list()`. */
|
|
316
|
+
description: string;
|
|
317
|
+
/** Coarse classification stamped on every emitted finding (`failure-mode`, `knowledge-gap`, ...). */
|
|
318
|
+
area: string;
|
|
319
|
+
/** Bump on any breaking change to the actor prompt or output schema. */
|
|
320
|
+
version: string;
|
|
321
|
+
/** Actor system prompt. Must instruct the LLM to emit `findings` per the schema. */
|
|
322
|
+
actorDescription: string;
|
|
323
|
+
/** Tool functions the actor may call. Pick narrow subsets via `ANALYST_TOOL_GROUPS`. */
|
|
324
|
+
buildTools: (store: TraceAnalysisStore) => AxFunction[];
|
|
325
|
+
/** Bounded semantic subqueries. `maxCalls: 0` disables model fan-out. */
|
|
326
|
+
subqueries?: {
|
|
327
|
+
maxCalls: number;
|
|
328
|
+
maxParallel?: number;
|
|
329
|
+
};
|
|
330
|
+
/** Actor turn cap. Default 12. */
|
|
331
|
+
maxTurns?: number;
|
|
332
|
+
/** Runtime char cap. Default 6000. */
|
|
333
|
+
maxRuntimeChars?: number;
|
|
334
|
+
/** Maximum output tokens for every actor and subquery model call. Default 4096. */
|
|
335
|
+
maxOutputTokens?: number;
|
|
336
|
+
/** Cost classification surfaced in `registry.list()` and budget enforcement. */
|
|
337
|
+
cost: AnalystCost;
|
|
338
|
+
/** Per-finding-row hook — kinds may reject / rewrite before lifting. */
|
|
339
|
+
postProcess?: (row: RawAnalystFinding, ctx: AnalystContext) => RawAnalystFinding | null;
|
|
340
|
+
/** Minimum citations per finding. Default 1; rows below it are rejected. */
|
|
341
|
+
minimumEvidenceCitations?: number;
|
|
342
|
+
/** Optional optimizer hook — populated when a kind wants to fit its prompt against labeled examples. */
|
|
343
|
+
goldens?: TraceAnalystGolden[];
|
|
344
|
+
}
|
|
345
|
+
/**
|
|
346
|
+
* One labeled example consumed by Ax optimizers (MIPRO / GEPA / Bootstrap).
|
|
347
|
+
* Each input is the same `{question}` an analyst would receive; `expected`
|
|
348
|
+
* is the ground-truth finding set a fitted prompt should produce on this
|
|
349
|
+
* input. Metric: kind-specific (default: F1 on `finding_id` overlap).
|
|
350
|
+
*/
|
|
351
|
+
interface TraceAnalystGolden {
|
|
352
|
+
question: string;
|
|
353
|
+
expected: ReadonlyArray<Omit<RawAnalystFinding, 'confidence'>>;
|
|
354
|
+
}
|
|
355
|
+
interface CreateTraceAnalystKindOpts {
|
|
356
|
+
/** AxAIService bound at registration time. */
|
|
357
|
+
ai: AxAIService;
|
|
358
|
+
/** Required unless `ai` was created by {@link createAnalystAi}. */
|
|
359
|
+
model?: string;
|
|
360
|
+
/** Override the spec's `version` (e.g. when an optimizer has fitted a new prompt). */
|
|
361
|
+
versionSuffix?: string;
|
|
362
|
+
/**
|
|
363
|
+
* Optional two-phase recovery: when the agentic harvest is empty but the
|
|
364
|
+
* actor produced a substantive free-form `report`, extract findings from that
|
|
365
|
+
* prose via a tolerant chat-completions pass (`structureFindings`) — no
|
|
366
|
+
* strict-emission contract, so it works on weak models. Omit to leave the
|
|
367
|
+
* actor's harvest as-is (the report is still surfaced fail-loud either way).
|
|
368
|
+
*/
|
|
369
|
+
recovery?: {
|
|
370
|
+
baseUrl: string;
|
|
371
|
+
apiKey?: string;
|
|
372
|
+
model?: string;
|
|
373
|
+
fetchImpl?: typeof fetch;
|
|
374
|
+
};
|
|
375
|
+
/** Maximum post-cancellation wait for a provider receipt. Default 5 seconds. */
|
|
376
|
+
settlementTimeoutMs?: number;
|
|
377
|
+
}
|
|
378
|
+
/**
|
|
379
|
+
* Build an `Analyst<TraceAnalysisStore>` from a kind spec.
|
|
380
|
+
*
|
|
381
|
+
* Lifts the Ax pipeline once at registration time so the registry
|
|
382
|
+
* gets a stateless analyst. The Ax agent is freshly constructed per
|
|
383
|
+
* `analyze()` call (the agent carries chat-log + usage state we don't
|
|
384
|
+
* want shared across analyst runs).
|
|
385
|
+
*/
|
|
386
|
+
declare function createTraceAnalystKind(spec: TraceAnalystKindSpec, opts: CreateTraceAnalystKindOpts): Analyst<TraceAnalysisStore>;
|
|
387
|
+
/**
|
|
388
|
+
* Render a compact prior-findings block the actor reads alongside its
|
|
389
|
+
* brief. Each row is one line so the actor can scan dozens cheaply.
|
|
390
|
+
* The kind's prompt instructs the actor to (a) check whether a new
|
|
391
|
+
* cluster matches a prior `finding_id` (carry the id forward via
|
|
392
|
+
* `id_basis` to keep diffs stable) and (b) raise severity / confidence
|
|
393
|
+
* when a prior finding has reappeared without remediation.
|
|
394
|
+
*
|
|
395
|
+
* Returns the empty string when there are no prior findings — most
|
|
396
|
+
* runs are "first-of-its-kind" and the prompt stays unchanged.
|
|
397
|
+
*
|
|
398
|
+
* Exported for tests + for consumers that build their own actor
|
|
399
|
+
* prompts (e.g. specialized analysts living outside the default kinds).
|
|
400
|
+
*/
|
|
401
|
+
declare function renderPriorFindings(prior: AnalystContext['priorFindings']): string;
|
|
402
|
+
/** Render findings produced earlier in this same registry run. */
|
|
403
|
+
declare function renderUpstreamFindings(upstream: AnalystContext['upstreamFindings']): string;
|
|
404
|
+
//#endregion
|
|
405
|
+
//#region src/analyst/registry.d.ts
|
|
406
|
+
interface AnalystHooks {
|
|
407
|
+
/** Before analyze() — last chance to mutate ctx (e.g. inject tags, override budget). */
|
|
408
|
+
onBeforeAnalyze?(args: {
|
|
409
|
+
analyst: Analyst;
|
|
410
|
+
ctx: AnalystContext;
|
|
411
|
+
runId: string;
|
|
412
|
+
}): void | Promise<void>;
|
|
413
|
+
/** After every analyst (ok | failed | skipped). Use for telemetry, ingestion, rotation. */
|
|
414
|
+
onAfterAnalyze?(args: {
|
|
415
|
+
analyst: Analyst;
|
|
416
|
+
summary: AnalystRunSummary;
|
|
417
|
+
findings: AnalystFinding[];
|
|
418
|
+
runId: string;
|
|
419
|
+
}): void | Promise<void>;
|
|
420
|
+
/**
|
|
421
|
+
* On analyst exception. Hook MAY return findings to convert the
|
|
422
|
+
* error into structured findings; the summary still reports 'failed'.
|
|
423
|
+
* Return void to keep the default empty-findings behavior.
|
|
424
|
+
*/
|
|
425
|
+
onError?(args: {
|
|
426
|
+
analyst: Analyst;
|
|
427
|
+
error: Error;
|
|
428
|
+
runId: string;
|
|
429
|
+
}): AnalystFinding[] | undefined | Promise<AnalystFinding[] | undefined>;
|
|
430
|
+
/** Once after registry.run() completes. Use for final aggregation, persistence. */
|
|
431
|
+
onComplete?(args: {
|
|
432
|
+
result: AnalystRunResult;
|
|
433
|
+
}): void | Promise<void>;
|
|
434
|
+
}
|
|
435
|
+
interface BudgetPolicy {
|
|
436
|
+
/** Overall USD cap across the registry.run(). */
|
|
437
|
+
totalUsd?: number;
|
|
438
|
+
/** Per-analyst weight for the default allocator. Missing ids get weight 1. */
|
|
439
|
+
weights?: Record<string, number>;
|
|
440
|
+
/**
|
|
441
|
+
* Custom allocator — receives the analyst, remaining/total budget, and
|
|
442
|
+
* the count of analysts that will run. Returns the per-analyst budget
|
|
443
|
+
* (or undefined only when the run has no overall cap). Overrides weights
|
|
444
|
+
* when set.
|
|
445
|
+
*/
|
|
446
|
+
allocate?: (args: {
|
|
447
|
+
analyst: Analyst;
|
|
448
|
+
totalUsd: number | undefined;
|
|
449
|
+
remainingUsd: number | undefined;
|
|
450
|
+
runningCount: number;
|
|
451
|
+
}) => number | undefined;
|
|
452
|
+
}
|
|
453
|
+
interface AnalystRegistryOptions {
|
|
454
|
+
/** Shared chat client passed to every LLM analyst via AnalystContext. */
|
|
455
|
+
chat?: ChatClient;
|
|
456
|
+
/** Logger callback. Defaults to a no-op. */
|
|
457
|
+
log?: (msg: string, fields?: Record<string, unknown>) => void;
|
|
458
|
+
/** Hooks invoked around analyze() — observability + customization seam. */
|
|
459
|
+
hooks?: AnalystHooks;
|
|
460
|
+
/** Default budget when run() doesn't override. */
|
|
461
|
+
defaultBudget?: BudgetPolicy;
|
|
462
|
+
}
|
|
463
|
+
interface RegistryRunOpts {
|
|
464
|
+
/** Restrict to a subset of registered analysts by id. */
|
|
465
|
+
only?: string[];
|
|
466
|
+
/** Skip these analysts even if registered. Useful for cheap iteration. */
|
|
467
|
+
skip?: string[];
|
|
468
|
+
/** Budget policy — totalUsd + optional weights/allocator. Falls back to options.defaultBudget. */
|
|
469
|
+
budget?: BudgetPolicy;
|
|
470
|
+
/** Active-work cap for the complete registry run. Model receipt settlement may follow. */
|
|
471
|
+
timeoutMs?: number;
|
|
472
|
+
/** Abort signal — forwarded into every analyst's context. */
|
|
473
|
+
signal?: AbortSignal;
|
|
474
|
+
/** Shared paid-call account forwarded to every analyst. */
|
|
475
|
+
costLedger?: CostLedgerHandle;
|
|
476
|
+
/** Attribution phase for calls written to `costLedger`. */
|
|
477
|
+
costPhase?: string;
|
|
478
|
+
/** Tags echoed into AnalystContext.tags — useful for tracking environment/version in findings. */
|
|
479
|
+
tags?: Record<string, string>;
|
|
480
|
+
/**
|
|
481
|
+
* Prior-run findings made available as retrieval context to every
|
|
482
|
+
* analyst via `ctx.priorFindings`. The registry forwards the slice
|
|
483
|
+
* whose `analyst_id` matches each registered analyst so a kind sees
|
|
484
|
+
* only its own history. Pass `{ '*': findings }` to broadcast to
|
|
485
|
+
* every analyst (useful when several kinds share the same historical
|
|
486
|
+
* context). For findings from this run, use `chainFindings` instead.
|
|
487
|
+
*/
|
|
488
|
+
priorFindings?: ReadonlyArray<AnalystFinding> | Record<string, ReadonlyArray<AnalystFinding>>;
|
|
489
|
+
/**
|
|
490
|
+
* Pass findings produced earlier in this registry run to each later analyst
|
|
491
|
+
* via `ctx.upstreamFindings`. Registration order is dependency order.
|
|
492
|
+
* Disabled by default because independent analyst suites must opt in.
|
|
493
|
+
*/
|
|
494
|
+
chainFindings?: boolean;
|
|
495
|
+
}
|
|
496
|
+
declare class AnalystRegistry {
|
|
497
|
+
private readonly analysts;
|
|
498
|
+
private readonly options;
|
|
499
|
+
constructor(options?: AnalystRegistryOptions);
|
|
500
|
+
register(analyst: Analyst): void;
|
|
501
|
+
list(): ReadonlyArray<{
|
|
502
|
+
id: string;
|
|
503
|
+
description: string;
|
|
504
|
+
version: string;
|
|
505
|
+
cost: Analyst['cost'];
|
|
506
|
+
}>;
|
|
507
|
+
run(runId: string, inputs: AnalystRunInputs, runOpts?: RegistryRunOpts): Promise<AnalystRunResult>;
|
|
508
|
+
/**
|
|
509
|
+
* Streaming counterpart to `run()`. Emits `AnalystRunEvent` values
|
|
510
|
+
* in real time — `run-started`, then per-analyst `skipped` /
|
|
511
|
+
* `started` / `completed`, then a terminal `run-completed` whose
|
|
512
|
+
* payload is the full `AnalystRunResult`. UIs use this to render
|
|
513
|
+
* progress; persistence consumers use `run()` and read the result.
|
|
514
|
+
*
|
|
515
|
+
* Hooks (`onBeforeAnalyze` / `onAfterAnalyze` / `onError` /
|
|
516
|
+
* `onComplete`) fire as before — streaming is additive, not a hook
|
|
517
|
+
* replacement.
|
|
518
|
+
*/
|
|
519
|
+
runStream(runId: string, inputs: AnalystRunInputs, runOpts?: RegistryRunOpts): AsyncGenerator<AnalystRunEvent, void, void>;
|
|
520
|
+
private selectAnalysts;
|
|
521
|
+
private routeInput;
|
|
522
|
+
}
|
|
523
|
+
//#endregion
|
|
524
|
+
//#region src/analyst/default-registry.d.ts
|
|
525
|
+
interface DefaultAnalystRegistryOptions {
|
|
526
|
+
/** Ax service for the agentic RLM kinds. Omit → only the deterministic analyst. */
|
|
527
|
+
ai?: AxAIService;
|
|
528
|
+
/** Required unless `ai` was created by `createAnalystAi`. */
|
|
529
|
+
model?: string;
|
|
530
|
+
/** Which agentic kinds to register when `ai` is present. Default = the shipped suite. */
|
|
531
|
+
kinds?: readonly TraceAnalystKindSpec[];
|
|
532
|
+
/** Set false to omit the deterministic behavioral analyst (default: include). */
|
|
533
|
+
includeBehavioral?: boolean;
|
|
534
|
+
/** Forwarded to the AnalystRegistry constructor (signal, tags, priorFindings). */
|
|
535
|
+
registry?: AnalystRegistryOptions;
|
|
536
|
+
}
|
|
537
|
+
declare function buildDefaultAnalystRegistry(opts?: DefaultAnalystRegistryOptions): AnalystRegistry;
|
|
538
|
+
//#endregion
|
|
539
|
+
export { AnalystRunResult as A, AnalystContext as C, AnalystRequirements as D, AnalystInputKind as E, computeFindingId as F, makeFinding as I, AnalystSeverity as M, AnalystUsageReceipt as N, AnalystRunEvent as O, EvidenceRef as P, Analyst as S, AnalystFinding as T, RawAnalystEvidenceSchema as _, AnalystRegistryOptions as a, evidenceRefsFromRawFinding as b, CreateTraceAnalystKindOpts as c, createTraceAnalystKind as d, renderPriorFindings as f, RawAnalystEvidence as g, RAW_FINDING_SCHEMA_PROMPT as h, AnalystRegistry as i, AnalystRunSummary as j, AnalystRunInputs as k, TraceAnalystGolden as l, ANALYST_SEVERITIES as m, buildDefaultAnalystRegistry as n, BudgetPolicy as o, renderUpstreamFindings as p, AnalystHooks as r, RegistryRunOpts as s, DefaultAnalystRegistryOptions as t, TraceAnalystKindSpec as u, RawAnalystFinding as v, AnalystCost as w, parseRawFinding as x, RawAnalystFindingSchema as y };
|
|
540
|
+
//# sourceMappingURL=default-registry-CNPo-Vsb.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"default-registry-CNPo-Vsb.d.ts","names":[],"sources":["../src/analyst/types.ts","../src/analyst/finding-signature.ts","../src/analyst/kind-factory.ts","../src/analyst/registry.ts","../src/analyst/default-registry.ts"],"mappings":";;;;;;;;;;;UA4BiB;EACf;;;;;;;EAOA;EACA;EACA;EACA,UAAU;;;;;;;EAOV;EACA;EACA;EACA,eAAe;EACf;EACA;;EAEA;;;;;;EAMA;;;;;;;;;EASA;;EAEA,WAAW;;KAGD;UAEK;;;;;;;EAOf;EACA;EACA;;;;;;;;KAWU;UAOK;;EAEf;;EAEA;;EAEA;;EAEA;;UAGe;;EAEf;;EAEA;;;;;;UAOe;EACf,aAAa;EACb;EACA,YAAY;EACZ,aAAa;;EAEb,SAAS;;UAGM;EACf;;EAEA;;EAEA;;EAEA;;EAEA,aAAa;;EAEb;;;;;;;EAOA,OAAO;;;;;;;;;;EAUP,gBAAgB,cAAc;;;;;;;EAO9B,mBAAmB,cAAc;;;;;EAKjC,eAAe,SAAS;;EAExB,OAAO;;EAEP,OAAO,aAAa,SAAS;;EAE7B,SAAS;;;;;;;;UASM,QAAQ;;WAEd;;WAEA;WACA,WAAW;WACX,MAAM;WACN,WAAW;;WAEX;EACT,QAAQ,OAAO,QAAQ,KAAK,iBAAiB,QAAQ;;;UAItC;;EAEf;;EAEA,QAAQ;;EAER,MAAM;;EAEN;;;;;;;;;;iBAac,iBAAiB;EAC/B;EACA;EACA;EACA;;EAEA;;;;;;iBAyBc,YACd,MAAM,KAAK;EACT;EACA;IAED;UAkBc;EACf;EACA;;EAEA;EACA;EACA;;EAEA,OAAO;;EAEP;IAAU;IAAe;;;UAGV;EACf;EACA;EACA;EACA;EACA,UAAU;EACV,aAAa;;EAEb;;;;;EAKA,wBAAwB;;;;;;;;;;;;;;;KAkBd;EAEN;EACA;EACA;EACA;;EAEA,aAAa;;EAGb;EACA,SAAS;;EAGT;EACA;EACA;;EAGA;;EAEA,SAAS;EACT,UAAU,cAAc;;EAGxB;EACA,QAAQ;;;;cCxUD;cAEA,0BAAwB,EAAA;;;GAK1B,EAAA,KAAA;KAEC,qBAAqB,EAAE,aAAa;cAiBnC,yBAAuB,EAAA;;;;;;;;;;;;;;;;;GAKzB,EAAA,KAAA;KAEC,oBAAoB,EAAE,aAAa;;;;;;cAOlC;;iBAYG,2BAA2B,SAAS,oBAAoB;iBAQxD,gBACd,cACA,OAAO,aAAa,SAAS,mCAC5B;;;;;;;UCnCc;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA,aAAa,OAAO,uBAAuB;;EAE3C;IAAe;IAAkB;;;EAEjC;;EAEA;;EAEA;;EAEA,MAAM;;EAEN,eAAe,KAAK,mBAAmB,KAAK,mBAAmB;;EAE/D;;EAEA,UAAU;;;;;;;;UASK;EACf;EACA,UAAU,cAAc,KAAK;;UAGd;;EAEf,IAAI;;EAEJ;;EAEA;;;;;;;;EAQA;IAAa;IAAiB;IAAiB;IAAgB,mBAAmB;;;EAElF;;;;;;;;;;iBAWc,uBACd,MAAM,sBACN,MAAM,6BACL,QAAQ;;;;;;;;;;;;;;;iBAgRK,oBAAoB,OAAO;;iBAyB3B,uBAAuB,UAAU;;;UC3XhC;;EAEf,iBAAiB;IACf,SAAS;IACT,KAAK;IACL;aACS;;EAEX,gBAAgB;IACd,SAAS;IACT,SAAS;IACT,UAAU;IACV;aACS;;;;;;EAMX,SAAS;IACP,SAAS;IACT,OAAO;IACP;MACE,+BAA+B,QAAQ;;EAE3C,YAAY;IAAQ,QAAQ;aAA4B;;UAGzC;;EAEf;;EAEA,UAAU;;;;;;;EAOV,YAAY;IACV,SAAS;IACT;IACA;IACA;;;UAIa;;EAEf,OAAO;;EAEP,OAAO,aAAa,SAAS;;EAE7B,QAAQ;;EAER,gBAAgB;;UAGD;;EAEf;;EAEA;;EAEA,SAAS;;EAET;;EAEA,SAAS;;EAET,aAAa;;EAEb;;EAEA,OAAO;;;;;;;;;EASP,gBAAgB,cAAc,kBAAkB,eAAe,cAAc;;;;;;EAM7E;;cAGW;mBACM;mBACA;EAEjB,YAAY,UAAS;EAIrB,SAAS,SAAS;EAmBlB,QAAQ;IACN;IACA;IACA;IACA,MAAM;;EAUF,IACJ,eACA,QAAQ,kBACR,UAAS,kBACR,QAAQ;;;;;;;;;;;;EAoBJ,UACL,eACA,QAAQ,kBACR,UAAS,kBACR,eAAe;UA0OV;UAaA;;;;UC5aO;;EAEf,KAAK;;EAEL;;EAEA,iBAAiB;;EAEjB;;EAEA,WAAW;;iBAGG,4BACd,OAAM,gCACL"}
|