@tangle-network/agent-eval 0.129.0 → 0.130.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +14 -0
- package/README.md +1 -0
- package/dist/active-curriculum-C4mk67HP.js +214 -0
- package/dist/active-curriculum-C4mk67HP.js.map +1 -0
- package/dist/adversarial-smnADNFS.d.ts +21 -0
- package/dist/adversarial-smnADNFS.d.ts.map +1 -0
- package/dist/analyst/index.d.ts +81 -2872
- package/dist/analyst/index.d.ts.map +1 -0
- package/dist/analyst/index.js +319 -360
- package/dist/analyst/index.js.map +1 -1
- package/dist/analyst-BkTS3C58.d.ts +89 -0
- package/dist/analyst-BkTS3C58.d.ts.map +1 -0
- package/dist/analyst-LsnNpSkm.js +152 -0
- package/dist/analyst-LsnNpSkm.js.map +1 -0
- package/dist/analyze-runs-C1CavBMk.js +1067 -0
- package/dist/analyze-runs-C1CavBMk.js.map +1 -0
- package/dist/analyze-runs-FsgYCinh.d.ts +72 -0
- package/dist/analyze-runs-FsgYCinh.d.ts.map +1 -0
- package/dist/attribute-vocabulary-DLJ6303h.d.ts +54 -0
- package/dist/attribute-vocabulary-DLJ6303h.d.ts.map +1 -0
- package/dist/authenticity/index.d.ts +81 -79
- package/dist/authenticity/index.d.ts.map +1 -0
- package/dist/authenticity/index.js +209 -193
- package/dist/authenticity/index.js.map +1 -1
- package/dist/baseline-HsBvw_dk.js +550 -0
- package/dist/baseline-HsBvw_dk.js.map +1 -0
- package/dist/baseline-hG3K85h4.d.ts +125 -0
- package/dist/baseline-hG3K85h4.d.ts.map +1 -0
- package/dist/belief-state/index.d.ts +448 -1188
- package/dist/belief-state/index.d.ts.map +1 -0
- package/dist/belief-state/index.js +1617 -1709
- package/dist/belief-state/index.js.map +1 -1
- package/dist/benchmarks/index.d.ts +2 -891
- package/dist/benchmarks/index.js +2 -60
- package/dist/benchmarks-DviOvUNr.js +754 -0
- package/dist/benchmarks-DviOvUNr.js.map +1 -0
- package/dist/builder-eval/index.d.ts +150 -662
- package/dist/builder-eval/index.d.ts.map +1 -0
- package/dist/builder-eval/index.js +356 -345
- package/dist/builder-eval/index.js.map +1 -1
- package/dist/calibration-CNWWA6K8.js +94 -0
- package/dist/calibration-CNWWA6K8.js.map +1 -0
- package/dist/campaign/index.d.ts +5 -6381
- package/dist/campaign/index.js +3 -213
- package/dist/campaign-CBKZvQ1H.js +3885 -0
- package/dist/campaign-CBKZvQ1H.js.map +1 -0
- package/dist/cli.d.ts +1 -1
- package/dist/cli.js +164 -175
- package/dist/cli.js.map +1 -1
- package/dist/client-C97NMzqi.d.ts +581 -0
- package/dist/client-C97NMzqi.d.ts.map +1 -0
- package/dist/client-CYzbdJOZ.js +637 -0
- package/dist/client-CYzbdJOZ.js.map +1 -0
- package/dist/code-agent-session-BjkMTQ7H.js +1390 -0
- package/dist/code-agent-session-BjkMTQ7H.js.map +1 -0
- package/dist/code-agent-session-DqqgOJaz.d.ts +143 -0
- package/dist/code-agent-session-DqqgOJaz.d.ts.map +1 -0
- package/dist/concurrency-DIxRZF_J.js +85 -0
- package/dist/concurrency-DIxRZF_J.js.map +1 -0
- package/dist/contract/index.d.ts +645 -5565
- package/dist/contract/index.d.ts.map +1 -0
- package/dist/contract/index.js +1730 -1938
- package/dist/contract/index.js.map +1 -1
- package/dist/control.d.ts +3 -1030
- package/dist/control.js +2 -33
- package/dist/cost-ledger-DIgQUFZZ.js +801 -0
- package/dist/cost-ledger-DIgQUFZZ.js.map +1 -0
- package/dist/cost-ledger-Dye6jCgg.d.ts +236 -0
- package/dist/cost-ledger-Dye6jCgg.d.ts.map +1 -0
- package/dist/dataset-BvtnC8Dc.d.ts +115 -0
- package/dist/dataset-BvtnC8Dc.d.ts.map +1 -0
- package/dist/default-registry-C-vFCSEc.js +2579 -0
- package/dist/default-registry-C-vFCSEc.js.map +1 -0
- package/dist/default-registry-CNPo-Vsb.d.ts +540 -0
- package/dist/default-registry-CNPo-Vsb.d.ts.map +1 -0
- package/dist/emitter-CPBAhxum.js +266 -0
- package/dist/emitter-CPBAhxum.js.map +1 -0
- package/dist/emitter-DGQGoLyj.d.ts +113 -0
- package/dist/emitter-DGQGoLyj.d.ts.map +1 -0
- package/dist/errors-8YnH8WlF.js +64 -0
- package/dist/errors-8YnH8WlF.js.map +1 -0
- package/dist/errors-CEk209JS.d.ts +76 -0
- package/dist/errors-CEk209JS.d.ts.map +1 -0
- package/dist/eval-campaign-DEm6c8ru.js +349 -0
- package/dist/eval-campaign-DEm6c8ru.js.map +1 -0
- package/dist/execution-tracks-CpgFPpS5.js +93 -0
- package/dist/execution-tracks-CpgFPpS5.js.map +1 -0
- package/dist/exporters-q9iL-2Jf.js +148 -0
- package/dist/exporters-q9iL-2Jf.js.map +1 -0
- package/dist/extract-usage-BrQ8mCLX.js +155 -0
- package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
- package/dist/failure-cluster-CqcvCcdR.d.ts +59 -0
- package/dist/failure-cluster-CqcvCcdR.d.ts.map +1 -0
- package/dist/feedback-trajectory-CVaeREXV.d.ts +340 -0
- package/dist/feedback-trajectory-CVaeREXV.d.ts.map +1 -0
- package/dist/fuzz.d.ts +320 -646
- package/dist/fuzz.d.ts.map +1 -0
- package/dist/fuzz.js +670 -618
- package/dist/fuzz.js.map +1 -1
- package/dist/hf-dataset-DBJXXoY1.js +763 -0
- package/dist/hf-dataset-DBJXXoY1.js.map +1 -0
- package/dist/hosted/index.d.ts +11 -831
- package/dist/hosted/index.d.ts.map +1 -0
- package/dist/hosted/index.js +2 -37
- package/dist/index-2JJSA6-r2.d.ts +926 -0
- package/dist/index-2JJSA6-r2.d.ts.map +1 -0
- package/dist/index-6N0aYmpW.d.ts +217 -0
- package/dist/index-6N0aYmpW.d.ts.map +1 -0
- package/dist/index-BAvgST_9.d.ts +131 -0
- package/dist/index-BAvgST_9.d.ts.map +1 -0
- package/dist/index-BvnJuTGD.d.ts +68 -0
- package/dist/index-BvnJuTGD.d.ts.map +1 -0
- package/dist/index-C61Wi7yg.d.ts +547 -0
- package/dist/index-C61Wi7yg.d.ts.map +1 -0
- package/dist/index-CAPUUKaM.d.ts +335 -0
- package/dist/index-CAPUUKaM.d.ts.map +1 -0
- package/dist/index-DE5fb3EC.d.ts +2244 -0
- package/dist/index-DE5fb3EC.d.ts.map +1 -0
- package/dist/index-DSC51roc.d.ts +102 -0
- package/dist/index-DSC51roc.d.ts.map +1 -0
- package/dist/index.d.ts +3755 -15555
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +11182 -11216
- package/dist/index.js.map +1 -1
- package/dist/integrity-BzRbCHzi.js +424 -0
- package/dist/integrity-BzRbCHzi.js.map +1 -0
- package/dist/integrity-rmVhXWA7.d.ts +61 -0
- package/dist/integrity-rmVhXWA7.d.ts.map +1 -0
- package/dist/judge-calibration-DFtEMlde.d.ts +146 -0
- package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
- package/dist/ledger-core/index.d.ts +2 -0
- package/dist/ledger-core/index.js +2 -0
- package/dist/ledger-core-DtZz1RG0.js +388 -0
- package/dist/ledger-core-DtZz1RG0.js.map +1 -0
- package/dist/llm-client--GR4JbZE.js +687 -0
- package/dist/llm-client--GR4JbZE.js.map +1 -0
- package/dist/llm-client-B_nIBlYo.d.ts +290 -0
- package/dist/llm-client-B_nIBlYo.d.ts.map +1 -0
- package/dist/matrix/index.d.ts +3 -155
- package/dist/matrix/index.js +2 -8
- package/dist/matrix-BzQnu2S6.js +270 -0
- package/dist/matrix-BzQnu2S6.js.map +1 -0
- package/dist/meta-eval/index.d.ts +4 -1027
- package/dist/meta-eval/index.js +393 -390
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/metrics-C9YY1OcL.js +239 -0
- package/dist/metrics-C9YY1OcL.js.map +1 -0
- package/dist/mint-yN2M2eh0.js +201 -0
- package/dist/mint-yN2M2eh0.js.map +1 -0
- package/dist/multi-layer-verifier-BHY1gWAc.d.ts +138 -0
- package/dist/multi-layer-verifier-BHY1gWAc.d.ts.map +1 -0
- package/dist/multishot/index.d.ts +273 -480
- package/dist/multishot/index.d.ts.map +1 -0
- package/dist/multishot/index.js +595 -548
- package/dist/multishot/index.js.map +1 -1
- package/dist/off-policy-DvgzvtIx.js +220 -0
- package/dist/off-policy-DvgzvtIx.js.map +1 -0
- package/dist/off-policy-mskQw8Mb.d.ts +153 -0
- package/dist/off-policy-mskQw8Mb.d.ts.map +1 -0
- package/dist/openapi.json +1 -1
- package/dist/opencode-sqlite-BGrHeDu3.js +318 -0
- package/dist/opencode-sqlite-BGrHeDu3.js.map +1 -0
- package/dist/outcome-store-BYHIuO0e.d.ts +65 -0
- package/dist/outcome-store-BYHIuO0e.d.ts.map +1 -0
- package/dist/outcome-store-ChBKlTd_.js +75 -0
- package/dist/outcome-store-ChBKlTd_.js.map +1 -0
- package/dist/paired-arms-D9D0wXj2.js +260 -0
- package/dist/paired-arms-D9D0wXj2.js.map +1 -0
- package/dist/pipelines/index.d.ts +95 -532
- package/dist/pipelines/index.d.ts.map +1 -0
- package/dist/pipelines/index.js +497 -478
- package/dist/pipelines/index.js.map +1 -1
- package/dist/pre-registration-DakwTRXk.js +96 -0
- package/dist/pre-registration-DakwTRXk.js.map +1 -0
- package/dist/propose-review-control-LhIWGzHl.js +1458 -0
- package/dist/propose-review-control-LhIWGzHl.js.map +1 -0
- package/dist/query-CJ_DX8vl.d.ts +28 -0
- package/dist/query-CJ_DX8vl.d.ts.map +1 -0
- package/dist/query-Di7eEQ79.js +83 -0
- package/dist/query-Di7eEQ79.js.map +1 -0
- package/dist/raw-provider-sink-BQd7mzyT.js +201 -0
- package/dist/raw-provider-sink-BQd7mzyT.js.map +1 -0
- package/dist/raw-provider-sink-BU29Sh8h.d.ts +134 -0
- package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +1 -0
- package/dist/redact-7Aq1ukl-.js +107 -0
- package/dist/redact-7Aq1ukl-.js.map +1 -0
- package/dist/release-report-Cz9NKH39.js +603 -0
- package/dist/release-report-Cz9NKH39.js.map +1 -0
- package/dist/release-report-mvB2G4_J.d.ts +244 -0
- package/dist/release-report-mvB2G4_J.d.ts.map +1 -0
- package/dist/replay-GnyotH0J.js +1741 -0
- package/dist/replay-GnyotH0J.js.map +1 -0
- package/dist/replay-OoidtG1E.d.ts +749 -0
- package/dist/replay-OoidtG1E.d.ts.map +1 -0
- package/dist/reporting.d.ts +6 -1312
- package/dist/reporting.js +6 -51
- package/dist/researcher-CwTdwXG1.d.ts +314 -0
- package/dist/researcher-CwTdwXG1.d.ts.map +1 -0
- package/dist/reward-hacking-eAnOsynk.d.ts +224 -0
- package/dist/reward-hacking-eAnOsynk.d.ts.map +1 -0
- package/dist/reward-hacking-qipEpKvY.js +596 -0
- package/dist/reward-hacking-qipEpKvY.js.map +1 -0
- package/dist/reward-nw2xZGZG.js +137 -0
- package/dist/reward-nw2xZGZG.js.map +1 -0
- package/dist/rl.d.ts +760 -4010
- package/dist/rl.d.ts.map +1 -0
- package/dist/rl.js +2325 -1958
- package/dist/rl.js.map +1 -1
- package/dist/rolldown-runtime-8H4AJuhK.js +14 -0
- package/dist/rollout/index.d.ts +3 -2087
- package/dist/rollout/index.js +8 -168
- package/dist/rollout-BOYjemfR.js +624 -0
- package/dist/rollout-BOYjemfR.js.map +1 -0
- package/dist/rubric-predictive-validity-B3xmbmS1.js +141 -0
- package/dist/rubric-predictive-validity-B3xmbmS1.js.map +1 -0
- package/dist/rubric-predictive-validity-Ku_clp_1.d.ts +74 -0
- package/dist/rubric-predictive-validity-Ku_clp_1.d.ts.map +1 -0
- package/dist/run-evidence-ClRX_8A9.d.ts +225 -0
- package/dist/run-evidence-ClRX_8A9.d.ts.map +1 -0
- package/dist/run-record-BuoE80Dq.js +467 -0
- package/dist/{chunk-56TAVBOK.js.map → run-record-BuoE80Dq.js.map} +1 -1
- package/dist/run-record-CnZu_gjl.d.ts +357 -0
- package/dist/run-record-CnZu_gjl.d.ts.map +1 -0
- package/dist/run-score-iEEAWiBY.js +41 -0
- package/dist/run-score-iEEAWiBY.js.map +1 -0
- package/dist/runtime-trajectory-1gyaTOoC.js +93 -0
- package/dist/runtime-trajectory-1gyaTOoC.js.map +1 -0
- package/dist/runtime-trajectory-BvSZcCHD.d.ts +50 -0
- package/dist/runtime-trajectory-BvSZcCHD.d.ts.map +1 -0
- package/dist/schema-BtVldJ3T.d.ts +206 -0
- package/dist/schema-BtVldJ3T.d.ts.map +1 -0
- package/dist/schema-C6DW4ZHR.js +821 -0
- package/dist/schema-C6DW4ZHR.js.map +1 -0
- package/dist/schema-CRhEY1SO.js +69 -0
- package/dist/schema-CRhEY1SO.js.map +1 -0
- package/dist/schema-Cef2cFmb.d.ts +408 -0
- package/dist/schema-Cef2cFmb.d.ts.map +1 -0
- package/dist/semantic-concept-judge-B6cWNJ2K.js +725 -0
- package/dist/semantic-concept-judge-B6cWNJ2K.js.map +1 -0
- package/dist/sequential-Br0mAPHA.js +148 -0
- package/dist/sequential-Br0mAPHA.js.map +1 -0
- package/dist/sequential-CYwq6Ff_.d.ts +141 -0
- package/dist/sequential-CYwq6Ff_.d.ts.map +1 -0
- package/dist/series-convergence-CjO2QdRW.js +43 -0
- package/dist/series-convergence-CjO2QdRW.js.map +1 -0
- package/dist/series-convergence-ofsqPWhs.d.ts +35 -0
- package/dist/series-convergence-ofsqPWhs.d.ts.map +1 -0
- package/dist/server-m5D9cvnG.js +1040 -0
- package/dist/server-m5D9cvnG.js.map +1 -0
- package/dist/skill-usage-C_pXm7lP.d.ts +534 -0
- package/dist/skill-usage-C_pXm7lP.d.ts.map +1 -0
- package/dist/skillopt-optimization-method-B7wX7XkF.d.ts +1739 -0
- package/dist/skillopt-optimization-method-B7wX7XkF.d.ts.map +1 -0
- package/dist/skillopt-optimization-method-D4ODwFVV.js +7712 -0
- package/dist/skillopt-optimization-method-D4ODwFVV.js.map +1 -0
- package/dist/statistics-Cmj6nynr.d.ts +514 -0
- package/dist/statistics-Cmj6nynr.d.ts.map +1 -0
- package/dist/statistics-CnnxdpOg.js +1437 -0
- package/dist/statistics-CnnxdpOg.js.map +1 -0
- package/dist/store-CT9YIIve.d.ts +117 -0
- package/dist/store-CT9YIIve.d.ts.map +1 -0
- package/dist/store-CxJry_cs.d.ts +229 -0
- package/dist/store-CxJry_cs.d.ts.map +1 -0
- package/dist/storyboard/index.d.ts +84 -203
- package/dist/storyboard/index.d.ts.map +1 -0
- package/dist/storyboard/index.js +609 -542
- package/dist/storyboard/index.js.map +1 -1
- package/dist/summary-report-BNs5nmXI.js +862 -0
- package/dist/summary-report-BNs5nmXI.js.map +1 -0
- package/dist/summary-report-Cj9gdw4i.d.ts +407 -0
- package/dist/summary-report-Cj9gdw4i.d.ts.map +1 -0
- package/dist/supervisor-run/index.d.ts +2 -959
- package/dist/supervisor-run/index.js +2 -65
- package/dist/supervisor-run-_lnTLM3z.js +1679 -0
- package/dist/supervisor-run-_lnTLM3z.js.map +1 -0
- package/dist/task-failure-attributes-CQZlB3et.js +311 -0
- package/dist/task-failure-attributes-CQZlB3et.js.map +1 -0
- package/dist/test-graded-scenario-BsqWLmPt.js +318 -0
- package/dist/test-graded-scenario-BsqWLmPt.js.map +1 -0
- package/dist/test-graded-scenario-D1TaI2va.d.ts +141 -0
- package/dist/test-graded-scenario-D1TaI2va.d.ts.map +1 -0
- package/dist/tools-BmuN627J.js +1085 -0
- package/dist/tools-BmuN627J.js.map +1 -0
- package/dist/trace-attributes.d.ts +2 -52
- package/dist/trace-attributes.js +131 -61
- package/dist/trace-attributes.js.map +1 -1
- package/dist/traces.d.ts +12 -2365
- package/dist/traces.js +12 -252
- package/dist/trajectory-D_7rLrvE.js +56 -0
- package/dist/trajectory-D_7rLrvE.js.map +1 -0
- package/dist/types-DGsxbAEd.d.ts +387 -0
- package/dist/types-DGsxbAEd.d.ts.map +1 -0
- package/dist/types-k9tZGKUg.d.ts +640 -0
- package/dist/types-k9tZGKUg.d.ts.map +1 -0
- package/dist/verdict-Dps8_okt.d.ts +37 -0
- package/dist/verdict-Dps8_okt.d.ts.map +1 -0
- package/dist/wire/index.d.ts +702 -1173
- package/dist/wire/index.d.ts.map +1 -0
- package/dist/wire/index.js +2 -81
- package/package.json +17 -9
- package/dist/benchmarks/index.js.map +0 -1
- package/dist/campaign/index.js.map +0 -1
- package/dist/chunk-2QU3YOPR.js +0 -7374
- package/dist/chunk-2QU3YOPR.js.map +0 -1
- package/dist/chunk-3OCR4R5I.js +0 -728
- package/dist/chunk-3OCR4R5I.js.map +0 -1
- package/dist/chunk-3RF76KTD.js +0 -84
- package/dist/chunk-3RF76KTD.js.map +0 -1
- package/dist/chunk-56TAVBOK.js +0 -698
- package/dist/chunk-5DTSBUL2.js +0 -159
- package/dist/chunk-5DTSBUL2.js.map +0 -1
- package/dist/chunk-7FO3TNPI.js +0 -232
- package/dist/chunk-7FO3TNPI.js.map +0 -1
- package/dist/chunk-7ZZMD7UK.js +0 -386
- package/dist/chunk-7ZZMD7UK.js.map +0 -1
- package/dist/chunk-BOD4O7OF.js +0 -40
- package/dist/chunk-BOD4O7OF.js.map +0 -1
- package/dist/chunk-BSO5JDQH.js +0 -2335
- package/dist/chunk-BSO5JDQH.js.map +0 -1
- package/dist/chunk-C6LXANRU.js +0 -1550
- package/dist/chunk-C6LXANRU.js.map +0 -1
- package/dist/chunk-DODXQREJ.js +0 -752
- package/dist/chunk-DODXQREJ.js.map +0 -1
- package/dist/chunk-DRYIUNWY.js +0 -622
- package/dist/chunk-DRYIUNWY.js.map +0 -1
- package/dist/chunk-E7QXT7SX.js +0 -183
- package/dist/chunk-E7QXT7SX.js.map +0 -1
- package/dist/chunk-EG66UGL4.js +0 -341
- package/dist/chunk-EG66UGL4.js.map +0 -1
- package/dist/chunk-FXTVJPYD.js +0 -576
- package/dist/chunk-FXTVJPYD.js.map +0 -1
- package/dist/chunk-G7MGMCZD.js +0 -153
- package/dist/chunk-G7MGMCZD.js.map +0 -1
- package/dist/chunk-GGE4NNQT.js +0 -65
- package/dist/chunk-GGE4NNQT.js.map +0 -1
- package/dist/chunk-H23X7XKK.js +0 -181
- package/dist/chunk-H23X7XKK.js.map +0 -1
- package/dist/chunk-HHWE3POT.js +0 -94
- package/dist/chunk-HHWE3POT.js.map +0 -1
- package/dist/chunk-HPWUNB47.js +0 -289
- package/dist/chunk-HPWUNB47.js.map +0 -1
- package/dist/chunk-IYCLP2N2.js +0 -766
- package/dist/chunk-IYCLP2N2.js.map +0 -1
- package/dist/chunk-JHCHEVET.js +0 -274
- package/dist/chunk-JHCHEVET.js.map +0 -1
- package/dist/chunk-JQSF5DQT.js +0 -701
- package/dist/chunk-JQSF5DQT.js.map +0 -1
- package/dist/chunk-K4DBDHLK.js +0 -158
- package/dist/chunk-K4DBDHLK.js.map +0 -1
- package/dist/chunk-K6N6XJJX.js +0 -306
- package/dist/chunk-K6N6XJJX.js.map +0 -1
- package/dist/chunk-M4YBQKIJ.js +0 -1040
- package/dist/chunk-M4YBQKIJ.js.map +0 -1
- package/dist/chunk-MA6HLL3S.js +0 -65
- package/dist/chunk-MA6HLL3S.js.map +0 -1
- package/dist/chunk-MAZ26DC7.js +0 -99
- package/dist/chunk-MAZ26DC7.js.map +0 -1
- package/dist/chunk-NPCTHQIO.js +0 -91
- package/dist/chunk-NPCTHQIO.js.map +0 -1
- package/dist/chunk-NY44NC4A.js +0 -1056
- package/dist/chunk-NY44NC4A.js.map +0 -1
- package/dist/chunk-OIUOT4QD.js +0 -44
- package/dist/chunk-OIUOT4QD.js.map +0 -1
- package/dist/chunk-ONWEPEDO.js +0 -57
- package/dist/chunk-ONWEPEDO.js.map +0 -1
- package/dist/chunk-OWN5NPMC.js +0 -152
- package/dist/chunk-OWN5NPMC.js.map +0 -1
- package/dist/chunk-P6FYH6K4.js +0 -1161
- package/dist/chunk-P6FYH6K4.js.map +0 -1
- package/dist/chunk-PC4UYEBM.js +0 -166
- package/dist/chunk-PC4UYEBM.js.map +0 -1
- package/dist/chunk-PC5DOSM7.js +0 -579
- package/dist/chunk-PC5DOSM7.js.map +0 -1
- package/dist/chunk-PXE2VKMX.js +0 -140
- package/dist/chunk-PXE2VKMX.js.map +0 -1
- package/dist/chunk-PZ5AY32C.js +0 -10
- package/dist/chunk-PZ5AY32C.js.map +0 -1
- package/dist/chunk-QB6BDBP2.js +0 -4464
- package/dist/chunk-QB6BDBP2.js.map +0 -1
- package/dist/chunk-RXHCETDZ.js +0 -536
- package/dist/chunk-RXHCETDZ.js.map +0 -1
- package/dist/chunk-RZTMDUO7.js +0 -49
- package/dist/chunk-RZTMDUO7.js.map +0 -1
- package/dist/chunk-SFLLL76A.js +0 -669
- package/dist/chunk-SFLLL76A.js.map +0 -1
- package/dist/chunk-SZLVEKMJ.js +0 -1446
- package/dist/chunk-SZLVEKMJ.js.map +0 -1
- package/dist/chunk-T4SQEITX.js +0 -95
- package/dist/chunk-T4SQEITX.js.map +0 -1
- package/dist/chunk-T6RLYGAD.js +0 -158
- package/dist/chunk-T6RLYGAD.js.map +0 -1
- package/dist/chunk-TJVT4QFF.js +0 -911
- package/dist/chunk-TJVT4QFF.js.map +0 -1
- package/dist/chunk-TQ7LNKZ3.js +0 -136
- package/dist/chunk-TQ7LNKZ3.js.map +0 -1
- package/dist/chunk-U4L7JRPZ.js +0 -1706
- package/dist/chunk-U4L7JRPZ.js.map +0 -1
- package/dist/chunk-U4PHLT2N.js +0 -419
- package/dist/chunk-U4PHLT2N.js.map +0 -1
- package/dist/chunk-VCZ5FQYW.js +0 -928
- package/dist/chunk-VCZ5FQYW.js.map +0 -1
- package/dist/chunk-VI2UW6B6.js +0 -162
- package/dist/chunk-VI2UW6B6.js.map +0 -1
- package/dist/chunk-VQMK5FMP.js +0 -247
- package/dist/chunk-VQMK5FMP.js.map +0 -1
- package/dist/chunk-WGXIEX7P.js +0 -116
- package/dist/chunk-WGXIEX7P.js.map +0 -1
- package/dist/chunk-WVATSFCP.js +0 -1553
- package/dist/chunk-WVATSFCP.js.map +0 -1
- package/dist/chunk-X4YIBDER.js +0 -1662
- package/dist/chunk-X4YIBDER.js.map +0 -1
- package/dist/chunk-YQN4ICPP.js +0 -355
- package/dist/chunk-YQN4ICPP.js.map +0 -1
- package/dist/chunk-ZET2UAYW.js +0 -89
- package/dist/chunk-ZET2UAYW.js.map +0 -1
- package/dist/chunk-ZHTZ4EYI.js +0 -1212
- package/dist/chunk-ZHTZ4EYI.js.map +0 -1
- package/dist/control.js.map +0 -1
- package/dist/hosted/index.js.map +0 -1
- package/dist/matrix/index.js.map +0 -1
- package/dist/reporting.js.map +0 -1
- package/dist/rollout/index.js.map +0 -1
- package/dist/run-campaign-OJJ7CZF4.js +0 -18
- package/dist/run-campaign-OJJ7CZF4.js.map +0 -1
- package/dist/supervisor-run/index.js.map +0 -1
- package/dist/traces.js.map +0 -1
- package/dist/wire/index.js.map +0 -1
|
@@ -0,0 +1,749 @@
|
|
|
1
|
+
import { o as NotFoundError, s as ReplayError } from "./errors-CEk209JS.js";
|
|
2
|
+
import { a as RunRecord, l as RunTerminalOutcome, s as RunSplitTag, u as RunTokenUsage } from "./run-record-CnZu_gjl.js";
|
|
3
|
+
import { S as ToolSpan, f as Run } from "./schema-BtVldJ3T.js";
|
|
4
|
+
import { c as RawProviderEvent, l as RawProviderSink } from "./raw-provider-sink-BU29Sh8h.js";
|
|
5
|
+
import { s as TraceStore } from "./store-CT9YIIve.js";
|
|
6
|
+
import { n as RunCompleteHookContext, t as RunCompleteHook } from "./emitter-DGQGoLyj.js";
|
|
7
|
+
import { a as QueryTracesPage, d as TraceAnalystFilters, g as ViewSpansResult, m as TraceAnalystSpanStatus, o as SearchSpanResult, p as TraceAnalystSpanKind, r as DatasetOverview, s as SearchTraceResult, t as TraceAnalysisStore, v as ViewTraceResult } from "./store-CxJry_cs.js";
|
|
8
|
+
import { n as AnalyzeTracesOptions, r as AnalyzeTracesResult } from "./analyst-BkTS3C58.js";
|
|
9
|
+
import { AxFunction } from "@ax-llm/ax";
|
|
10
|
+
//#region src/trace/otel.d.ts
|
|
11
|
+
declare const OTEL_AGENT_EVAL_SCOPE: {
|
|
12
|
+
name: string;
|
|
13
|
+
version: string;
|
|
14
|
+
};
|
|
15
|
+
interface OtlpSpan {
|
|
16
|
+
traceId: string;
|
|
17
|
+
spanId: string;
|
|
18
|
+
parentSpanId?: string;
|
|
19
|
+
name: string;
|
|
20
|
+
kind: number;
|
|
21
|
+
startTimeUnixNano: string;
|
|
22
|
+
endTimeUnixNano: string;
|
|
23
|
+
attributes: Array<{
|
|
24
|
+
key: string;
|
|
25
|
+
value: {
|
|
26
|
+
stringValue?: string;
|
|
27
|
+
intValue?: string;
|
|
28
|
+
doubleValue?: number;
|
|
29
|
+
boolValue?: boolean;
|
|
30
|
+
};
|
|
31
|
+
}>;
|
|
32
|
+
events?: Array<{
|
|
33
|
+
timeUnixNano: string;
|
|
34
|
+
name: string;
|
|
35
|
+
attributes?: OtlpSpan['attributes'];
|
|
36
|
+
}>;
|
|
37
|
+
status?: {
|
|
38
|
+
code: number;
|
|
39
|
+
message?: string;
|
|
40
|
+
};
|
|
41
|
+
}
|
|
42
|
+
interface OtlpResourceSpans {
|
|
43
|
+
resource: {
|
|
44
|
+
attributes: OtlpSpan['attributes'];
|
|
45
|
+
};
|
|
46
|
+
scopeSpans: Array<{
|
|
47
|
+
scope: typeof OTEL_AGENT_EVAL_SCOPE;
|
|
48
|
+
spans: OtlpSpan[];
|
|
49
|
+
}>;
|
|
50
|
+
}
|
|
51
|
+
interface OtlpExport {
|
|
52
|
+
resourceSpans: OtlpResourceSpans[];
|
|
53
|
+
}
|
|
54
|
+
/** Export a single run's spans + events in OTLP/JSON. */
|
|
55
|
+
declare function exportRunAsOtlp(store: TraceStore, runId: string, resourceAttrs?: Record<string, string | number | boolean>): Promise<OtlpExport>;
|
|
56
|
+
//#endregion
|
|
57
|
+
//#region src/trace/otlp-attributes.d.ts
|
|
58
|
+
type ToolSpanOtlpInput = Pick<ToolSpan, 'toolName' | 'args' | 'argsCaptured' | 'result' | 'latencyMs'>;
|
|
59
|
+
type OtlpSpanRole = 'AGENT' | 'CHAIN' | 'EVALUATOR' | 'GUARDRAIL' | 'LLM' | 'SPAN' | 'TOOL' | 'UNKNOWN';
|
|
60
|
+
interface OtlpSpanRoleInput {
|
|
61
|
+
name: string;
|
|
62
|
+
attributes: Record<string, unknown>;
|
|
63
|
+
kind?: string | null;
|
|
64
|
+
}
|
|
65
|
+
/**
|
|
66
|
+
* Classify a span once for both measurement and error accounting.
|
|
67
|
+
* An explicit OpenInference kind wins; untyped spans use the same tool and
|
|
68
|
+
* model signals in online and offline intake.
|
|
69
|
+
*/
|
|
70
|
+
declare function classifyOtlpSpanRole(input: OtlpSpanRoleInput): OtlpSpanRole;
|
|
71
|
+
declare function isOtlpModelCall(input: OtlpSpanRoleInput): boolean;
|
|
72
|
+
declare function applyToolSpanOtlpAttributes(attributes: Record<string, unknown>, span: ToolSpanOtlpInput): void;
|
|
73
|
+
declare function traceSpanKindToOpenInferenceKind(kind: string): string;
|
|
74
|
+
//#endregion
|
|
75
|
+
//#region src/trace/otel-export.d.ts
|
|
76
|
+
interface OtelExportConfig {
|
|
77
|
+
/** OTLP endpoint. Reads OTEL_EXPORTER_OTLP_ENDPOINT env by default. */
|
|
78
|
+
endpoint?: string;
|
|
79
|
+
/** OTLP headers. Reads OTEL_EXPORTER_OTLP_HEADERS env by default. */
|
|
80
|
+
headers?: Record<string, string>;
|
|
81
|
+
/** Batch size before flush. Default 64. */
|
|
82
|
+
batchSize?: number;
|
|
83
|
+
/** Flush interval ms. Default 5000. */
|
|
84
|
+
flushIntervalMs?: number;
|
|
85
|
+
/** Resource attributes stamped on every export. */
|
|
86
|
+
resourceAttributes?: Record<string, string | number | boolean>;
|
|
87
|
+
/** Service name. Default 'agent-eval'. */
|
|
88
|
+
serviceName?: string;
|
|
89
|
+
}
|
|
90
|
+
interface OtelExporter {
|
|
91
|
+
/** Called by the TraceEmitter on every span close. */
|
|
92
|
+
exportSpan(span: ExportableSpan): void;
|
|
93
|
+
/** Force flush pending spans. */
|
|
94
|
+
flush(): Promise<void>;
|
|
95
|
+
/** Shutdown cleanly — flushes remaining spans and stops the timer. */
|
|
96
|
+
shutdown(): Promise<void>;
|
|
97
|
+
}
|
|
98
|
+
interface ExportableSpan {
|
|
99
|
+
traceId: string;
|
|
100
|
+
spanId: string;
|
|
101
|
+
parentSpanId?: string;
|
|
102
|
+
name: string;
|
|
103
|
+
kind: string;
|
|
104
|
+
startedAt: number;
|
|
105
|
+
endedAt?: number;
|
|
106
|
+
status?: string;
|
|
107
|
+
error?: string;
|
|
108
|
+
model?: string;
|
|
109
|
+
inputTokens?: number;
|
|
110
|
+
outputTokens?: number;
|
|
111
|
+
reasoningTokens?: number;
|
|
112
|
+
cachedTokens?: number;
|
|
113
|
+
cacheWriteTokens?: number;
|
|
114
|
+
costUsd?: number;
|
|
115
|
+
tool?: ToolSpanOtlpInput;
|
|
116
|
+
attributes?: Record<string, unknown>;
|
|
117
|
+
}
|
|
118
|
+
/**
|
|
119
|
+
* Create an OTEL exporter. Returns undefined when no endpoint is configured
|
|
120
|
+
* (neither via config nor env) — callers should check before attaching.
|
|
121
|
+
*/
|
|
122
|
+
declare function createOtelExporter(config?: OtelExportConfig): OtelExporter | undefined;
|
|
123
|
+
//#endregion
|
|
124
|
+
//#region src/trace/otel-bridge.d.ts
|
|
125
|
+
/**
|
|
126
|
+
* Create a RunCompleteHook that exports all spans from the completed run
|
|
127
|
+
* to the OTEL exporter, then flushes.
|
|
128
|
+
*/
|
|
129
|
+
declare function otelRunCompleteHook(exporter: OtelExporter): RunCompleteHook;
|
|
130
|
+
/**
|
|
131
|
+
* Create an auto-exporting TraceStore wrapper that intercepts updateSpan
|
|
132
|
+
* calls. When a span gets an endedAt, it's exported immediately. This
|
|
133
|
+
* gives real-time streaming instead of batch-at-end.
|
|
134
|
+
*
|
|
135
|
+
* This is the preferred integration path: wrap the store before
|
|
136
|
+
* constructing the TraceEmitter.
|
|
137
|
+
*/
|
|
138
|
+
declare function createOtelTracingStore(inner: TraceStore, exporter: OtelExporter, traceId: string): TraceStore;
|
|
139
|
+
//#endregion
|
|
140
|
+
//#region src/trace/redact.d.ts
|
|
141
|
+
/**
|
|
142
|
+
* Redaction — remove PII / secrets from trace payloads before persist.
|
|
143
|
+
*
|
|
144
|
+
* Pre-persistence rules mean raw traces in storage are already scrubbed.
|
|
145
|
+
* Unredacted variants (for debugging / post-mortems) live in a separate
|
|
146
|
+
* storage layer with stricter access controls; this module only covers
|
|
147
|
+
* the default scrub-then-persist path.
|
|
148
|
+
*
|
|
149
|
+
* Rules compose: pass an array of `RedactionRule`, each is applied in
|
|
150
|
+
* order. Strings that match get replaced with a tagged sentinel so the
|
|
151
|
+
* eval framework can count how many redactions happened per run
|
|
152
|
+
* (surfaced via `redaction_applied` events).
|
|
153
|
+
*/
|
|
154
|
+
interface RedactionRule {
|
|
155
|
+
id: string;
|
|
156
|
+
pattern: RegExp;
|
|
157
|
+
/** Replacement — e.g. '[PII:email]'. Defaults to `[redacted:{id}]`. */
|
|
158
|
+
replacement?: string;
|
|
159
|
+
}
|
|
160
|
+
interface RedactionReport {
|
|
161
|
+
redactionCount: number;
|
|
162
|
+
byRule: Record<string, number>;
|
|
163
|
+
}
|
|
164
|
+
/** OWASP / common-sense defaults — extend per-domain. */
|
|
165
|
+
declare const DEFAULT_REDACTION_RULES: RedactionRule[];
|
|
166
|
+
declare const REDACTION_VERSION = "1.0.0";
|
|
167
|
+
/**
|
|
168
|
+
* Redact a single string. Returns the new string and a per-rule count of
|
|
169
|
+
* how many substitutions fired.
|
|
170
|
+
*/
|
|
171
|
+
declare function redactString(input: string, rules?: RedactionRule[]): {
|
|
172
|
+
output: string;
|
|
173
|
+
report: RedactionReport;
|
|
174
|
+
};
|
|
175
|
+
/**
|
|
176
|
+
* Walk a JSON-ish value applying `redactString` to every string leaf.
|
|
177
|
+
* Arrays and plain objects are recursed; other types pass through
|
|
178
|
+
* untouched. Circular references throw — traces should be tree-shaped.
|
|
179
|
+
*/
|
|
180
|
+
declare function redactValue(value: unknown, rules?: RedactionRule[], report?: RedactionReport): {
|
|
181
|
+
value: unknown;
|
|
182
|
+
report: RedactionReport;
|
|
183
|
+
};
|
|
184
|
+
//#endregion
|
|
185
|
+
//#region src/trace/store-to-otlp.d.ts
|
|
186
|
+
interface TracesToOtlpResult {
|
|
187
|
+
/** Total spans emitted across every cell. */
|
|
188
|
+
spanCount: number;
|
|
189
|
+
/** Total run-anchor spans (one per Run) appended for analyst visibility. */
|
|
190
|
+
runCount: number;
|
|
191
|
+
/** Cells whose shards parsed cleanly. */
|
|
192
|
+
cellCount: number;
|
|
193
|
+
/** Cells that errored mid-conversion — surfaced so a partial conversion
|
|
194
|
+
* isn't silently masked. */
|
|
195
|
+
cellErrorCount: number;
|
|
196
|
+
}
|
|
197
|
+
/**
|
|
198
|
+
* A trace source the analyst should ingest. Two layouts are supported:
|
|
199
|
+
*
|
|
200
|
+
* - `celled` (default): the root holds one cell subdirectory per
|
|
201
|
+
* persona/variant, each a `FileSystemTraceStore` (`runs.ndjson`,
|
|
202
|
+
* `spans.ndjson`, `events.ndjson`).
|
|
203
|
+
* - `flat`: the root is itself a single `FileSystemTraceStore` (e.g. a
|
|
204
|
+
* production-ingestion sidecar that appends one ndjson set directly under
|
|
205
|
+
* the chosen directory).
|
|
206
|
+
*/
|
|
207
|
+
interface TraceStoreSource {
|
|
208
|
+
/** Absolute path to the trace store root. */
|
|
209
|
+
root: string;
|
|
210
|
+
/** Layout — `celled` (default) or `flat`. */
|
|
211
|
+
layout?: 'celled' | 'flat';
|
|
212
|
+
/** OTLP `service.name` for this source. Overrides the options default. */
|
|
213
|
+
serviceName?: string;
|
|
214
|
+
}
|
|
215
|
+
/** Domain hooks. The walker is generic; consumers inject the namespaced
|
|
216
|
+
* attributes that ride along on every run + span for analyst discovery. */
|
|
217
|
+
interface TraceStoreToOtlpOptions {
|
|
218
|
+
/** Default OTLP `service.name` when a source doesn't set its own.
|
|
219
|
+
* Default `agent-eval`. */
|
|
220
|
+
serviceName?: string;
|
|
221
|
+
/** Extra resource attributes per run, e.g.
|
|
222
|
+
* `(run) => ({ 'legal.persona_id': run.tags?.personaId ?? '' })`. */
|
|
223
|
+
resourceAttributes?: (run: Run) => Record<string, unknown>;
|
|
224
|
+
/** Extra attributes on the per-run anchor span, e.g. domain outcome fields. */
|
|
225
|
+
runAttributes?: (run: Run) => Record<string, unknown>;
|
|
226
|
+
}
|
|
227
|
+
/**
|
|
228
|
+
* Read every per-cell shard under each source root and write a flat OTLP-JSONL
|
|
229
|
+
* view of the corpus to `outPath`. Each cell directory is a
|
|
230
|
+
* `FileSystemTraceStore` — NDJSON append-only with size-based rotation;
|
|
231
|
+
* `updateRun`/`updateSpan` append `{ id, ...patch, _update: true }` rows
|
|
232
|
+
* rather than rewriting, so readers must merge those patches in (done here).
|
|
233
|
+
*
|
|
234
|
+
* A `string` source is treated as a celled root.
|
|
235
|
+
*/
|
|
236
|
+
declare function convertTraceStoresToOtlp(source: string | TraceStoreSource | readonly TraceStoreSource[], outPath: string, opts?: TraceStoreToOtlpOptions): TracesToOtlpResult;
|
|
237
|
+
//#endregion
|
|
238
|
+
//#region src/trace-analyst/hook.d.ts
|
|
239
|
+
interface TraceAnalystHookOptions {
|
|
240
|
+
/**
|
|
241
|
+
* Options forwarded to `analyzeTraces`. The hook supplies the question
|
|
242
|
+
* if you don't pass one — defaulting to a launch-grade prompt that asks
|
|
243
|
+
* for failure modes, surprising findings, and a recommendation.
|
|
244
|
+
*/
|
|
245
|
+
analyze: Omit<AnalyzeTracesOptions, 'source'> & {
|
|
246
|
+
source?: AnalyzeTracesOptions['source'];
|
|
247
|
+
};
|
|
248
|
+
/**
|
|
249
|
+
* Override the question. The default is intentionally generic:
|
|
250
|
+
* "Summarise what happened in this run, surface any failure modes,
|
|
251
|
+
* surprising findings, or evidence the verdict is wrong."
|
|
252
|
+
*/
|
|
253
|
+
question?: string;
|
|
254
|
+
/**
|
|
255
|
+
* Persist the result. The hook calls this with the analysis output and
|
|
256
|
+
* the run context. Common implementations write to a TraceAnalysisStore
|
|
257
|
+
* or append to a per-run JSONL.
|
|
258
|
+
*/
|
|
259
|
+
save?: (result: AnalyzeTracesResult, ctx: RunCompleteHookContext) => Promise<void>;
|
|
260
|
+
/**
|
|
261
|
+
* Predicate gating execution per run. Default: every completed run.
|
|
262
|
+
* Use to skip aborted runs, debug runs, or runs without LLM activity.
|
|
263
|
+
*/
|
|
264
|
+
shouldRun?: (ctx: RunCompleteHookContext) => boolean;
|
|
265
|
+
/**
|
|
266
|
+
* Optional gate: if set and returns false, the hook records the failure
|
|
267
|
+
* as a log event on the run instead of staying quiet. The caller can
|
|
268
|
+
* then trigger downstream alerts off `analyst_gate_failed` log events.
|
|
269
|
+
*/
|
|
270
|
+
gateOn?: (result: AnalyzeTracesResult, ctx: RunCompleteHookContext) => boolean;
|
|
271
|
+
}
|
|
272
|
+
declare function traceAnalystOnRunComplete(opts: TraceAnalystHookOptions): RunCompleteHook;
|
|
273
|
+
//#endregion
|
|
274
|
+
//#region src/trace-analyst/insights.d.ts
|
|
275
|
+
interface TraceInsightTask {
|
|
276
|
+
id: string;
|
|
277
|
+
name: string;
|
|
278
|
+
prompt?: string;
|
|
279
|
+
difficulty?: string;
|
|
280
|
+
tags?: string[];
|
|
281
|
+
outcome?: string;
|
|
282
|
+
score?: number;
|
|
283
|
+
gaps?: string[];
|
|
284
|
+
}
|
|
285
|
+
interface TraceInsightSuite {
|
|
286
|
+
name: string;
|
|
287
|
+
collectionId?: string;
|
|
288
|
+
tasks: TraceInsightTask[];
|
|
289
|
+
}
|
|
290
|
+
interface TraceInsightFinding {
|
|
291
|
+
kind: string;
|
|
292
|
+
severity?: string;
|
|
293
|
+
taskIds: string[];
|
|
294
|
+
evidence?: string;
|
|
295
|
+
proposedFixClass?: string;
|
|
296
|
+
}
|
|
297
|
+
interface TraceInsightQuestion {
|
|
298
|
+
id: string;
|
|
299
|
+
question: string;
|
|
300
|
+
why: string;
|
|
301
|
+
}
|
|
302
|
+
interface TraceInsightPanelRole {
|
|
303
|
+
id: string;
|
|
304
|
+
name: string;
|
|
305
|
+
responsibility: string;
|
|
306
|
+
}
|
|
307
|
+
interface TraceInsightPromptInput {
|
|
308
|
+
suite: TraceInsightSuite;
|
|
309
|
+
findings?: TraceInsightFinding[];
|
|
310
|
+
agent?: Record<string, unknown>;
|
|
311
|
+
totals?: Record<string, unknown>;
|
|
312
|
+
maxRepresentativeTraces?: number;
|
|
313
|
+
}
|
|
314
|
+
interface TraceInsightContext {
|
|
315
|
+
suite: TraceInsightSuite;
|
|
316
|
+
scope: string;
|
|
317
|
+
keywords: string[];
|
|
318
|
+
questions: TraceInsightQuestion[];
|
|
319
|
+
panel: TraceInsightPanelRole[];
|
|
320
|
+
findings: TraceInsightFinding[];
|
|
321
|
+
agent: Record<string, unknown> | null;
|
|
322
|
+
totals: Record<string, unknown> | null;
|
|
323
|
+
}
|
|
324
|
+
interface TraceInsightQualityGate {
|
|
325
|
+
id: string;
|
|
326
|
+
label: string;
|
|
327
|
+
passed: boolean;
|
|
328
|
+
severity: 'critical' | 'high' | 'medium' | 'low';
|
|
329
|
+
detail: string;
|
|
330
|
+
}
|
|
331
|
+
interface TraceInsightReadiness {
|
|
332
|
+
score: number;
|
|
333
|
+
grade: 'external-ready' | 'internal-review' | 'raw-analysis';
|
|
334
|
+
gates: TraceInsightQualityGate[];
|
|
335
|
+
}
|
|
336
|
+
declare function tokenizeDomainWords(value: string): string[];
|
|
337
|
+
declare function inferDomainKeywords(suite: TraceInsightSuite): string[];
|
|
338
|
+
declare function domainEvidencePattern(keywords: string[]): RegExp;
|
|
339
|
+
declare function describeTraceInsightScope(suite: TraceInsightSuite): string;
|
|
340
|
+
declare function planTraceInsightQuestions(input: TraceInsightPromptInput): TraceInsightQuestion[];
|
|
341
|
+
declare function buildTraceInsightContext(input: TraceInsightPromptInput): TraceInsightContext;
|
|
342
|
+
declare function scoreTraceInsightReadiness(context: TraceInsightContext): TraceInsightReadiness;
|
|
343
|
+
declare function defaultTraceInsightPanel(): TraceInsightPanelRole[];
|
|
344
|
+
declare function buildTraceInsightPrompt(input: TraceInsightPromptInput): string;
|
|
345
|
+
//#endregion
|
|
346
|
+
//#region src/trace-analyst/otlp-flatten.d.ts
|
|
347
|
+
interface OtlpFlatLine {
|
|
348
|
+
trace_id: string;
|
|
349
|
+
span_id: string;
|
|
350
|
+
parent_span_id: string | null;
|
|
351
|
+
name: string;
|
|
352
|
+
kind: string;
|
|
353
|
+
start_time: string;
|
|
354
|
+
end_time: string;
|
|
355
|
+
status: {
|
|
356
|
+
code: 'STATUS_CODE_OK' | 'STATUS_CODE_ERROR' | 'STATUS_CODE_UNSET';
|
|
357
|
+
message?: string;
|
|
358
|
+
};
|
|
359
|
+
resource: {
|
|
360
|
+
attributes: Record<string, string | number | boolean>;
|
|
361
|
+
};
|
|
362
|
+
attributes: Record<string, string | number | boolean>;
|
|
363
|
+
events?: Array<{
|
|
364
|
+
name: string;
|
|
365
|
+
timeUnixNano?: string;
|
|
366
|
+
attributes?: Record<string, unknown>;
|
|
367
|
+
}>;
|
|
368
|
+
}
|
|
369
|
+
interface FlattenOtlpOptions {
|
|
370
|
+
/** `'openinference'` (default) maps source per-span attributes into the
|
|
371
|
+
* canonical OpenInference vocabulary the analyst readers consume. `'none'`
|
|
372
|
+
* passes attributes through untouched. */
|
|
373
|
+
attributeVocabulary?: 'openinference' | 'none';
|
|
374
|
+
/** Override the numeric-kind → otlp-string mapping. */
|
|
375
|
+
kindMap?: Partial<Record<number, string>>;
|
|
376
|
+
}
|
|
377
|
+
declare function flattenOtlpExportToNdjson(otlpExport: OtlpExport, opts?: FlattenOtlpOptions): OtlpFlatLine[];
|
|
378
|
+
//#endregion
|
|
379
|
+
//#region src/trace-analyst/otlp-span.d.ts
|
|
380
|
+
/**
|
|
381
|
+
* The structural fields a flat OTLP-JSONL line projects to. `attributes`
|
|
382
|
+
* is the merged resource+span attribute map (span overrides resource);
|
|
383
|
+
* the named fields are the pivots every reader of a trace needs without
|
|
384
|
+
* paying the full attribute materialisation.
|
|
385
|
+
*/
|
|
386
|
+
interface ProjectedOtlpSpan {
|
|
387
|
+
trace_id: string;
|
|
388
|
+
span_id: string;
|
|
389
|
+
parent_span_id: string | null;
|
|
390
|
+
name: string;
|
|
391
|
+
kind: TraceAnalystSpanKind;
|
|
392
|
+
start_time: string;
|
|
393
|
+
end_time: string;
|
|
394
|
+
duration_ms: number;
|
|
395
|
+
status: TraceAnalystSpanStatus;
|
|
396
|
+
status_message: string | undefined;
|
|
397
|
+
service_name: string | null;
|
|
398
|
+
agent_name: string | null;
|
|
399
|
+
model_name: string | null;
|
|
400
|
+
tool_name: string | null;
|
|
401
|
+
/** Merged resource + span attributes, span winning on overlap. */
|
|
402
|
+
attributes: Record<string, unknown>;
|
|
403
|
+
}
|
|
404
|
+
/**
|
|
405
|
+
* Project one parsed OTLP-JSONL object to `ProjectedOtlpSpan`, or `null`
|
|
406
|
+
* when the line is missing the mandatory `trace_id` + `span_id`.
|
|
407
|
+
*/
|
|
408
|
+
declare function projectOtlpFlatLine(raw: Record<string, unknown>): ProjectedOtlpSpan | null;
|
|
409
|
+
declare function readOtlpStatus(raw: Record<string, unknown>): {
|
|
410
|
+
code: TraceAnalystSpanStatus;
|
|
411
|
+
message: string | undefined;
|
|
412
|
+
};
|
|
413
|
+
declare function inferOtlpKind(attrs: Record<string, unknown>): TraceAnalystSpanKind;
|
|
414
|
+
/**
|
|
415
|
+
* Flatten OTLP `attributes` + `resource.attributes` into a single
|
|
416
|
+
* dotted-key map. Span attributes override resource attributes when keys
|
|
417
|
+
* overlap. Nested objects/arrays are preserved as-is.
|
|
418
|
+
*/
|
|
419
|
+
declare function extractOtlpAttributes(raw: Record<string, unknown>): Record<string, unknown>;
|
|
420
|
+
declare function stringField(raw: Record<string, unknown>, key: string): string | undefined;
|
|
421
|
+
declare function asString(v: unknown): string | null;
|
|
422
|
+
/** First non-empty string value across a list of candidate attribute keys. */
|
|
423
|
+
declare function firstStringAttr(attrs: Record<string, unknown>, keys: readonly string[]): string | null;
|
|
424
|
+
//#endregion
|
|
425
|
+
//#region src/trace-analyst/otlp-to-run-records.d.ts
|
|
426
|
+
interface OtlpToRunRecordsOptions {
|
|
427
|
+
/** Logical experiment grouping for every produced record. */
|
|
428
|
+
experimentId: string;
|
|
429
|
+
/** Candidate (variant) id — the surface these traces exercised. The
|
|
430
|
+
* bench passes the proposer label here so `compareOptimizationMethods` can pair rows. */
|
|
431
|
+
candidateId: string;
|
|
432
|
+
/** Split assignment for every produced record. Default `'holdout'` —
|
|
433
|
+
* ingested traces are evidence, not the optimizer's training pool. */
|
|
434
|
+
splitTag?: RunSplitTag;
|
|
435
|
+
/** Git SHA the traces were produced from. Default `'unknown'`. */
|
|
436
|
+
commitSha?: string;
|
|
437
|
+
/** sha256 of the effective prompt surface. Default `'unknown'`. */
|
|
438
|
+
promptHash?: string;
|
|
439
|
+
/** sha256 of the effective config. Default `'unknown'`. */
|
|
440
|
+
configHash?: string;
|
|
441
|
+
/** RNG seed recorded on every row. Default 0. */
|
|
442
|
+
seed?: number;
|
|
443
|
+
/**
|
|
444
|
+
* Fallback model snapshot when the trace exposes no LLM model attribute
|
|
445
|
+
* OR exposes a bare alias `validateRunRecord` would reject. The trace's
|
|
446
|
+
* own model wins when it already carries a snapshot. Default
|
|
447
|
+
* `'unknown@otlp'` (opaque-snapshot form the validator accepts).
|
|
448
|
+
*/
|
|
449
|
+
fallbackModel?: string;
|
|
450
|
+
/**
|
|
451
|
+
* USD per total token (input+output) used to price a trace when no
|
|
452
|
+
* per-span cost attribute is present. When unset, an unpriced trace
|
|
453
|
+
* records `costUsd: null` and `raw.cost_unpriced = 1`.
|
|
454
|
+
*/
|
|
455
|
+
priceUsdPerToken?: number;
|
|
456
|
+
/**
|
|
457
|
+
* Map each OTLP `trace_id` to the logical run it belongs to. Use this when a
|
|
458
|
+
* provider emits several traces for one task attempt. All mapped traces are
|
|
459
|
+
* aggregated into one record; duplicate span ids remain isolated by their
|
|
460
|
+
* source trace. The callback must return a non-empty id for every trace.
|
|
461
|
+
*
|
|
462
|
+
* When omitted, every `trace_id` remains an independent run.
|
|
463
|
+
*/
|
|
464
|
+
logicalRunIdForTrace?: (traceId: string) => string;
|
|
465
|
+
/**
|
|
466
|
+
* Score for a produced run's outcome (AppWorld `world.evaluate()` →
|
|
467
|
+
* TGC/SGC, or
|
|
468
|
+
* any [0,1] task-success signal). Keyed by the logical run id when
|
|
469
|
+
* `logicalRunIdForTrace` is supplied, otherwise by `trace_id`. When the map
|
|
470
|
+
* has no entry or the function returns undefined, the record remains
|
|
471
|
+
* unlabeled.
|
|
472
|
+
*/
|
|
473
|
+
scoreForTrace?: (runId: string, span: TraceAggregate) => number | undefined;
|
|
474
|
+
/**
|
|
475
|
+
* Per-record judge metadata when an external judge produced the score.
|
|
476
|
+
* Keyed by the logical run id when supplied, otherwise by `trace_id`.
|
|
477
|
+
*/
|
|
478
|
+
judgeMetadataForTrace?: (runId: string) => RunRecord['judgeMetadata'] | undefined;
|
|
479
|
+
}
|
|
480
|
+
/** A `RunRecord` plus the verbatim prompt/completion text when the trace's
|
|
481
|
+
* LLM spans exposed it. The text is NOT on the validated `RunRecord`
|
|
482
|
+
* (`outcome.raw` is numeric-only) but consumers ingesting full traces want
|
|
483
|
+
* it — so it rides alongside. */
|
|
484
|
+
interface OtlpTraceRunRecord {
|
|
485
|
+
record: RunRecord;
|
|
486
|
+
/** Verbatim first-LLM-span `input.value`, when present. */
|
|
487
|
+
promptText?: string;
|
|
488
|
+
/** Verbatim last-LLM-span `output.value`, when present. */
|
|
489
|
+
completionText?: string;
|
|
490
|
+
}
|
|
491
|
+
/** Per-trace rollup the score callback can inspect. */
|
|
492
|
+
interface TraceAggregate {
|
|
493
|
+
traceId: string;
|
|
494
|
+
/** Number of source OTLP traces folded into this run. */
|
|
495
|
+
sourceTraceCount: number;
|
|
496
|
+
/** Source OTLP trace ids folded into this run, sorted for deterministic audit output. */
|
|
497
|
+
sourceTraceIds: readonly string[];
|
|
498
|
+
spanCount: number;
|
|
499
|
+
llmSpanCount: number;
|
|
500
|
+
toolSpanCount: number;
|
|
501
|
+
agentSpanCount: number;
|
|
502
|
+
errorSpanCount: number;
|
|
503
|
+
executionErrorCount: number;
|
|
504
|
+
processErrorCount: number;
|
|
505
|
+
guardrailErrorCount: number;
|
|
506
|
+
judgeErrorCount: number;
|
|
507
|
+
propagatedErrorCount: number;
|
|
508
|
+
unclassifiedErrorCount: number;
|
|
509
|
+
tokenUsage: RunTokenUsage;
|
|
510
|
+
/** First error span's normalized status message, if any. */
|
|
511
|
+
firstErrorMessage?: string;
|
|
512
|
+
model: string;
|
|
513
|
+
startTime: string;
|
|
514
|
+
endTime: string;
|
|
515
|
+
wallMs: number;
|
|
516
|
+
/** Root-span terminal result. Child span errors do not change this value. */
|
|
517
|
+
terminalOutcome: RunTerminalOutcome;
|
|
518
|
+
}
|
|
519
|
+
/**
|
|
520
|
+
* Parse + aggregate an OTLP traces.jsonl string into validated
|
|
521
|
+
* `RunRecord[]` (one per trace). Use {@link otlpToTraceRunRecords} when you
|
|
522
|
+
* also want the verbatim prompt/completion text alongside each record.
|
|
523
|
+
*/
|
|
524
|
+
declare function otlpToRunRecords(otlpJsonl: string, opts: OtlpToRunRecordsOptions): RunRecord[];
|
|
525
|
+
/**
|
|
526
|
+
* Aggregate already-parsed OTLP flat rows without serializing them back to
|
|
527
|
+
* JSONL. This is the in-memory counterpart to {@link otlpToRunRecords}; both
|
|
528
|
+
* paths share projection, reconciliation, validation, and ordering.
|
|
529
|
+
*/
|
|
530
|
+
declare function otlpRowsToRunRecords(rows: Iterable<object>, opts: OtlpToRunRecordsOptions): RunRecord[];
|
|
531
|
+
/** As {@link otlpToRunRecords} but returns the prompt/completion text too. */
|
|
532
|
+
declare function otlpToTraceRunRecords(otlpJsonl: string, opts: OtlpToRunRecordsOptions): OtlpTraceRunRecord[];
|
|
533
|
+
/** Parsed-row counterpart to {@link otlpToTraceRunRecords}. */
|
|
534
|
+
declare function otlpRowsToTraceRunRecords(rows: Iterable<object>, opts: OtlpToRunRecordsOptions): OtlpTraceRunRecord[];
|
|
535
|
+
//#endregion
|
|
536
|
+
//#region src/trace-analyst/prompts.d.ts
|
|
537
|
+
/** Ax RLM prompt for bounded trace discovery and evidence-backed analysis. */
|
|
538
|
+
declare const TRACE_ANALYST_ACTOR_DESCRIPTION = "You answer questions about an OTLP-shaped JSONL trace dataset using the trace tools provided in the `traces` namespace.\n\nDISCOVERY → NARROW → DEEP-READ protocol — follow exactly:\n\n1. ALWAYS call `traces.getDatasetOverview({})` FIRST without a regex_pattern. The result tells you total_traces, raw_jsonl_bytes, services, agents, models, and sample_trace_ids (real ids — never fabricate one).\n\n2. Use raw_jsonl_bytes to gauge how expensive raw scans will be. `filters.regex_pattern` is the one scan-heavy filter on getDatasetOverview / queryTraces / countTraces — narrow with indexed fields (has_errors, model_names, service_names, agent_names, time bounds) BEFORE adding a regex on a large dataset.\n\n3. To list more traces than the sample, call `traces.queryTraces({ filters?, limit, offset? })`. Each summary carries raw_jsonl_bytes — use it to choose between viewTrace and searchTrace BEFORE calling either.\n\n4. Per-trace inspection:\n - SMALL trace (raw_jsonl_bytes well under 150_000): call `traces.viewTrace({ trace_id })`. Returns all spans. Per-attribute payloads are head-capped at ~4KB; large `input.value` / `output.value` / `llm.input_messages` will show a `[trace-analyst truncated: N bytes]` marker.\n - LARGE trace (raw_jsonl_bytes near or above 150_000, or you saw an `oversized` response): use `traces.searchTrace({ trace_id, regex_pattern })` to get bounded SpanMatchRecords (span metadata + matched text + surrounding context). Then call `traces.viewSpans({ trace_id, span_ids: [...] })` for surgical reads (~16KB cap, 4× higher than discovery), or `traces.searchSpan({ trace_id, span_id, regex_pattern })` for one large span. Stays bounded regardless of trace size.\n - Useful regex patterns: `STATUS_CODE_ERROR` (failures), tool names like `grep` or `view_trace`, error strings like `MaxTurnsExceeded`, model names, attribute keys.\n\n5. ONLY call viewTrace / viewSpans / searchTrace / searchSpan with trace/span ids you have already seen in sample_trace_ids, a queryTraces page, or a previous search result. Never invent ids.\n\n5a. **Result-shape contract** — searchTrace and searchSpan return `{ trace_id, hits, total_matches, has_more }`. Iterate `result.hits` (NOT result.matches). Each hit has `{ span_id, span_name, span_kind, attribute_path, matched_text, context_before, context_after, match_offset }`. viewTrace returns `{ trace_id, spans }` (or `oversized`). viewSpans returns `{ trace_id, spans, missing_span_ids, truncated_attribute_count }`. Never assume a field name — log the result shape first if unsure.\n\n6. If viewTrace returns an `oversized` summary instead of `spans`, DO NOT retry the same call. Read the summary's top_span_names, span_count, span_response_bytes_max, error_span_count to plan a follow-up: switch to searchTrace (or searchSpan for one large span), then viewSpans on a smaller, surgical span_ids set.\n\n7. If searchTrace or searchSpan returns has_more=true, REFINE the regex to be more specific rather than blindly raising max_matches.\n\n8. If a tool errors (invalid regex, range error), STOP and reconsider — don't retry with a guessed id or argument. Use the discovery tools above to recover.\n\n9. If a ~4KB-truncated payload from viewTrace / searchTrace matters for your answer, first try viewSpans on that span id (~16KB cap). If a 16KB-truncated payload from viewSpans still matters, narrow further with searchSpan against a more specific regex rather than asking for the full payload again.\n\n10. If the question splits into independent reasoning branches, use bounded `llmQuery(...)` calls over evidence you already loaded. Subqueries cannot inspect the trace store, so pass the exact trace excerpts they need. Example:\n\n const reviews = await llmQuery([\n { query: 'Classify the failure mechanism in this excerpt.', context: traceAbcExcerpt },\n { query: 'Classify the failure mechanism in this excerpt.', context: traceDefExcerpt },\n ]);\n\nOBSERVABILITY rules:\n- Each discovery turn must emit at least one concise `console.log(...)` showing what evidence was learned.\n- Finish gathering evidence before submitting the analysis.\n- Reuse runtime variables across turns; don't recompute.\n\nOUTPUT contract — your final answer must include:\n- A clear prose conclusion answering the user's question.\n- Trace ids and span ids cited as evidence for each claim.\n- Failure modes named in the user's domain language, with frequency and concrete examples.\n- A concise findings array containing only claims supported by inspected evidence.\n\nDo NOT invent trace ids, span ids, error messages, or model names. Every fact must be traceable to a tool result.";
|
|
539
|
+
declare const TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION = "trace-analyst-actor-v6-2026-07-14";
|
|
540
|
+
//#endregion
|
|
541
|
+
//#region src/trace-analyst/store-otlp.d.ts
|
|
542
|
+
interface OtlpFileTraceStoreOptions {
|
|
543
|
+
/** Path to the OTLP-JSONL file. */
|
|
544
|
+
path: string;
|
|
545
|
+
/** Override the discovery (`viewTrace`) per-attribute byte cap. */
|
|
546
|
+
perAttributeViewBudget?: number;
|
|
547
|
+
/** Override the surgical (`viewSpans`) per-attribute byte cap. */
|
|
548
|
+
perAttributeSpanBudget?: number;
|
|
549
|
+
/** Override the per-call ceiling that triggers oversized summaries. */
|
|
550
|
+
perCallByteCeiling?: number;
|
|
551
|
+
/** Override the per-match text budget. */
|
|
552
|
+
perMatchTextBudget?: number;
|
|
553
|
+
/**
|
|
554
|
+
* Hard ceiling on the trace file size in bytes. The store reads the
|
|
555
|
+
* whole file into one Buffer and indexes it in memory, so an
|
|
556
|
+
* unbounded file OOMs the process. Above this size the store fails
|
|
557
|
+
* loud with `TraceFileTooLargeError` instead of degrading silently.
|
|
558
|
+
* Default 256 MiB.
|
|
559
|
+
*/
|
|
560
|
+
maxFileBytes?: number;
|
|
561
|
+
}
|
|
562
|
+
declare class OtlpFileTraceStore implements TraceAnalysisStore {
|
|
563
|
+
private readonly path;
|
|
564
|
+
private readonly perAttributeViewBudget;
|
|
565
|
+
private readonly perAttributeSpanBudget;
|
|
566
|
+
private readonly perCallByteCeiling;
|
|
567
|
+
private readonly perMatchTextBudget;
|
|
568
|
+
private readonly maxFileBytes;
|
|
569
|
+
private indexPromise?;
|
|
570
|
+
/** Cached UTF-8 buffer of the file. We pin it once because every
|
|
571
|
+
* read needs slice access and re-reading on each call balloons the
|
|
572
|
+
* syscall count. */
|
|
573
|
+
private bufferPromise?;
|
|
574
|
+
constructor(opts: OtlpFileTraceStoreOptions);
|
|
575
|
+
getOverview(filters?: TraceAnalystFilters): Promise<DatasetOverview>;
|
|
576
|
+
queryTraces(opts: {
|
|
577
|
+
filters?: TraceAnalystFilters;
|
|
578
|
+
limit: number;
|
|
579
|
+
offset?: number;
|
|
580
|
+
}): Promise<QueryTracesPage>;
|
|
581
|
+
countTraces(filters?: TraceAnalystFilters): Promise<number>;
|
|
582
|
+
viewTrace(opts: {
|
|
583
|
+
trace_id: string;
|
|
584
|
+
per_attribute_byte_cap?: number;
|
|
585
|
+
}): Promise<ViewTraceResult>;
|
|
586
|
+
viewSpans(opts: {
|
|
587
|
+
trace_id: string;
|
|
588
|
+
span_ids: readonly string[];
|
|
589
|
+
per_attribute_byte_cap?: number;
|
|
590
|
+
}): Promise<ViewSpansResult>;
|
|
591
|
+
searchTrace(opts: {
|
|
592
|
+
trace_id: string;
|
|
593
|
+
regex_pattern: string;
|
|
594
|
+
max_matches?: number;
|
|
595
|
+
}): Promise<SearchTraceResult>;
|
|
596
|
+
searchSpan(opts: {
|
|
597
|
+
trace_id: string;
|
|
598
|
+
span_id: string;
|
|
599
|
+
regex_pattern: string;
|
|
600
|
+
max_matches?: number;
|
|
601
|
+
}): Promise<SearchSpanResult>;
|
|
602
|
+
/** Force the index to materialise. Useful to amortise startup cost
|
|
603
|
+
* before the first agent call. */
|
|
604
|
+
ensureIndexed(): Promise<void>;
|
|
605
|
+
private buffer;
|
|
606
|
+
/** Stat-then-read so an oversized file fails loud BEFORE we allocate a
|
|
607
|
+
* multi-hundred-MB Buffer and OOM the process. A missing file surfaces
|
|
608
|
+
* as TraceFileMissingError; any other stat/read error propagates. */
|
|
609
|
+
private readGuarded;
|
|
610
|
+
private index;
|
|
611
|
+
private buildIndex;
|
|
612
|
+
private matchedTraces;
|
|
613
|
+
private toSummary;
|
|
614
|
+
private projectSpan;
|
|
615
|
+
private buildOversizedSummary;
|
|
616
|
+
private scanSpanForMatches;
|
|
617
|
+
}
|
|
618
|
+
declare class TraceFileMissingError extends NotFoundError {
|
|
619
|
+
constructor(path: string);
|
|
620
|
+
}
|
|
621
|
+
declare class TraceNotFoundError extends NotFoundError {
|
|
622
|
+
readonly trace_id: string;
|
|
623
|
+
constructor(trace_id: string);
|
|
624
|
+
}
|
|
625
|
+
declare class SpanNotFoundError extends NotFoundError {
|
|
626
|
+
readonly trace_id: string;
|
|
627
|
+
readonly span_id: string;
|
|
628
|
+
constructor(trace_id: string, span_id: string);
|
|
629
|
+
}
|
|
630
|
+
//#endregion
|
|
631
|
+
//#region src/trace-analyst/tools.d.ts
|
|
632
|
+
interface BuildTraceAnalystToolsOpts {
|
|
633
|
+
store: TraceAnalysisStore;
|
|
634
|
+
/** Override the default sample-trace-id slot count (20). Mostly for tests. */
|
|
635
|
+
sampleTraceLimit?: number;
|
|
636
|
+
}
|
|
637
|
+
/**
|
|
638
|
+
* Build the trace-analyst function set. Pass the result into
|
|
639
|
+
* `agent(...).functions.local`.
|
|
640
|
+
*/
|
|
641
|
+
declare function buildTraceAnalystTools(opts: BuildTraceAnalystToolsOpts): AxFunction[];
|
|
642
|
+
/**
|
|
643
|
+
* Convenience: same shape as `buildTraceAnalystTools` but returns the
|
|
644
|
+
* grouped form expected when registering trace tools alongside other
|
|
645
|
+
* agent function modules. */
|
|
646
|
+
declare function traceAnalystFunctionGroup(opts: BuildTraceAnalystToolsOpts): {
|
|
647
|
+
namespace: string;
|
|
648
|
+
title: string;
|
|
649
|
+
selectionCriteria: string;
|
|
650
|
+
description: string;
|
|
651
|
+
functions: AxFunction[];
|
|
652
|
+
};
|
|
653
|
+
//#endregion
|
|
654
|
+
//#region src/replay.d.ts
|
|
655
|
+
declare class ReplayCacheMissError extends ReplayError {
|
|
656
|
+
readonly url: string;
|
|
657
|
+
readonly requestKey: string;
|
|
658
|
+
constructor(url: string, requestKey: string, message?: string);
|
|
659
|
+
}
|
|
660
|
+
interface ReplayCacheEntry {
|
|
661
|
+
request: RawProviderEvent;
|
|
662
|
+
response: RawProviderEvent;
|
|
663
|
+
}
|
|
664
|
+
interface ReplayCacheStats {
|
|
665
|
+
total: number;
|
|
666
|
+
byProvider: Record<string, number>;
|
|
667
|
+
byModel: Record<string, number>;
|
|
668
|
+
/** Spans for which we have a request but no response (run aborted mid-call). */
|
|
669
|
+
orphanRequests: number;
|
|
670
|
+
}
|
|
671
|
+
/**
|
|
672
|
+
* In-memory deterministic cache of (request → response) keyed on a stable
|
|
673
|
+
* hash of the request body. Built from a `RawProviderSink` containing
|
|
674
|
+
* paired `request` and `response` events from a previous run.
|
|
675
|
+
*
|
|
676
|
+
* The cache is the source of truth for replay; `createReplayFetch` is a
|
|
677
|
+
* thin wrapper that reads from it.
|
|
678
|
+
*/
|
|
679
|
+
declare class ReplayCache {
|
|
680
|
+
private byKey;
|
|
681
|
+
private orphans;
|
|
682
|
+
private byProvider;
|
|
683
|
+
private byModel;
|
|
684
|
+
/**
|
|
685
|
+
* Build a cache from a sink's events. The sink must implement `list()`.
|
|
686
|
+
* Filter by `runId` / `spanId` to scope to a specific replay.
|
|
687
|
+
*/
|
|
688
|
+
static fromSink(sink: RawProviderSink, filter?: {
|
|
689
|
+
runId?: string;
|
|
690
|
+
spanId?: string;
|
|
691
|
+
}): Promise<ReplayCache>;
|
|
692
|
+
/** Build a cache from an in-memory event list. */
|
|
693
|
+
static fromEvents(events: RawProviderEvent[]): Promise<ReplayCache>;
|
|
694
|
+
/** Number of cacheable (request, response) pairs in the cache. */
|
|
695
|
+
size(): number;
|
|
696
|
+
stats(): ReplayCacheStats;
|
|
697
|
+
/** Iterate every cached `(request, response)` pair in insertion order. */
|
|
698
|
+
entries(): IterableIterator<ReplayCacheEntry>;
|
|
699
|
+
/**
|
|
700
|
+
* Look up a cached response by hashing the (model, messages, temperature,
|
|
701
|
+
* maxTokens, response_format) shape. Returns `undefined` on miss; the
|
|
702
|
+
* caller decides whether to throw, fall back to the network, or skip.
|
|
703
|
+
*/
|
|
704
|
+
lookup(requestBody: unknown): Promise<ReplayCacheEntry | undefined>;
|
|
705
|
+
}
|
|
706
|
+
interface ReplayFetchOptions {
|
|
707
|
+
/**
|
|
708
|
+
* Behaviour on cache miss. Default `'throw'`. `'fallback'` calls the
|
|
709
|
+
* `fallbackFetch` (typically `globalThis.fetch`) so a partial replay can
|
|
710
|
+
* still complete; `'fail-closed'` returns a synthetic 599 response so the
|
|
711
|
+
* call site sees a non-retriable failure.
|
|
712
|
+
*/
|
|
713
|
+
onMiss?: 'throw' | 'fallback' | 'fail-closed';
|
|
714
|
+
fallbackFetch?: typeof fetch;
|
|
715
|
+
/** Optional callback fired once per replayed call (for telemetry / counters). */
|
|
716
|
+
onHit?: (info: {
|
|
717
|
+
url: string;
|
|
718
|
+
provider: string;
|
|
719
|
+
model: string;
|
|
720
|
+
}) => void;
|
|
721
|
+
/** Optional callback fired on cache miss before the `onMiss` policy applies. */
|
|
722
|
+
onMissNotify?: (info: {
|
|
723
|
+
url: string;
|
|
724
|
+
requestBody: unknown;
|
|
725
|
+
}) => void;
|
|
726
|
+
}
|
|
727
|
+
/**
|
|
728
|
+
* Build a `fetch`-shaped function that serves cached responses out of a
|
|
729
|
+
* `ReplayCache` for any URL ending in `/chat/completions`. Pass through
|
|
730
|
+
* `LlmClientOptions.fetch` and `callLlm` becomes free.
|
|
731
|
+
*
|
|
732
|
+
* Non-`/chat/completions` URLs are passed straight to the fallback fetch
|
|
733
|
+
* (default: `globalThis.fetch`). This matters because non-LLM HTTP work
|
|
734
|
+
* (judge HTTP servers, sandbox callbacks) sometimes flows through the same
|
|
735
|
+
* `fetch` and shouldn't be intercepted.
|
|
736
|
+
*/
|
|
737
|
+
declare function createReplayFetch(cache: ReplayCache, opts?: ReplayFetchOptions): typeof fetch;
|
|
738
|
+
/**
|
|
739
|
+
* Convenience iterator over `(request, response)` pairs in a sink — for
|
|
740
|
+
* post-hoc scoring that doesn't need a `fetch` shim. The judge or scorer
|
|
741
|
+
* runs purely in-process over cached LLM outputs.
|
|
742
|
+
*/
|
|
743
|
+
declare function iterateRawCalls(sink: RawProviderSink, filter?: {
|
|
744
|
+
runId?: string;
|
|
745
|
+
spanId?: string;
|
|
746
|
+
}): AsyncGenerator<ReplayCacheEntry>;
|
|
747
|
+
//#endregion
|
|
748
|
+
export { TraceAnalystHookOptions as $, readOtlpStatus as A, TraceInsightQuestion as B, otlpToTraceRunRecords as C, traceSpanKindToOpenInferenceKind as Ct, firstStringAttr as D, OtlpSpan as Dt, extractOtlpAttributes as E, OtlpResourceSpans as Et, TraceInsightContext as F, buildTraceInsightPrompt as G, TraceInsightSuite as H, TraceInsightFinding as I, domainEvidencePattern as J, defaultTraceInsightPanel as K, TraceInsightPanelRole as L, FlattenOtlpOptions as M, OtlpFlatLine as N, inferOtlpKind as O, exportRunAsOtlp as Ot, flattenOtlpExportToNdjson as P, tokenizeDomainWords as Q, TraceInsightPromptInput as R, otlpToRunRecords as S, isOtlpModelCall as St, asString as T, OtlpExport as Tt, TraceInsightTask as U, TraceInsightReadiness as V, buildTraceInsightContext as W, planTraceInsightQuestions as X, inferDomainKeywords as Y, scoreTraceInsightReadiness as Z, OtlpToRunRecordsOptions as _, OtlpSpanRole as _t, ReplayFetchOptions as a, DEFAULT_REDACTION_RULES as at, otlpRowsToRunRecords as b, applyToolSpanOtlpAttributes as bt, buildTraceAnalystTools as c, RedactionRule as ct, OtlpFileTraceStoreOptions as d, createOtelTracingStore as dt, traceAnalystOnRunComplete as et, SpanNotFoundError as f, otelRunCompleteHook as ft, TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION as g, createOtelExporter as gt, TRACE_ANALYST_ACTOR_DESCRIPTION as h, OtelExporter as ht, ReplayCacheStats as i, convertTraceStoresToOtlp as it, stringField as j, projectOtlpFlatLine as k, traceAnalystFunctionGroup as l, redactString as lt, TraceNotFoundError as m, OtelExportConfig as mt, ReplayCacheEntry as n, TraceStoreToOtlpOptions as nt, createReplayFetch as o, REDACTION_VERSION as ot, TraceFileMissingError as p, ExportableSpan as pt, describeTraceInsightScope as q, ReplayCacheMissError as r, TracesToOtlpResult as rt, iterateRawCalls as s, RedactionReport as st, ReplayCache as t, TraceStoreSource as tt, OtlpFileTraceStore as u, redactValue as ut, OtlpTraceRunRecord as v, OtlpSpanRoleInput as vt, ProjectedOtlpSpan as w, OTEL_AGENT_EVAL_SCOPE as wt, otlpRowsToTraceRunRecords as x, classifyOtlpSpanRole as xt, TraceAggregate as y, ToolSpanOtlpInput as yt, TraceInsightQualityGate as z };
|
|
749
|
+
//# sourceMappingURL=replay-OoidtG1E.d.ts.map
|