@tangle-network/agent-eval 0.144.11 → 0.144.13
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +11 -0
- package/dist/agent-profile-B9_GGsG8.d.ts +84 -0
- package/dist/agent-profile-B9_GGsG8.d.ts.map +1 -0
- package/dist/{agent-profile-cell-BOP-iA9Q.d.ts → agent-profile-cell-BkcRDikH.d.ts} +2 -2
- package/dist/{agent-profile-cell-BOP-iA9Q.d.ts.map → agent-profile-cell-BkcRDikH.d.ts.map} +1 -1
- package/dist/analyst/index.d.ts +134 -16
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +364 -10
- package/dist/analyst/index.js.map +1 -1
- package/dist/backend-integrity-CsVin_Wb.d.ts +280 -0
- package/dist/backend-integrity-CsVin_Wb.d.ts.map +1 -0
- package/dist/{baseline-C-GocmIW.js → baseline-BhPRQBVn.js} +3 -2
- package/dist/{baseline-C-GocmIW.js.map → baseline-BhPRQBVn.js.map} +1 -1
- package/dist/{benchmark-CWeqGl7x.js → benchmark-BhT16ep9.js} +2 -2
- package/dist/{benchmark-CWeqGl7x.js.map → benchmark-BhT16ep9.js.map} +1 -1
- package/dist/{agent-profile-DPi7IZg7.d.ts → benchmark-Ceoan7vk.d.ts} +4 -91
- package/dist/benchmark-Ceoan7vk.d.ts.map +1 -0
- package/dist/{benchmark-command-BteMFN62.js → benchmark-command-CY6cbhOB.js} +36 -79
- package/dist/benchmark-command-CY6cbhOB.js.map +1 -0
- package/dist/benchmarks/index.d.ts +244 -2
- package/dist/benchmarks/index.d.ts.map +1 -0
- package/dist/benchmarks/index.js +733 -1
- package/dist/benchmarks/index.js.map +1 -0
- package/dist/builder-eval/index.d.ts +23 -2
- package/dist/builder-eval/index.d.ts.map +1 -1
- package/dist/builder-eval/index.js +227 -3
- package/dist/builder-eval/index.js.map +1 -1
- package/dist/campaign/index.d.ts +10 -8
- package/dist/campaign/index.js +9 -6
- package/dist/{campaign-C2TTzQII.js → campaign-BYjBAypg.js} +21 -634
- package/dist/campaign-BYjBAypg.js.map +1 -0
- package/dist/{canonical-D011XM8r.js → canonical-D-XsTQ6_.js} +2 -2
- package/dist/{canonical-D011XM8r.js.map → canonical-D-XsTQ6_.js.map} +1 -1
- package/dist/{index-CQTZ-4XN.d.ts → capture-fetch-DDvpjVRU.d.ts} +3 -3
- package/dist/capture-fetch-DDvpjVRU.d.ts.map +1 -0
- package/dist/{default-registry-BmktKy8r.js → chat-client-Bp4Ebuuc.js} +1276 -1276
- package/dist/chat-client-Bp4Ebuuc.js.map +1 -0
- package/dist/cli.js +2 -2
- package/dist/{client-C9gzZE59.d.ts → client-KUbGulm_.d.ts} +4 -4
- package/dist/{client-C9gzZE59.d.ts.map → client-KUbGulm_.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +13 -390
- package/dist/contract/index.d.ts.map +1 -1
- package/dist/contract/index.js +18 -542
- package/dist/contract/index.js.map +1 -1
- package/dist/{cost-ledger-DMFxsLKr.js → cost-ledger-BSe92yAV.js} +3 -3
- package/dist/{cost-ledger-DMFxsLKr.js.map → cost-ledger-BSe92yAV.js.map} +1 -1
- package/dist/{cost-ledger-Bv_e8XHY.d.ts → cost-ledger-DbQdN3nO.d.ts} +2 -2
- package/dist/{cost-ledger-Bv_e8XHY.d.ts.map → cost-ledger-DbQdN3nO.d.ts.map} +1 -1
- package/dist/{counterfactual-CxmxAONP.d.ts → counterfactual--bpysZF0.d.ts} +2 -14
- package/dist/{counterfactual-CxmxAONP.d.ts.map → counterfactual--bpysZF0.d.ts.map} +1 -1
- package/dist/{counterfactual-CWPTrMH7.js → counterfactual-lDfCx0Uz.js} +3 -31
- package/dist/{counterfactual-CWPTrMH7.js.map → counterfactual-lDfCx0Uz.js.map} +1 -1
- package/dist/{dataset-C8xaLXdY.d.ts → dataset-CJjKqQfA.d.ts} +2 -8
- package/dist/dataset-CJjKqQfA.d.ts.map +1 -0
- package/dist/{default-registry-Bf8Woqmq.d.ts → default-registry-Di6HP6pG.d.ts} +7 -12
- package/dist/default-registry-Di6HP6pG.d.ts.map +1 -0
- package/dist/{analyze-runs-C30yljDJ.js → define-agent-eval-iqjT--ZZ.js} +542 -130
- package/dist/define-agent-eval-iqjT--ZZ.js.map +1 -0
- package/dist/define-agent-eval-sH24zBfM.d.ts +388 -0
- package/dist/define-agent-eval-sH24zBfM.d.ts.map +1 -0
- package/dist/descriptive-B5MwKfbf.js +144 -0
- package/dist/descriptive-B5MwKfbf.js.map +1 -0
- package/dist/{dspy-rlm-engine-DbTk4JdR.js → dspy-rlm-engine-B_qhSc21.js} +4 -4
- package/dist/{dspy-rlm-engine-DbTk4JdR.js.map → dspy-rlm-engine-B_qhSc21.js.map} +1 -1
- package/dist/effect-sizes-DiH8MGOH.js +82 -0
- package/dist/effect-sizes-DiH8MGOH.js.map +1 -0
- package/dist/{engine-3hL-XqwJ.d.ts → engine-otFpE2gF.d.ts} +10 -38
- package/dist/engine-otFpE2gF.d.ts.map +1 -0
- package/dist/{errors-CKPfb2aH.d.ts → errors-DEE6u6ot.d.ts} +2 -14
- package/dist/{errors-CKPfb2aH.d.ts.map → errors-DEE6u6ot.d.ts.map} +1 -1
- package/dist/{errors-D-LKuDhb.js → errors-Dngq5h35.js} +2 -8
- package/dist/{errors-D-LKuDhb.js.map → errors-Dngq5h35.js.map} +1 -1
- package/dist/{eval-campaign-B_7wcnav.js → eval-campaign-UB-usSQ2.js} +6 -6
- package/dist/{eval-campaign-B_7wcnav.js.map → eval-campaign-UB-usSQ2.js.map} +1 -1
- package/dist/{exact-types-DSFFpLLI.d.ts → exact-types-BH1twmAJ.d.ts} +2 -2
- package/dist/{exact-types-DSFFpLLI.d.ts.map → exact-types-BH1twmAJ.d.ts.map} +1 -1
- package/dist/experiment/index.d.ts +9 -6
- package/dist/experiment/index.d.ts.map +1 -1
- package/dist/experiment/index.js +11 -7
- package/dist/experiment/index.js.map +1 -1
- package/dist/experiment-tracker-C29gXM4B.js +269 -0
- package/dist/experiment-tracker-C29gXM4B.js.map +1 -0
- package/dist/{experiment-tracker-IMntXr6J.d.ts → experiment-tracker-DWHZBAYL.d.ts} +77 -121
- package/dist/experiment-tracker-DWHZBAYL.d.ts.map +1 -0
- package/dist/{external-optimizer-contracts-CZuJNcT5.d.ts → external-optimizer-contracts-lixrOZdX.d.ts} +3 -3
- package/dist/{external-optimizer-contracts-CZuJNcT5.d.ts.map → external-optimizer-contracts-lixrOZdX.d.ts.map} +1 -1
- package/dist/external-optimizer-process-BTiNB-RH.js +301 -0
- package/dist/external-optimizer-process-BTiNB-RH.js.map +1 -0
- package/dist/{single-run-lock-DFWHEB09.js → external-optimizer-subprocess-DrJ9hR8u.js} +144 -438
- package/dist/external-optimizer-subprocess-DrJ9hR8u.js.map +1 -0
- package/dist/extract-usage-BrQ8mCLX.js +155 -0
- package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
- package/dist/{failure-cluster-CqcvCcdR.d.ts → failure-cluster-BLURuWG4.d.ts} +2 -3
- package/dist/failure-cluster-BLURuWG4.d.ts.map +1 -0
- package/dist/{feedback-trajectory-CacHpxsp.d.ts → feedback-trajectory-juOozjAc.d.ts} +5 -36
- package/dist/feedback-trajectory-juOozjAc.d.ts.map +1 -0
- package/dist/fuzz.d.ts +2 -2
- package/dist/fuzz.js +2 -2
- package/dist/hosted/index.d.ts +3 -3
- package/dist/{index-CvSN3IG1.d.ts → index-B8Ui1mr1.d.ts} +2 -2
- package/dist/{index-CvSN3IG1.d.ts.map → index-B8Ui1mr1.d.ts.map} +1 -1
- package/dist/{skill-usage-CO9OLRBx.d.ts → index-BWDrSVfw.d.ts} +11 -136
- package/dist/index-BWDrSVfw.d.ts.map +1 -0
- package/dist/index-Ba3YrbAL.d.ts +1 -0
- package/dist/{index-BnEuDAK2.d.ts → index-COQYtuRF.d.ts} +3 -3
- package/dist/{index-BnEuDAK2.d.ts.map → index-COQYtuRF.d.ts.map} +1 -1
- package/dist/{index-C3ssXVLv.d.ts → index-CvjYbU0D.d.ts} +2 -2
- package/dist/{index-C3ssXVLv.d.ts.map → index-CvjYbU0D.d.ts.map} +1 -1
- package/dist/{index-DPPGNJ_R.d.ts → index-DSmEylT9.d.ts} +17 -115
- package/dist/index-DSmEylT9.d.ts.map +1 -0
- package/dist/index.d.ts +2397 -5308
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +5914 -10496
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-CgX_s0Ez.d.ts → insight-report-CRi-Ufrj.d.ts} +4 -4
- package/dist/{insight-report-CgX_s0Ez.d.ts.map → insight-report-CRi-Ufrj.d.ts.map} +1 -1
- package/dist/{integrity-MLzHOfV9.js → integrity-Cy9WHAtb.js} +2 -2
- package/dist/{integrity-MLzHOfV9.js.map → integrity-Cy9WHAtb.js.map} +1 -1
- package/dist/{integrity-DY6tIbl0.js → integrity-DysDBWDu.js} +2 -2
- package/dist/{integrity-DY6tIbl0.js.map → integrity-DysDBWDu.js.map} +1 -1
- package/dist/{integrity-BuqEKu-x.d.ts → integrity-OrcI9Nau.d.ts} +3 -3
- package/dist/{integrity-BuqEKu-x.d.ts.map → integrity-OrcI9Nau.d.ts.map} +1 -1
- package/dist/internal-BDHPCnjk.js +230 -0
- package/dist/internal-BDHPCnjk.js.map +1 -0
- package/dist/judge-calibration-C5CbMYce.d.ts +117 -0
- package/dist/judge-calibration-C5CbMYce.d.ts.map +1 -0
- package/dist/judge-calibration-DZkWrm5H.js +317 -0
- package/dist/judge-calibration-DZkWrm5H.js.map +1 -0
- package/dist/{kind-factory-B8-r8-y8.js → kind-factory-CPmSd58s.js} +208 -208
- package/dist/{kind-factory-B8-r8-y8.js.map → kind-factory-CPmSd58s.js.map} +1 -1
- package/dist/ledger-core/index.d.ts +1 -1
- package/dist/ledger-core/index.js +2 -2
- package/dist/{ledger-core-DXZIqu17.js → ledger-core-BmZt19oQ.js} +110 -110
- package/dist/{ledger-core-DXZIqu17.js.map → ledger-core-BmZt19oQ.js.map} +1 -1
- package/dist/{llm-client-DzvMUsS_.js → llm-client-d0-2TT1g.js} +4 -64
- package/dist/{llm-client-DzvMUsS_.js.map → llm-client-d0-2TT1g.js.map} +1 -1
- package/dist/{skillopt-optimization-method-CQdVeM8k.js → llm-judge-dZ8P6nGI.js} +3349 -5401
- package/dist/llm-judge-dZ8P6nGI.js.map +1 -0
- package/dist/matrix/index.d.ts +2 -2
- package/dist/meta-eval/index.d.ts +3 -3
- package/dist/meta-eval/index.js +3 -3
- package/dist/{metrics-C9YY1OcL.js → metrics-Cl0L1KUy.js} +2 -108
- package/dist/{metrics-C9YY1OcL.js.map → metrics-Cl0L1KUy.js.map} +1 -1
- package/dist/{mint-B2O60ACG.js → mint-Dj9Ww_3I.js} +4 -4
- package/dist/{mint-B2O60ACG.js.map → mint-Dj9Ww_3I.js.map} +1 -1
- package/dist/{multi-layer-verifier-DnAqwl0h.d.ts → multi-layer-verifier-DIguZc8Z.d.ts} +3 -3
- package/dist/{multi-layer-verifier-DnAqwl0h.d.ts.map → multi-layer-verifier-DIguZc8Z.d.ts.map} +1 -1
- package/dist/multiplicity-DIWHvysC.d.ts +43 -0
- package/dist/multiplicity-DIWHvysC.d.ts.map +1 -0
- package/dist/multishot/index.d.ts +3 -3
- package/dist/multishot/index.js +1 -1
- package/dist/openapi.json +1 -1
- package/dist/{opencode-sqlite-8r6WUfHc.js → opencode-sqlite-DJWAXLms.js} +2 -2
- package/dist/{opencode-sqlite-8r6WUfHc.js.map → opencode-sqlite-DJWAXLms.js.map} +1 -1
- package/dist/package-version-D7lQHt_-.js +34 -0
- package/dist/package-version-D7lQHt_-.js.map +1 -0
- package/dist/paired-arms-D-XRF_fy.js +1045 -0
- package/dist/paired-arms-D-XRF_fy.js.map +1 -0
- package/dist/paired-promotion-decision-CGzg0cI_.d.ts +251 -0
- package/dist/paired-promotion-decision-CGzg0cI_.d.ts.map +1 -0
- package/dist/paired-tests-BHIhYVdu.js +213 -0
- package/dist/paired-tests-BHIhYVdu.js.map +1 -0
- package/dist/pareto-BqNW3LJR.d.ts +117 -0
- package/dist/pareto-BqNW3LJR.d.ts.map +1 -0
- package/dist/pipelines/index.d.ts +3 -64
- package/dist/pipelines/index.d.ts.map +1 -1
- package/dist/pipelines/index.js +4 -284
- package/dist/pipelines/index.js.map +1 -1
- package/dist/power-and-mde-CHIrXJll.js +195 -0
- package/dist/power-and-mde-CHIrXJll.js.map +1 -0
- package/dist/{promotion-policy-CrLrmys8.js → power-preflight-DEw-uC7q.js} +4 -184
- package/dist/power-preflight-DEw-uC7q.js.map +1 -0
- package/dist/pre-registration-CZwSFQS4.d.ts +577 -0
- package/dist/pre-registration-CZwSFQS4.d.ts.map +1 -0
- package/dist/{prime-protocol-BfSalTfR.js → prime-protocol-6tZTVsWm.js} +72 -21
- package/dist/prime-protocol-6tZTVsWm.js.map +1 -0
- package/dist/produced-state-DU79a81m.js +586 -0
- package/dist/produced-state-DU79a81m.js.map +1 -0
- package/dist/profile-cell.d.ts +1 -1
- package/dist/profile-cell.js +1 -1
- package/dist/promotion-policy-D0nPhkSy.d.ts +134 -0
- package/dist/promotion-policy-D0nPhkSy.d.ts.map +1 -0
- package/dist/promotion-policy-xzA40Evo.js +186 -0
- package/dist/promotion-policy-xzA40Evo.js.map +1 -0
- package/dist/{skillopt-optimization-method-USDKhxSA.d.ts → provenance-CCdxgLDT.d.ts} +57 -478
- package/dist/provenance-CCdxgLDT.d.ts.map +1 -0
- package/dist/registry-oJeeI4-a.d.ts +178 -0
- package/dist/registry-oJeeI4-a.d.ts.map +1 -0
- package/dist/{release-report-DA2BCu5a.d.ts → release-confidence-BFRE5WSp.d.ts} +180 -112
- package/dist/release-confidence-BFRE5WSp.d.ts.map +1 -0
- package/dist/{release-report-BUYmoKo2.js → release-confidence-CxDuiAev.js} +125 -217
- package/dist/release-confidence-CxDuiAev.js.map +1 -0
- package/dist/reporting.d.ts +6 -5
- package/dist/reporting.js +7 -5
- package/dist/{researcher-BchpD55R.d.ts → researcher-DaN4GST-.d.ts} +7 -40
- package/dist/{researcher-BchpD55R.d.ts.map → researcher-DaN4GST-.d.ts.map} +1 -1
- package/dist/{reward-hacking-BDToousL.js → reward-hacking-DNgjilrV.js} +3 -3
- package/dist/reward-hacking-DNgjilrV.js.map +1 -0
- package/dist/{reward-hacking-BI0OMAlo.d.ts → reward-hacking-DSSTuI9r.d.ts} +5 -5
- package/dist/{reward-hacking-BI0OMAlo.d.ts.map → reward-hacking-DSSTuI9r.d.ts.map} +1 -1
- package/dist/rl.d.ts +7 -7
- package/dist/rl.js +11 -10
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +2 -2
- package/dist/rollout/index.js +4 -4
- package/dist/{rollout-DRrksrcV.js → rollout-BWtw0I_6.js} +3 -3
- package/dist/{rollout-DRrksrcV.js.map → rollout-BWtw0I_6.js.map} +1 -1
- package/dist/{rubric-predictive-validity-DvkjPYCe.d.ts → rubric-predictive-validity-BgxtKe4G.d.ts} +2 -2
- package/dist/{rubric-predictive-validity-DvkjPYCe.d.ts.map → rubric-predictive-validity-BgxtKe4G.d.ts.map} +1 -1
- package/dist/{rubric-predictive-validity-BRR632r1.js → rubric-predictive-validity-Cwwyd7ah.js} +2 -2
- package/dist/{rubric-predictive-validity-BRR632r1.js.map → rubric-predictive-validity-Cwwyd7ah.js.map} +1 -1
- package/dist/{run-record-BmSPWXJR.js → run-record-BvHPVS-i.js} +3 -3
- package/dist/{run-record-BmSPWXJR.js.map → run-record-BvHPVS-i.js.map} +1 -1
- package/dist/{run-record-CF4Dwpxr.d.ts → run-record-CKiihE6f.d.ts} +4 -4
- package/dist/{run-record-CF4Dwpxr.d.ts.map → run-record-CKiihE6f.d.ts.map} +1 -1
- package/dist/{proposal-findings-2GIUo1et.js → run-score-lDzV0X8j.js} +31 -31
- package/dist/run-score-lDzV0X8j.js.map +1 -0
- package/dist/sandbox-harness-BlSOu4LX.d.ts +70 -0
- package/dist/sandbox-harness-BlSOu4LX.d.ts.map +1 -0
- package/dist/{schema-Cef2cFmb2.d.ts → schema-Cef2cFmb.d.ts} +1 -1
- package/dist/schema-Cef2cFmb.d.ts.map +1 -0
- package/dist/semantic-concept-judge-BiJxScqe.js +406 -0
- package/dist/semantic-concept-judge-BiJxScqe.js.map +1 -0
- package/dist/{sequential-D-BLJBKU.js → sequential-C458DXNf.js} +4 -3
- package/dist/{sequential-D-BLJBKU.js.map → sequential-C458DXNf.js.map} +1 -1
- package/dist/sequential-eprocess-CbUt2htw.js +83 -0
- package/dist/sequential-eprocess-CbUt2htw.js.map +1 -0
- package/dist/series-convergence-D1cL1f-4.d.ts +129 -0
- package/dist/series-convergence-D1cL1f-4.d.ts.map +1 -0
- package/dist/{server-iu0ede49.js → server-ulsOdrTI.js} +5 -21
- package/dist/server-ulsOdrTI.js.map +1 -0
- package/dist/skillopt-optimization-method-C5cotF4E.d.ts +427 -0
- package/dist/skillopt-optimization-method-C5cotF4E.d.ts.map +1 -0
- package/dist/skillopt-optimization-method-jjdnc3YK.js +1999 -0
- package/dist/skillopt-optimization-method-jjdnc3YK.js.map +1 -0
- package/dist/{statistical-heldout-TQ-4CYiN.d.ts → statistical-heldout-DoFF5KJX.d.ts} +5 -278
- package/dist/statistical-heldout-DoFF5KJX.d.ts.map +1 -0
- package/dist/{store-otlp-Dw8PPIlL.js → store-otlp-CsptLYpN.js} +3 -3
- package/dist/{store-otlp-Dw8PPIlL.js.map → store-otlp-CsptLYpN.js.map} +1 -1
- package/dist/store-tool-spans-Cq9mFd-q.js +667 -0
- package/dist/store-tool-spans-Cq9mFd-q.js.map +1 -0
- package/dist/store-tool-spans-DbDLOBOb.d.ts +342 -0
- package/dist/store-tool-spans-DbDLOBOb.d.ts.map +1 -0
- package/dist/student-t-CvBq2mve.js +38 -0
- package/dist/student-t-CvBq2mve.js.map +1 -0
- package/dist/{summary-report-Lf-5I7xh.js → summary-report-Blysd6Z2.js} +6 -3
- package/dist/{summary-report-Lf-5I7xh.js.map → summary-report-Blysd6Z2.js.map} +1 -1
- package/dist/{summary-report-BNR7DWTj.d.ts → summary-report-CaL-Hnxt.d.ts} +5 -5
- package/dist/{summary-report-BNR7DWTj.d.ts.map → summary-report-CaL-Hnxt.d.ts.map} +1 -1
- package/dist/supervisor-run/index.d.ts +391 -3
- package/dist/supervisor-run/index.d.ts.map +1 -0
- package/dist/supervisor-run/index.js +1689 -2
- package/dist/{supervisor-run-D_sokXcO.js.map → supervisor-run/index.js.map} +1 -1
- package/dist/{extract-usage-CdZdoj1s.js → task-failure-attributes-CpQ4y5RD.js} +5 -157
- package/dist/task-failure-attributes-CpQ4y5RD.js.map +1 -0
- package/dist/{tool-groups-CZPGGlHf.d.ts → tool-groups-Cteb03Ps.d.ts} +3 -3
- package/dist/tool-groups-Cteb03Ps.d.ts.map +1 -0
- package/dist/tool-waste-BDdBZG1F.js +803 -0
- package/dist/tool-waste-BDdBZG1F.js.map +1 -0
- package/dist/tool-waste-DjRDEsuI.d.ts +128 -0
- package/dist/tool-waste-DjRDEsuI.d.ts.map +1 -0
- package/dist/trace-repair/index.d.ts +14 -5
- package/dist/trace-repair/index.d.ts.map +1 -1
- package/dist/trace-repair/index.js +35 -7
- package/dist/trace-repair/index.js.map +1 -1
- package/dist/traces.d.ts +406 -7
- package/dist/traces.d.ts.map +1 -0
- package/dist/traces.js +1011 -10
- package/dist/traces.js.map +1 -0
- package/dist/trajectory-replay/index.d.ts +16 -3
- package/dist/trajectory-replay/index.d.ts.map +1 -1
- package/dist/trajectory-replay/index.js +52 -5
- package/dist/trajectory-replay/index.js.map +1 -1
- package/dist/types-BEPZc6eo.d.ts +93 -0
- package/dist/types-BEPZc6eo.d.ts.map +1 -0
- package/dist/{usage-receipt-t7vAzCRQ.js → types-BI4fT3HN.js} +67 -67
- package/dist/types-BI4fT3HN.js.map +1 -0
- package/dist/{types-BnjdJ70P.d.ts → types-BZ59Ahr8.d.ts} +5 -5
- package/dist/{types-BnjdJ70P.d.ts.map → types-BZ59Ahr8.d.ts.map} +1 -1
- package/dist/{types-D216SgwM.d.ts → types-Cx3YUh2r.d.ts} +4 -241
- package/dist/types-Cx3YUh2r.d.ts.map +1 -0
- package/dist/{types-BvDKaULh.d.ts → types-DLQx4mKU.d.ts} +4 -4
- package/dist/{types-BvDKaULh.d.ts.map → types-DLQx4mKU.d.ts.map} +1 -1
- package/dist/{types-D4mog56g.d.ts → types-yLK8gXE9.d.ts} +2 -2
- package/dist/{types-D4mog56g.d.ts.map → types-yLK8gXE9.d.ts.map} +1 -1
- package/dist/verdict-BndeTAh_.js +61 -0
- package/dist/verdict-BndeTAh_.js.map +1 -0
- package/dist/verdict-E4eRNf7-.d.ts +392 -0
- package/dist/verdict-E4eRNf7-.d.ts.map +1 -0
- package/dist/{verdict-cache-BCcOh0kF.js → verdict-cache-mZf5FEiY.js} +3 -55
- package/dist/{verdict-cache-BCcOh0kF.js.map → verdict-cache-mZf5FEiY.js.map} +1 -1
- package/dist/wire/index.d.ts +3 -3
- package/dist/wire/index.d.ts.map +1 -1
- package/dist/wire/index.js +1 -1
- package/docs/charter.md +3 -3
- package/docs/control-runtime.md +3 -42
- package/docs/experiment.md +0 -1
- package/docs/feature-guide.md +2 -2
- package/docs/trace-repair-grader.md +1 -0
- package/docs/trajectory-replay.md +1 -0
- package/docs/verdicts.md +43 -0
- package/docs/verification-strategies.md +3 -2
- package/package.json +6 -11
- package/dist/agent-profile-DPi7IZg7.d.ts.map +0 -1
- package/dist/analyze-runs-C30yljDJ.js.map +0 -1
- package/dist/baseline-CavEbRyH.d.ts +0 -136
- package/dist/baseline-CavEbRyH.d.ts.map +0 -1
- package/dist/benchmark-command-BteMFN62.js.map +0 -1
- package/dist/benchmarks-Dzs8CKb1.js +0 -755
- package/dist/benchmarks-Dzs8CKb1.js.map +0 -1
- package/dist/campaign-C2TTzQII.js.map +0 -1
- package/dist/completion-verifier-DJA5BhPb.d.ts +0 -414
- package/dist/completion-verifier-DJA5BhPb.d.ts.map +0 -1
- package/dist/control.d.ts +0 -3
- package/dist/control.js +0 -2
- package/dist/dataset-C8xaLXdY.d.ts.map +0 -1
- package/dist/default-registry-Bf8Woqmq.d.ts.map +0 -1
- package/dist/default-registry-BmktKy8r.js.map +0 -1
- package/dist/engine-3hL-XqwJ.d.ts.map +0 -1
- package/dist/experiment-tracker-CnRICnMl.js +0 -500
- package/dist/experiment-tracker-CnRICnMl.js.map +0 -1
- package/dist/experiment-tracker-IMntXr6J.d.ts.map +0 -1
- package/dist/extract-usage-CdZdoj1s.js.map +0 -1
- package/dist/failure-cluster-CqcvCcdR.d.ts.map +0 -1
- package/dist/feedback-trajectory-CacHpxsp.d.ts.map +0 -1
- package/dist/index-BZ3-y4YL.d.ts +0 -391
- package/dist/index-BZ3-y4YL.d.ts.map +0 -1
- package/dist/index-CQTZ-4XN.d.ts.map +0 -1
- package/dist/index-DPPGNJ_R.d.ts.map +0 -1
- package/dist/index-YE4KdKbO2.d.ts +0 -335
- package/dist/index-YE4KdKbO2.d.ts.map +0 -1
- package/dist/paired-arms-iZ08VFMN.js +0 -260
- package/dist/paired-arms-iZ08VFMN.js.map +0 -1
- package/dist/paired-promotion-decision-B6zJ3gYM.d.ts +0 -114
- package/dist/paired-promotion-decision-B6zJ3gYM.d.ts.map +0 -1
- package/dist/prime-protocol-BfSalTfR.js.map +0 -1
- package/dist/promotion-policy-Ckjhzg_4.d.ts +0 -289
- package/dist/promotion-policy-Ckjhzg_4.d.ts.map +0 -1
- package/dist/promotion-policy-CrLrmys8.js.map +0 -1
- package/dist/proposal-findings-2GIUo1et.js.map +0 -1
- package/dist/propose-review-control-dSNPjFUH.js +0 -1458
- package/dist/propose-review-control-dSNPjFUH.js.map +0 -1
- package/dist/release-report-BUYmoKo2.js.map +0 -1
- package/dist/release-report-DA2BCu5a.d.ts.map +0 -1
- package/dist/replay-CohS93nE.js +0 -1859
- package/dist/replay-CohS93nE.js.map +0 -1
- package/dist/replay-DbhZ4Ked.d.ts +0 -834
- package/dist/replay-DbhZ4Ked.d.ts.map +0 -1
- package/dist/reward-hacking-BDToousL.js.map +0 -1
- package/dist/run-evidence-Dhi3C81V.d.ts +0 -225
- package/dist/run-evidence-Dhi3C81V.d.ts.map +0 -1
- package/dist/schema-Cef2cFmb2.d.ts.map +0 -1
- package/dist/semantic-concept-judge-D1z-KepS.js +0 -767
- package/dist/semantic-concept-judge-D1z-KepS.js.map +0 -1
- package/dist/series-convergence-ofsqPWhs.d.ts +0 -35
- package/dist/series-convergence-ofsqPWhs.d.ts.map +0 -1
- package/dist/server-iu0ede49.js.map +0 -1
- package/dist/single-run-lock-DFWHEB09.js.map +0 -1
- package/dist/skill-usage-CO9OLRBx.d.ts.map +0 -1
- package/dist/skillopt-optimization-method-CQdVeM8k.js.map +0 -1
- package/dist/skillopt-optimization-method-USDKhxSA.d.ts.map +0 -1
- package/dist/statistical-heldout-TQ-4CYiN.d.ts.map +0 -1
- package/dist/statistics-ByxzSiOM.js +0 -2212
- package/dist/statistics-ByxzSiOM.js.map +0 -1
- package/dist/statistics-D6Uebe_4.d.ts +0 -968
- package/dist/statistics-D6Uebe_4.d.ts.map +0 -1
- package/dist/supervisor-run-D_sokXcO.js +0 -1690
- package/dist/test-graded-scenario-D1TaI2va.d.ts +0 -141
- package/dist/test-graded-scenario-D1TaI2va.d.ts.map +0 -1
- package/dist/test-graded-scenario-JHcKQNpq.js +0 -318
- package/dist/test-graded-scenario-JHcKQNpq.js.map +0 -1
- package/dist/tool-groups-CZPGGlHf.d.ts.map +0 -1
- package/dist/tool-use-metrics-DEGMKycK.js +0 -370
- package/dist/tool-use-metrics-DEGMKycK.js.map +0 -1
- package/dist/types-D216SgwM.d.ts.map +0 -1
- package/dist/usage-receipt-t7vAzCRQ.js.map +0 -1
- package/dist/verdict-DExhxfgR.d.ts +0 -201
- package/dist/verdict-DExhxfgR.d.ts.map +0 -1
package/dist/replay-CohS93nE.js
DELETED
|
@@ -1,1859 +0,0 @@
|
|
|
1
|
-
import { n as CaptureIntegrityError, s as ReplayError } from "./errors-D-LKuDhb.js";
|
|
2
|
-
import { r as hashJson, t as canonicalize } from "./pre-registration-DakwTRXk.js";
|
|
3
|
-
import { a as providerFromBaseUrl, i as defaultProviderRedactor } from "./raw-provider-sink-BQd7mzyT.js";
|
|
4
|
-
import { LLM_INPUT_TOKENS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, OPENINFERENCE_SPAN_KIND, TOOL_NAME, applyLlmSpanOtlpAttributes } from "./trace-attributes.js";
|
|
5
|
-
import { J as firstStringAttr, K as compareSpanTime, Q as spanEpochMillis, X as projectOtlpFlatLine, et as applyToolSpanOtlpAttributes, i as runTraceAnalyst, nt as isOtlpModelCall, rt as traceSpanKindToOpenInferenceKind, tt as classifyOtlpSpanRole } from "./kind-factory-B8-r8-y8.js";
|
|
6
|
-
import { r as modelHasSnapshot, s as validateRunRecord } from "./run-record-BmSPWXJR.js";
|
|
7
|
-
import { n as OtlpFileTraceStore, r as createOtlpBufferTraceStore } from "./store-otlp-Dw8PPIlL.js";
|
|
8
|
-
import { a as recordAggregateMeasurements, i as readTaskFailureLabels, o as summarizeExecutionMeasurements, r as extractUsageFromSse, s as summarizeTraceErrors, t as extractUsage } from "./extract-usage-CdZdoj1s.js";
|
|
9
|
-
import { readFileSync, readdirSync, statSync, writeFileSync } from "node:fs";
|
|
10
|
-
import { join } from "node:path";
|
|
11
|
-
import { randomUUID } from "node:crypto";
|
|
12
|
-
import { deriveHexId, isW3CSpanId, isW3CTraceId } from "@tangle-network/agent-trace-contract";
|
|
13
|
-
//#region src/trace-analyst/prompts.ts
|
|
14
|
-
/** General policy for recursive, evidence-backed trace analysis. */
|
|
15
|
-
const TRACE_ANALYST_ACTOR_DESCRIPTION = `Answer the question by inspecting the OTLP trace dataset with the available tools.
|
|
16
|
-
|
|
17
|
-
1. Call getDatasetOverview first. Use its real trace ids and dataset size to plan the investigation.
|
|
18
|
-
2. Narrow with queryTraces and countTraces before scanning large payloads.
|
|
19
|
-
3. For a small trace, use viewTrace. For a large trace, use searchTrace and then viewSpans or searchSpan.
|
|
20
|
-
4. Never invent a trace id, span id, tool result, error, frequency, or final outcome.
|
|
21
|
-
5. When a search reports has_more, refine the query before drawing a conclusion.
|
|
22
|
-
6. Use llm_query only over evidence already loaded. A recursive query cannot inspect traces itself.
|
|
23
|
-
7. Cite exact evidence URIs returned by the tools. Include a short exact excerpt when it supports the claim.
|
|
24
|
-
8. Return no finding when the available evidence cannot support one.
|
|
25
|
-
|
|
26
|
-
The prose answer must directly answer the question and state important uncertainty.
|
|
27
|
-
The findings array contains only actionable or decision-relevant claims supported by inspected evidence.`;
|
|
28
|
-
const TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION = "trace-analyst-research-v1-2026-07-30";
|
|
29
|
-
//#endregion
|
|
30
|
-
//#region src/trace-analyst/analyst.ts
|
|
31
|
-
/**
|
|
32
|
-
* Answer one question by recursively inspecting a trace store.
|
|
33
|
-
*
|
|
34
|
-
* The returned answer, cited findings, engine steps, call counts, and runtime
|
|
35
|
-
* identity are one audit record. A direct one-shot model call is not used.
|
|
36
|
-
*/
|
|
37
|
-
async function analyzeTraces(input, options) {
|
|
38
|
-
if (typeof input.question !== "string" || !input.question.trim()) throw new TypeError("analyzeTraces: input.question must be a non-empty string");
|
|
39
|
-
const id = input.id?.trim() || "trace-analysis";
|
|
40
|
-
const store = typeof options.source === "string" ? new OtlpFileTraceStore({ path: options.source }) : options.source;
|
|
41
|
-
if (store instanceof OtlpFileTraceStore) await store.ensureIndexed(options.signal ? { signal: options.signal } : void 0);
|
|
42
|
-
return runTraceAnalyst({
|
|
43
|
-
definition: {
|
|
44
|
-
id,
|
|
45
|
-
description: input.description?.trim() || "Answers a caller-defined question by recursively inspecting trace evidence.",
|
|
46
|
-
area: input.area?.trim() || "trace-analysis",
|
|
47
|
-
version: TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION,
|
|
48
|
-
question: input.question,
|
|
49
|
-
instructions: options.instructions ?? TRACE_ANALYST_ACTOR_DESCRIPTION,
|
|
50
|
-
toolGroup: options.toolGroup ?? "all",
|
|
51
|
-
limits: options.limits
|
|
52
|
-
},
|
|
53
|
-
engine: options.engine,
|
|
54
|
-
store,
|
|
55
|
-
context: {
|
|
56
|
-
runId: options.runId ?? id,
|
|
57
|
-
correlationId: randomUUID(),
|
|
58
|
-
budgetUsd: options.budgetUsd,
|
|
59
|
-
costLedger: options.costLedger,
|
|
60
|
-
costPhase: options.costPhase ?? "trace-analysis",
|
|
61
|
-
priorFindings: options.priorFindings,
|
|
62
|
-
upstreamFindings: options.upstreamFindings,
|
|
63
|
-
recordUsage: options.recordUsage,
|
|
64
|
-
tags: options.tags,
|
|
65
|
-
log: options.log,
|
|
66
|
-
signal: options.signal
|
|
67
|
-
}
|
|
68
|
-
});
|
|
69
|
-
}
|
|
70
|
-
//#endregion
|
|
71
|
-
//#region src/trace-analyst/hook.ts
|
|
72
|
-
const DEFAULT_QUESTION = "Summarise what happened in this run. Surface any failure modes, surprising findings, or evidence that the run's verdict is wrong.";
|
|
73
|
-
function traceAnalystOnRunComplete(opts) {
|
|
74
|
-
return async (ctx) => {
|
|
75
|
-
if (opts.shouldRun && !opts.shouldRun(ctx)) return;
|
|
76
|
-
const source = opts.analyze.source;
|
|
77
|
-
if (source === void 0) {
|
|
78
|
-
await ctx.store.appendEvent({
|
|
79
|
-
eventId: `analyst-skip-${ctx.runId}`,
|
|
80
|
-
runId: ctx.runId,
|
|
81
|
-
kind: "log",
|
|
82
|
-
timestamp: Date.now(),
|
|
83
|
-
payload: {
|
|
84
|
-
source: "trace_analyst_hook",
|
|
85
|
-
reason: "no source configured"
|
|
86
|
-
}
|
|
87
|
-
});
|
|
88
|
-
return;
|
|
89
|
-
}
|
|
90
|
-
const result = await analyzeTraces({ question: opts.question ?? DEFAULT_QUESTION }, {
|
|
91
|
-
...opts.analyze,
|
|
92
|
-
source
|
|
93
|
-
});
|
|
94
|
-
if (opts.save) await opts.save(result, ctx);
|
|
95
|
-
if (opts.gateOn && !opts.gateOn(result, ctx)) await ctx.store.appendEvent({
|
|
96
|
-
eventId: `analyst-gate-${ctx.runId}`,
|
|
97
|
-
runId: ctx.runId,
|
|
98
|
-
kind: "log",
|
|
99
|
-
timestamp: Date.now(),
|
|
100
|
-
payload: {
|
|
101
|
-
source: "trace_analyst_hook",
|
|
102
|
-
reason: "analyst_gate_failed",
|
|
103
|
-
findings: result.findings
|
|
104
|
-
}
|
|
105
|
-
});
|
|
106
|
-
};
|
|
107
|
-
}
|
|
108
|
-
//#endregion
|
|
109
|
-
//#region src/trace-analyst/insights.ts
|
|
110
|
-
const DOMAIN_STOP_WORDS = /* @__PURE__ */ new Set([
|
|
111
|
-
"and",
|
|
112
|
-
"advanced",
|
|
113
|
-
"app",
|
|
114
|
-
"build",
|
|
115
|
-
"create",
|
|
116
|
-
"easy",
|
|
117
|
-
"expert",
|
|
118
|
-
"extreme",
|
|
119
|
-
"for",
|
|
120
|
-
"from",
|
|
121
|
-
"hard",
|
|
122
|
-
"implementation",
|
|
123
|
-
"integrate",
|
|
124
|
-
"medium",
|
|
125
|
-
"project",
|
|
126
|
-
"task",
|
|
127
|
-
"the",
|
|
128
|
-
"this",
|
|
129
|
-
"with",
|
|
130
|
-
"workflow"
|
|
131
|
-
]);
|
|
132
|
-
function tokenizeDomainWords(value) {
|
|
133
|
-
return [...value.matchAll(/[A-Za-z][A-Za-z0-9.+#-]{2,}/g)].map((match) => match[0].toLowerCase()).filter((word) => !DOMAIN_STOP_WORDS.has(word));
|
|
134
|
-
}
|
|
135
|
-
function inferDomainKeywords(suite) {
|
|
136
|
-
const suiteWords = new Set(tokenizeDomainWords(`${suite.name} ${suite.collectionId ?? ""}`));
|
|
137
|
-
const source = [
|
|
138
|
-
suite.name,
|
|
139
|
-
suite.collectionId ?? "",
|
|
140
|
-
...suite.tasks.flatMap((task) => [
|
|
141
|
-
task.id,
|
|
142
|
-
task.name,
|
|
143
|
-
task.prompt ?? "",
|
|
144
|
-
task.difficulty ?? "",
|
|
145
|
-
...task.tags ?? [],
|
|
146
|
-
...task.gaps ?? []
|
|
147
|
-
])
|
|
148
|
-
].join(" ");
|
|
149
|
-
const counts = /* @__PURE__ */ new Map();
|
|
150
|
-
for (const word of tokenizeDomainWords(source)) counts.set(word, (counts.get(word) ?? 0) + 1);
|
|
151
|
-
return [...counts.entries()].filter(([word, count]) => count >= 2 || suiteWords.has(word)).sort((a, b) => b[1] - a[1] || a[0].localeCompare(b[0])).map(([word]) => word).slice(0, 18);
|
|
152
|
-
}
|
|
153
|
-
function domainEvidencePattern(keywords) {
|
|
154
|
-
const escaped = keywords.filter((keyword) => keyword.length >= 3).map((keyword) => keyword.replace(/[.*+?^${}()|[\]\\]/g, "\\$&"));
|
|
155
|
-
return escaped.length > 0 ? new RegExp(`(?<![A-Za-z0-9])(?:${escaped.join("|")})(?![A-Za-z0-9])`, "i") : /(?<![A-Za-z0-9])(?:sdk|api|css|dns|xml|provider|client|service|integration|webhook|transaction|auth|oauth|graphql|rest)(?![A-Za-z0-9])/i;
|
|
156
|
-
}
|
|
157
|
-
function describeTraceInsightScope(suite) {
|
|
158
|
-
const taskLabel = suite.tasks.length === 1 ? "1 implementation task" : `${suite.tasks.length} implementation tasks`;
|
|
159
|
-
const tags = /* @__PURE__ */ new Map();
|
|
160
|
-
for (const task of suite.tasks) for (const tag of task.tags ?? []) tags.set(tag, (tags.get(tag) ?? 0) + 1);
|
|
161
|
-
const topTags = [...tags.entries()].sort((a, b) => b[1] - a[1] || a[0].localeCompare(b[0])).slice(0, 8).map(([tag]) => tag);
|
|
162
|
-
if (topTags.length > 0) return `${taskLabel} across ${topTags.join(", ")}.`;
|
|
163
|
-
return `${taskLabel} across ${[...new Set(suite.tasks.map((task) => task.difficulty).filter((value) => Boolean(value)))].join(", ") || "the selected benchmark scope"}.`;
|
|
164
|
-
}
|
|
165
|
-
function planTraceInsightQuestions(input) {
|
|
166
|
-
const hasFailures = input.suite.tasks.some((task) => task.outcome && task.outcome !== "satisfied");
|
|
167
|
-
const hasMultipleShots = input.suite.tasks.some((task) => (task.gaps ?? []).some((gap) => /shot|review|retry|continue/i.test(gap)));
|
|
168
|
-
const questions = [
|
|
169
|
-
{
|
|
170
|
-
id: "execution-path",
|
|
171
|
-
question: "What did the worker actually do before the first meaningful implementation edit?",
|
|
172
|
-
why: "Separates grounded execution from polished but shallow output."
|
|
173
|
-
},
|
|
174
|
-
{
|
|
175
|
-
id: "research-grounding",
|
|
176
|
-
question: "Did the worker inspect docs, source, examples, or package references before committing to an implementation path?",
|
|
177
|
-
why: "Identifies whether failures came from weak retrieval, weak examples, or premature coding."
|
|
178
|
-
},
|
|
179
|
-
{
|
|
180
|
-
id: "domain-proof",
|
|
181
|
-
question: "Which tasks produced executable domain proof versus UI copy, placeholders, or inferred behavior?",
|
|
182
|
-
why: "Keeps product-quality claims tied to concrete evidence."
|
|
183
|
-
},
|
|
184
|
-
{
|
|
185
|
-
id: "root-cause",
|
|
186
|
-
question: "For each major failure cluster, is the likely root cause prompt/scaffold, docs/examples, SDK/API ergonomics, evaluator, runtime, or model behavior?",
|
|
187
|
-
why: "Turns trace observations into actionable ownership."
|
|
188
|
-
},
|
|
189
|
-
{
|
|
190
|
-
id: "evidence-quality",
|
|
191
|
-
question: "Which external-facing claims are directly supported by trace ids, span ids, verifier findings, reviewer notes, or generated code?",
|
|
192
|
-
why: "Prevents unsupported customer-report conclusions."
|
|
193
|
-
}
|
|
194
|
-
];
|
|
195
|
-
if (hasMultipleShots) questions.push({
|
|
196
|
-
id: "reviewer-lift",
|
|
197
|
-
question: "Where did reviewer feedback improve score, stall, or regress across shots?",
|
|
198
|
-
why: "Shows whether the driver loop is learning or merely repeating work."
|
|
199
|
-
});
|
|
200
|
-
if (hasFailures) questions.push({
|
|
201
|
-
id: "optimization-targets",
|
|
202
|
-
question: "Which prompt, evaluator, scaffold, or workflow changes should feed the next GEPA/autoresearch optimization run?",
|
|
203
|
-
why: "Connects benchmark evidence to the optimization loop."
|
|
204
|
-
});
|
|
205
|
-
return questions;
|
|
206
|
-
}
|
|
207
|
-
function buildTraceInsightContext(input) {
|
|
208
|
-
return {
|
|
209
|
-
suite: input.suite,
|
|
210
|
-
scope: describeTraceInsightScope(input.suite),
|
|
211
|
-
keywords: inferDomainKeywords(input.suite),
|
|
212
|
-
questions: planTraceInsightQuestions(input),
|
|
213
|
-
panel: defaultTraceInsightPanel(),
|
|
214
|
-
findings: input.findings ?? [],
|
|
215
|
-
agent: input.agent ?? null,
|
|
216
|
-
totals: input.totals ?? null
|
|
217
|
-
};
|
|
218
|
-
}
|
|
219
|
-
function scoreTraceInsightReadiness(context) {
|
|
220
|
-
const failedTasks = context.suite.tasks.filter((task) => task.outcome && task.outcome !== "satisfied");
|
|
221
|
-
const findingTaskIds = new Set(context.findings.flatMap((finding) => finding.taskIds));
|
|
222
|
-
const failedTasksWithFindings = failedTasks.filter((task) => findingTaskIds.has(task.id));
|
|
223
|
-
const tasksWithGaps = context.suite.tasks.filter((task) => (task.gaps ?? []).length > 0);
|
|
224
|
-
const gates = [
|
|
225
|
-
{
|
|
226
|
-
id: "domain-context",
|
|
227
|
-
label: "Domain context inferred",
|
|
228
|
-
passed: context.keywords.length > 0,
|
|
229
|
-
severity: "high",
|
|
230
|
-
detail: context.keywords.length > 0 ? `${context.keywords.length} domain terms inferred: ${context.keywords.slice(0, 8).join(", ")}` : "No domain terms were inferred from suite, tasks, prompts, tags, or gaps."
|
|
231
|
-
},
|
|
232
|
-
{
|
|
233
|
-
id: "panel-coverage",
|
|
234
|
-
label: "Analyst panel planned",
|
|
235
|
-
passed: context.panel.length >= 4 && context.questions.length >= 5,
|
|
236
|
-
severity: "high",
|
|
237
|
-
detail: `${context.panel.length} panel roles and ${context.questions.length} investigation questions planned.`
|
|
238
|
-
},
|
|
239
|
-
{
|
|
240
|
-
id: "failure-coverage",
|
|
241
|
-
label: "Failures mapped to findings",
|
|
242
|
-
passed: failedTasks.length === 0 || failedTasksWithFindings.length / failedTasks.length >= .5,
|
|
243
|
-
severity: "critical",
|
|
244
|
-
detail: failedTasks.length === 0 ? "No failed tasks in suite." : `${failedTasksWithFindings.length}/${failedTasks.length} failed tasks appear in finding clusters.`
|
|
245
|
-
},
|
|
246
|
-
{
|
|
247
|
-
id: "gap-evidence",
|
|
248
|
-
label: "Task gaps captured",
|
|
249
|
-
passed: failedTasks.length === 0 || tasksWithGaps.length / failedTasks.length >= .5,
|
|
250
|
-
severity: "medium",
|
|
251
|
-
detail: `${tasksWithGaps.length} tasks include explicit evaluator or analyst gaps.`
|
|
252
|
-
}
|
|
253
|
-
];
|
|
254
|
-
const penalty = gates.reduce((sum, gate) => {
|
|
255
|
-
if (gate.passed) return sum;
|
|
256
|
-
if (gate.severity === "critical") return sum + 35;
|
|
257
|
-
if (gate.severity === "high") return sum + 20;
|
|
258
|
-
if (gate.severity === "medium") return sum + 10;
|
|
259
|
-
return sum + 5;
|
|
260
|
-
}, 0);
|
|
261
|
-
const score = Math.max(0, Math.min(1, 1 - penalty / 100));
|
|
262
|
-
return {
|
|
263
|
-
score,
|
|
264
|
-
grade: score >= .9 ? "external-ready" : score >= .7 ? "internal-review" : "raw-analysis",
|
|
265
|
-
gates
|
|
266
|
-
};
|
|
267
|
-
}
|
|
268
|
-
function defaultTraceInsightPanel() {
|
|
269
|
-
return [
|
|
270
|
-
{
|
|
271
|
-
id: "trace-forensics",
|
|
272
|
-
name: "Trace Forensics",
|
|
273
|
-
responsibility: "Reconstruct what the worker did in order, including research, edits, reviewer interventions, verifier feedback, and stop reason."
|
|
274
|
-
},
|
|
275
|
-
{
|
|
276
|
-
id: "root-cause",
|
|
277
|
-
name: "Root Cause",
|
|
278
|
-
responsibility: "Map failures to prompt/scaffold, docs/examples, SDK/API/product ergonomics, evaluator, runtime, or model behavior."
|
|
279
|
-
},
|
|
280
|
-
{
|
|
281
|
-
id: "optimization",
|
|
282
|
-
name: "Optimization",
|
|
283
|
-
responsibility: "Identify prompt, reviewer, evaluator, scaffold, and GEPA/autoresearch changes that should be tested next."
|
|
284
|
-
},
|
|
285
|
-
{
|
|
286
|
-
id: "external-evidence",
|
|
287
|
-
name: "External Evidence",
|
|
288
|
-
responsibility: "Separate customer-safe claims from internal harness findings and reject conclusions without task, trace, span, code, reviewer, or verifier evidence."
|
|
289
|
-
}
|
|
290
|
-
];
|
|
291
|
-
}
|
|
292
|
-
function buildTraceInsightPrompt(input) {
|
|
293
|
-
const context = buildTraceInsightContext(input);
|
|
294
|
-
const maxRepresentativeTraces = input.maxRepresentativeTraces ?? 6;
|
|
295
|
-
return `Analyze this benchmark run and produce evidence-backed trace intelligence.
|
|
296
|
-
|
|
297
|
-
Audience:
|
|
298
|
-
- internal AI/product leadership
|
|
299
|
-
- possible customer-facing report for ${input.suite.name}
|
|
300
|
-
|
|
301
|
-
Investigation plan:
|
|
302
|
-
${context.questions.map((item, index) => `${index + 1}. ${item.question} (${item.why})`).join("\n")}
|
|
303
|
-
|
|
304
|
-
Analyst panel:
|
|
305
|
-
${context.panel.map((role) => `- ${role.name}: ${role.responsibility}`).join("\n")}
|
|
306
|
-
|
|
307
|
-
If the task branches are independent, use subagents for the panel roles above and aggregate their findings. Do not run a panel role unless its answer will change the final report.
|
|
308
|
-
|
|
309
|
-
Required output:
|
|
310
|
-
1. Executive verdict: what this run proves and does not prove.
|
|
311
|
-
2. The investigation questions you answered and the evidence used.
|
|
312
|
-
3. Failure taxonomy: agent prompting, evaluator/harness, docs/examples, SDK/API/product integration, infra.
|
|
313
|
-
4. Evidence-backed examples with trace ids/task ids and concrete verifier findings.
|
|
314
|
-
5. Highest-ROI fixes for the benchmark harness, prompt/GEPA optimization, and customer-facing product/docs surface.
|
|
315
|
-
6. What is safe for an external report versus what must stay internal.
|
|
316
|
-
7. One rerun plan that would validate lift after optimization.
|
|
317
|
-
|
|
318
|
-
Budget:
|
|
319
|
-
- Inspect the dataset overview, the failure summary, and at most ${maxRepresentativeTraces} representative traces.
|
|
320
|
-
- Prefer traces named in the failure summary over broad exploration.
|
|
321
|
-
- Do not do exhaustive trace sweeps.
|
|
322
|
-
- Return the final report as soon as the taxonomy and examples are supported.
|
|
323
|
-
|
|
324
|
-
Run summary:
|
|
325
|
-
${JSON.stringify({
|
|
326
|
-
suite: input.suite.name,
|
|
327
|
-
scope: context.scope,
|
|
328
|
-
inferredKeywords: context.keywords,
|
|
329
|
-
agent: context.agent,
|
|
330
|
-
totals: context.totals,
|
|
331
|
-
findings: context.findings.map((finding) => ({
|
|
332
|
-
kind: finding.kind,
|
|
333
|
-
severity: finding.severity,
|
|
334
|
-
taskCount: finding.taskIds.length,
|
|
335
|
-
proposedFixClass: finding.proposedFixClass
|
|
336
|
-
})),
|
|
337
|
-
failures: input.suite.tasks.filter((task) => task.outcome && task.outcome !== "satisfied").map((task) => ({
|
|
338
|
-
task: task.id,
|
|
339
|
-
difficulty: task.difficulty,
|
|
340
|
-
outcome: task.outcome,
|
|
341
|
-
score: task.score,
|
|
342
|
-
gaps: task.gaps ?? []
|
|
343
|
-
}))
|
|
344
|
-
}, null, 2)}
|
|
345
|
-
|
|
346
|
-
Use the trace tools. Do not invent facts. Cite task ids. Separate customer-facing claims from internal harness/model findings.`;
|
|
347
|
-
}
|
|
348
|
-
//#endregion
|
|
349
|
-
//#region src/trace/otlp-flat.ts
|
|
350
|
-
function createOtlpFlatLine(input) {
|
|
351
|
-
return {
|
|
352
|
-
trace_id: input.traceId,
|
|
353
|
-
span_id: input.spanId,
|
|
354
|
-
parent_span_id: input.parentSpanId,
|
|
355
|
-
name: input.name,
|
|
356
|
-
kind: input.kind,
|
|
357
|
-
start_time: input.startTime,
|
|
358
|
-
end_time: input.endTime,
|
|
359
|
-
status: {
|
|
360
|
-
code: input.statusCode,
|
|
361
|
-
...input.statusMessage !== void 0 ? { message: input.statusMessage } : {}
|
|
362
|
-
},
|
|
363
|
-
resource: input.resource,
|
|
364
|
-
attributes: input.attributes,
|
|
365
|
-
...input.events && input.events.length > 0 ? { events: input.events } : {}
|
|
366
|
-
};
|
|
367
|
-
}
|
|
368
|
-
/** Map the canonical trace status while letting each caller choose its legacy default. */
|
|
369
|
-
function spanStatusToOtlp(status, error, defaultCode) {
|
|
370
|
-
if (status === "error" || error) return "STATUS_CODE_ERROR";
|
|
371
|
-
if (status === "ok") return "STATUS_CODE_OK";
|
|
372
|
-
return defaultCode;
|
|
373
|
-
}
|
|
374
|
-
/** Convert epoch milliseconds, returning `undefined` for invalid or out-of-range values. */
|
|
375
|
-
function epochMillisToIso(value) {
|
|
376
|
-
if (!Number.isFinite(value)) return void 0;
|
|
377
|
-
try {
|
|
378
|
-
return new Date(value).toISOString();
|
|
379
|
-
} catch {
|
|
380
|
-
return;
|
|
381
|
-
}
|
|
382
|
-
}
|
|
383
|
-
//#endregion
|
|
384
|
-
//#region src/trace-analyst/otlp-flatten.ts
|
|
385
|
-
const DEFAULT_KIND_MAP = {
|
|
386
|
-
0: "SPAN_KIND_UNSPECIFIED",
|
|
387
|
-
1: "SPAN_KIND_INTERNAL",
|
|
388
|
-
2: "SPAN_KIND_SERVER",
|
|
389
|
-
3: "SPAN_KIND_CLIENT",
|
|
390
|
-
4: "SPAN_KIND_PRODUCER",
|
|
391
|
-
5: "SPAN_KIND_CONSUMER"
|
|
392
|
-
};
|
|
393
|
-
const STATUS_MAP = {
|
|
394
|
-
0: "STATUS_CODE_UNSET",
|
|
395
|
-
1: "STATUS_CODE_OK",
|
|
396
|
-
2: "STATUS_CODE_ERROR"
|
|
397
|
-
};
|
|
398
|
-
/** Unwrap an OTLP attribute-value union to a scalar. */
|
|
399
|
-
function attrValue(v) {
|
|
400
|
-
if (v.stringValue !== void 0) return v.stringValue;
|
|
401
|
-
if (v.intValue !== void 0) return Number(v.intValue);
|
|
402
|
-
if (v.doubleValue !== void 0) return v.doubleValue;
|
|
403
|
-
if (v.boolValue !== void 0) return v.boolValue;
|
|
404
|
-
return "";
|
|
405
|
-
}
|
|
406
|
-
function attrsToRecord(attrs) {
|
|
407
|
-
const out = {};
|
|
408
|
-
for (const a of attrs) out[a.key] = attrValue(a.value);
|
|
409
|
-
return out;
|
|
410
|
-
}
|
|
411
|
-
function nanoToIso(nano) {
|
|
412
|
-
const ms = Number(nano) / 1e6;
|
|
413
|
-
return Number.isFinite(ms) ? new Date(ms).toISOString() : (/* @__PURE__ */ new Date(0)).toISOString();
|
|
414
|
-
}
|
|
415
|
-
/** Mirror selected attributes into the OpenInference vocabulary in place. */
|
|
416
|
-
function applyOpenInference(attrs) {
|
|
417
|
-
if ("llm.model" in attrs && !("llm.model_name" in attrs)) attrs[LLM_MODEL_NAME] = attrs["llm.model"];
|
|
418
|
-
if ("llm.input_tokens" in attrs && !("llm.token_count.prompt" in attrs)) attrs[LLM_INPUT_TOKENS] = attrs["llm.input_tokens"];
|
|
419
|
-
if ("inference.llm.input_tokens" in attrs && !("llm.token_count.prompt" in attrs)) attrs[LLM_INPUT_TOKENS] = attrs["inference.llm.input_tokens"];
|
|
420
|
-
if ("llm.output_tokens" in attrs && !("llm.token_count.completion" in attrs)) attrs[LLM_OUTPUT_TOKENS] = attrs["llm.output_tokens"];
|
|
421
|
-
if ("inference.llm.output_tokens" in attrs && !("llm.token_count.completion" in attrs)) attrs[LLM_OUTPUT_TOKENS] = attrs["inference.llm.output_tokens"];
|
|
422
|
-
if ("tool.name" in attrs && !("inference.tool.name" in attrs)) attrs["inference.tool.name"] = attrs[TOOL_NAME];
|
|
423
|
-
if ("span.kind" in attrs && !("openinference.span.kind" in attrs)) attrs[OPENINFERENCE_SPAN_KIND] = String(attrs["span.kind"]).toUpperCase();
|
|
424
|
-
}
|
|
425
|
-
function flattenOtlpExportToNdjson(otlpExport, opts = {}) {
|
|
426
|
-
const vocab = opts.attributeVocabulary ?? "openinference";
|
|
427
|
-
const kindMap = {
|
|
428
|
-
...DEFAULT_KIND_MAP,
|
|
429
|
-
...opts.kindMap
|
|
430
|
-
};
|
|
431
|
-
const lines = [];
|
|
432
|
-
for (const rs of otlpExport.resourceSpans ?? []) {
|
|
433
|
-
const resource = { attributes: attrsToRecord(rs.resource?.attributes ?? []) };
|
|
434
|
-
for (const scope of rs.scopeSpans ?? []) for (const span of scope.spans ?? []) {
|
|
435
|
-
const attributes = attrsToRecord(span.attributes ?? []);
|
|
436
|
-
if (vocab === "openinference") applyOpenInference(attributes);
|
|
437
|
-
const line = createOtlpFlatLine({
|
|
438
|
-
traceId: span.traceId,
|
|
439
|
-
spanId: span.spanId,
|
|
440
|
-
parentSpanId: span.parentSpanId ?? null,
|
|
441
|
-
name: span.name,
|
|
442
|
-
kind: kindMap[span.kind] ?? "SPAN_KIND_UNSPECIFIED",
|
|
443
|
-
startTime: nanoToIso(span.startTimeUnixNano),
|
|
444
|
-
endTime: nanoToIso(span.endTimeUnixNano),
|
|
445
|
-
statusCode: STATUS_MAP[span.status?.code ?? 0] ?? "STATUS_CODE_UNSET",
|
|
446
|
-
statusMessage: span.status?.message,
|
|
447
|
-
resource,
|
|
448
|
-
attributes,
|
|
449
|
-
events: span.events?.map((e) => ({
|
|
450
|
-
name: e.name,
|
|
451
|
-
timeUnixNano: e.timeUnixNano,
|
|
452
|
-
...e.attributes ? { attributes: attrsToRecord(e.attributes) } : {}
|
|
453
|
-
}))
|
|
454
|
-
});
|
|
455
|
-
lines.push(line);
|
|
456
|
-
}
|
|
457
|
-
}
|
|
458
|
-
return lines;
|
|
459
|
-
}
|
|
460
|
-
//#endregion
|
|
461
|
-
//#region src/trace-analyst/otlp-to-run-records.ts
|
|
462
|
-
/**
|
|
463
|
-
* `otlpToRunRecords` — fold an OTLP traces.jsonl (one OTLP span per line;
|
|
464
|
-
* the form AppWorld / HALO emit via their OpenInference OTLP exporter, the
|
|
465
|
-
* same shape `flattenOtlpExportToNdjson` produces) into validated
|
|
466
|
-
* `RunRecord[]` — one record per `trace_id` by default, or per caller-defined
|
|
467
|
-
* logical run when one task is fragmented across multiple OTLP traces.
|
|
468
|
-
*
|
|
469
|
-
* This is the offline ingestion primitive the AppWorld proposer bench and the
|
|
470
|
-
* hosted Intelligence product both stand on: traces in, paper-grade rows
|
|
471
|
-
* out, ready for `compareOptimizationMethods` / `analyzeRuns` / the promotion gate.
|
|
472
|
-
*
|
|
473
|
-
* Aggregation per trace:
|
|
474
|
-
* - tokenUsage: reconcile input, output, cache-read, and cache-write across
|
|
475
|
-
* nested model-call wrappers without double-counting parent aggregates.
|
|
476
|
-
* - costUsd: reconcile complete observed model-call cost when present; else priced via
|
|
477
|
-
* `opts.priceUsdPerToken` from the aggregated tokens; else `null` with a
|
|
478
|
-
* loud `raw.cost_unpriced = 1` marker.
|
|
479
|
-
* - task failure class and detail: read from process-root
|
|
480
|
-
* `tangle.task.failure_*` attributes; malformed or conflicting values throw.
|
|
481
|
-
* - terminalFailureReason: the failed root's normalized status message,
|
|
482
|
-
* when one unambiguous root supplies terminal failure evidence.
|
|
483
|
-
* - terminalOutcome: reduced from root-span status only. Child tool errors
|
|
484
|
-
* remain visible in `error_span_count` and `execution_error_count` without
|
|
485
|
-
* changing the run outcome. Root, guardrail, evaluator, propagated, and
|
|
486
|
-
* unknown errors retain separate counters.
|
|
487
|
-
* - model: the dominant LLM model in the trace (snapshot-padded to satisfy
|
|
488
|
-
* `validateRunRecord` when the trace's model is a bare alias).
|
|
489
|
-
* - outcome score: `opts.scoreForTrace` (AppWorld `world.evaluate()` →
|
|
490
|
-
* TGC/SGC) when supplied. Traces without an external task-quality signal
|
|
491
|
-
* remain unlabeled; execution errors never become a task score.
|
|
492
|
-
* - prompt / completion: carried into `raw` as token-count signals and,
|
|
493
|
-
* when the first/last LLM span exposes `input.value` / `output.value`,
|
|
494
|
-
* the verbatim text is preserved on the optional `promptText` /
|
|
495
|
-
* `completionText` of the returned `OtlpTraceRunRecord`.
|
|
496
|
-
*
|
|
497
|
-
* Fail-loud: an OTLP file with zero valid spans throws. A trace with no
|
|
498
|
-
* spans is impossible (a trace exists only because a span referenced it).
|
|
499
|
-
* `validateRunRecord` runs on every row — a malformed projection throws
|
|
500
|
-
* rather than silently producing a half-record.
|
|
501
|
-
*/
|
|
502
|
-
/**
|
|
503
|
-
* Parse + aggregate an OTLP traces.jsonl string into validated
|
|
504
|
-
* `RunRecord[]` (one per trace). Use {@link otlpToTraceRunRecords} when you
|
|
505
|
-
* also want the verbatim prompt/completion text alongside each record.
|
|
506
|
-
*/
|
|
507
|
-
function otlpToRunRecords(otlpJsonl, opts) {
|
|
508
|
-
return otlpToTraceRunRecords(otlpJsonl, opts).map((r) => r.record);
|
|
509
|
-
}
|
|
510
|
-
/**
|
|
511
|
-
* Aggregate already-parsed OTLP flat rows without serializing them back to
|
|
512
|
-
* JSONL. This is the in-memory counterpart to {@link otlpToRunRecords}; both
|
|
513
|
-
* paths share projection, reconciliation, validation, and ordering.
|
|
514
|
-
*/
|
|
515
|
-
function otlpRowsToRunRecords(rows, opts) {
|
|
516
|
-
return otlpRowsToTraceRunRecords(rows, opts).map((row) => row.record);
|
|
517
|
-
}
|
|
518
|
-
/** As {@link otlpToRunRecords} but returns the prompt/completion text too. */
|
|
519
|
-
function otlpToTraceRunRecords(otlpJsonl, opts) {
|
|
520
|
-
return traceRunRecordsFromSpans(groupSpansByLogicalRun(groupJsonlSpansByTrace(otlpJsonl), opts.logicalRunIdForTrace), opts);
|
|
521
|
-
}
|
|
522
|
-
/** Parsed-row counterpart to {@link otlpToTraceRunRecords}. */
|
|
523
|
-
function otlpRowsToTraceRunRecords(rows, opts) {
|
|
524
|
-
return traceRunRecordsFromSpans(groupSpansByLogicalRun(groupRowsByTrace(rows), opts.logicalRunIdForTrace), opts);
|
|
525
|
-
}
|
|
526
|
-
function traceRunRecordsFromSpans(byTrace, opts) {
|
|
527
|
-
const splitTag = opts.splitTag ?? "holdout";
|
|
528
|
-
const commitSha = opts.commitSha ?? "unknown";
|
|
529
|
-
const promptHash = opts.promptHash ?? "unknown";
|
|
530
|
-
const configHash = opts.configHash ?? "unknown";
|
|
531
|
-
const seed = opts.seed ?? 0;
|
|
532
|
-
const fallbackModel = opts.fallbackModel ?? "unknown@otlp";
|
|
533
|
-
if (byTrace.size === 0) throw new Error("otlpToRunRecords: OTLP input produced zero valid spans — every row was empty, malformed, or missing trace_id/span_id");
|
|
534
|
-
const traceIds = [...byTrace.keys()].sort();
|
|
535
|
-
const out = [];
|
|
536
|
-
for (const traceId of traceIds) {
|
|
537
|
-
const spans = byTrace.get(traceId);
|
|
538
|
-
const agg = aggregateTrace(traceId, spans, fallbackModel);
|
|
539
|
-
const score = resolveScore(opts, traceId, agg);
|
|
540
|
-
const { costUsd, costProvenance } = resolveCost(opts, agg);
|
|
541
|
-
const raw = {
|
|
542
|
-
source_trace_count: agg.sourceTraceCount,
|
|
543
|
-
span_count: agg.spanCount,
|
|
544
|
-
llm_span_count: agg.llmSpanCount,
|
|
545
|
-
tool_span_count: agg.toolSpanCount,
|
|
546
|
-
agent_span_count: agg.agentSpanCount,
|
|
547
|
-
error_span_count: agg.errorSpanCount,
|
|
548
|
-
execution_error_count: agg.executionErrorCount,
|
|
549
|
-
process_error_count: agg.processErrorCount,
|
|
550
|
-
guardrail_error_count: agg.guardrailErrorCount,
|
|
551
|
-
judge_error_count: agg.judgeErrorCount,
|
|
552
|
-
propagated_error_count: agg.propagatedErrorCount,
|
|
553
|
-
unclassified_error_count: agg.unclassifiedErrorCount,
|
|
554
|
-
prompt_tokens: agg.tokenUsage.input,
|
|
555
|
-
completion_tokens: agg.tokenUsage.output
|
|
556
|
-
};
|
|
557
|
-
if (agg.tokenUsage.reasoning !== void 0) raw.reasoning_tokens = agg.tokenUsage.reasoning;
|
|
558
|
-
if (agg.tokenUsage.cached !== void 0) raw.cached_tokens = agg.tokenUsage.cached;
|
|
559
|
-
if (agg.tokenUsage.cacheWrite !== void 0) raw.cache_write_tokens = agg.tokenUsage.cacheWrite;
|
|
560
|
-
if (agg.costMeasurement.value !== void 0 && !agg.costMeasurement.complete) raw.partial_observed_cost_usd = agg.costMeasurement.value;
|
|
561
|
-
recordAggregateMeasurements(raw, agg.aggregateMeasurement);
|
|
562
|
-
if (costProvenance.kind === "uncaptured") raw.cost_unpriced = 1;
|
|
563
|
-
const outcome = { raw };
|
|
564
|
-
if (score !== void 0) if (splitTag === "holdout") outcome.holdoutScore = score;
|
|
565
|
-
else outcome.searchScore = score;
|
|
566
|
-
const { promptText, completionText } = extractPromptCompletion(spans, agg.callSpanIds);
|
|
567
|
-
const judgeMetadata = opts.judgeMetadataForTrace?.(traceId);
|
|
568
|
-
const taskFailure = readTaskFailureLabels(spans.filter((span) => span.parent_span_id === null && isTerminalRootCandidate(span)), `otlpToRunRecords: run '${traceId}'`);
|
|
569
|
-
const record = validateRunRecord({
|
|
570
|
-
runId: `otlp:${opts.experimentId}:${opts.candidateId}:${traceId}`,
|
|
571
|
-
experimentId: opts.experimentId,
|
|
572
|
-
candidateId: opts.candidateId,
|
|
573
|
-
seed,
|
|
574
|
-
model: ensureSnapshot(agg.model, fallbackModel),
|
|
575
|
-
promptHash,
|
|
576
|
-
configHash,
|
|
577
|
-
commitSha,
|
|
578
|
-
wallMs: agg.wallMs,
|
|
579
|
-
costUsd,
|
|
580
|
-
costProvenance,
|
|
581
|
-
tokenUsage: agg.tokenUsage,
|
|
582
|
-
terminalOutcome: agg.terminalOutcome,
|
|
583
|
-
...agg.terminalFailureMessage ? { terminalFailureReason: agg.terminalFailureMessage } : {},
|
|
584
|
-
...judgeMetadata ? { judgeMetadata } : {},
|
|
585
|
-
outcome,
|
|
586
|
-
...taskFailure,
|
|
587
|
-
splitTag,
|
|
588
|
-
scenarioId: traceId
|
|
589
|
-
});
|
|
590
|
-
out.push({
|
|
591
|
-
record,
|
|
592
|
-
...promptText !== void 0 ? { promptText } : {},
|
|
593
|
-
...completionText !== void 0 ? { completionText } : {}
|
|
594
|
-
});
|
|
595
|
-
}
|
|
596
|
-
return out;
|
|
597
|
-
}
|
|
598
|
-
function* yieldJsonlRows(otlpJsonl) {
|
|
599
|
-
for (const line of otlpJsonl.split("\n")) {
|
|
600
|
-
const trimmed = line.trim();
|
|
601
|
-
if (trimmed.length === 0) continue;
|
|
602
|
-
let parsed;
|
|
603
|
-
try {
|
|
604
|
-
parsed = JSON.parse(trimmed);
|
|
605
|
-
} catch {
|
|
606
|
-
continue;
|
|
607
|
-
}
|
|
608
|
-
if (parsed && typeof parsed === "object") yield parsed;
|
|
609
|
-
}
|
|
610
|
-
}
|
|
611
|
-
function groupJsonlSpansByTrace(otlpJsonl) {
|
|
612
|
-
return groupRowsByTrace(yieldJsonlRows(otlpJsonl));
|
|
613
|
-
}
|
|
614
|
-
function groupRowsByTrace(rows) {
|
|
615
|
-
const byTrace = /* @__PURE__ */ new Map();
|
|
616
|
-
for (const row of rows) {
|
|
617
|
-
if (!row || typeof row !== "object") continue;
|
|
618
|
-
const span = projectOtlpFlatLine(row);
|
|
619
|
-
if (!span) continue;
|
|
620
|
-
const arr = byTrace.get(span.trace_id);
|
|
621
|
-
if (arr) arr.push(span);
|
|
622
|
-
else byTrace.set(span.trace_id, [span]);
|
|
623
|
-
}
|
|
624
|
-
return byTrace;
|
|
625
|
-
}
|
|
626
|
-
function groupSpansByLogicalRun(byTrace, logicalRunIdForTrace) {
|
|
627
|
-
if (!logicalRunIdForTrace) return byTrace;
|
|
628
|
-
const byRun = /* @__PURE__ */ new Map();
|
|
629
|
-
for (const [traceId, spans] of byTrace) {
|
|
630
|
-
const suppliedRunId = logicalRunIdForTrace(traceId);
|
|
631
|
-
if (typeof suppliedRunId !== "string" || suppliedRunId.trim().length === 0) throw new Error(`otlpToRunRecords: logicalRunIdForTrace('${traceId}') returned an empty run id`);
|
|
632
|
-
const runId = suppliedRunId.trim();
|
|
633
|
-
const target = byRun.get(runId) ?? [];
|
|
634
|
-
for (const span of spans) target.push({
|
|
635
|
-
...span,
|
|
636
|
-
span_id: qualifySpanId(traceId, span.span_id),
|
|
637
|
-
parent_span_id: span.parent_span_id ? qualifySpanId(traceId, span.parent_span_id) : null
|
|
638
|
-
});
|
|
639
|
-
byRun.set(runId, target);
|
|
640
|
-
}
|
|
641
|
-
return byRun;
|
|
642
|
-
}
|
|
643
|
-
function qualifySpanId(traceId, spanId) {
|
|
644
|
-
const prefix = `${traceId}:`;
|
|
645
|
-
return spanId.startsWith(prefix) ? spanId : `${prefix}${spanId}`;
|
|
646
|
-
}
|
|
647
|
-
function aggregateTrace(traceId, spans, fallbackModel) {
|
|
648
|
-
const ordered = [...spans].sort((a, b) => compareSpanTime(a.start_time, b.start_time) || a.span_id.localeCompare(b.span_id));
|
|
649
|
-
const measurements = summarizeExecutionMeasurements(ordered.map((span) => ({
|
|
650
|
-
id: span.span_id,
|
|
651
|
-
...span.parent_span_id ? { parentId: span.parent_span_id } : {},
|
|
652
|
-
attributes: span.attributes,
|
|
653
|
-
modelCall: isOtlpModelCall({
|
|
654
|
-
kind: span.kind,
|
|
655
|
-
name: span.name,
|
|
656
|
-
attributes: span.attributes
|
|
657
|
-
}),
|
|
658
|
-
aggregate: span.kind !== "LLM" && span.kind !== "UNKNOWN"
|
|
659
|
-
})));
|
|
660
|
-
let toolSpanCount = 0;
|
|
661
|
-
let agentSpanCount = 0;
|
|
662
|
-
let firstErrorMessage;
|
|
663
|
-
const modelVotes = /* @__PURE__ */ new Map();
|
|
664
|
-
let earliest = ordered[0]?.start_time ?? "";
|
|
665
|
-
let latest = ordered[0]?.end_time ?? "";
|
|
666
|
-
for (const s of ordered) {
|
|
667
|
-
if (s.start_time && (!earliest || compareSpanTime(s.start_time, earliest) < 0)) earliest = s.start_time;
|
|
668
|
-
if (s.end_time && (!latest || compareSpanTime(s.end_time, latest) > 0)) latest = s.end_time;
|
|
669
|
-
if (s.kind === "TOOL") toolSpanCount += 1;
|
|
670
|
-
else if (s.kind === "AGENT") agentSpanCount += 1;
|
|
671
|
-
if (s.status === "ERROR") {
|
|
672
|
-
if (firstErrorMessage === void 0) firstErrorMessage = (s.status_message ?? `${s.name} — STATUS_CODE_ERROR`).slice(0, 500);
|
|
673
|
-
}
|
|
674
|
-
}
|
|
675
|
-
const callSpanIds = new Set(measurements.callSpanIds);
|
|
676
|
-
for (const span of ordered) {
|
|
677
|
-
if (!callSpanIds.has(span.span_id)) continue;
|
|
678
|
-
const model = firstStringAttr(span.attributes, LLM_MODEL_ATTR_KEYS) ?? span.model_name;
|
|
679
|
-
if (model) modelVotes.set(model, (modelVotes.get(model) ?? 0) + 1);
|
|
680
|
-
}
|
|
681
|
-
const model = topVote(modelVotes) ?? firstModelAttr(ordered) ?? fallbackModel;
|
|
682
|
-
let wallMs = 0;
|
|
683
|
-
const a = spanEpochMillis(earliest);
|
|
684
|
-
const b = spanEpochMillis(latest);
|
|
685
|
-
if (a !== null && b !== null) wallMs = Math.max(0, b - a);
|
|
686
|
-
const sourceTraceIds = [...new Set(spans.map((span) => span.trace_id))].sort();
|
|
687
|
-
const terminal = terminalEvidenceFromRoots(ordered);
|
|
688
|
-
const errorSummary = summarizeTraceErrors(ordered.map((span) => ({
|
|
689
|
-
id: span.span_id,
|
|
690
|
-
...span.parent_span_id ? { parentId: span.parent_span_id } : {},
|
|
691
|
-
role: errorRoleForProjectedSpan(span),
|
|
692
|
-
error: span.status === "ERROR",
|
|
693
|
-
processRoot: span.parent_span_id === null && isTerminalRootCandidate(span)
|
|
694
|
-
})));
|
|
695
|
-
return {
|
|
696
|
-
traceId,
|
|
697
|
-
sourceTraceCount: sourceTraceIds.length,
|
|
698
|
-
sourceTraceIds,
|
|
699
|
-
spanCount: spans.length,
|
|
700
|
-
llmSpanCount: measurements.modelCallCount,
|
|
701
|
-
toolSpanCount,
|
|
702
|
-
agentSpanCount,
|
|
703
|
-
errorSpanCount: errorSummary.total,
|
|
704
|
-
executionErrorCount: errorSummary.execution,
|
|
705
|
-
processErrorCount: errorSummary.process,
|
|
706
|
-
guardrailErrorCount: errorSummary.guardrail,
|
|
707
|
-
judgeErrorCount: errorSummary.evaluation,
|
|
708
|
-
propagatedErrorCount: errorSummary.propagated,
|
|
709
|
-
unclassifiedErrorCount: errorSummary.unclassified,
|
|
710
|
-
tokenUsage: measurements.tokenUsage,
|
|
711
|
-
firstErrorMessage,
|
|
712
|
-
model,
|
|
713
|
-
startTime: earliest,
|
|
714
|
-
endTime: latest,
|
|
715
|
-
wallMs,
|
|
716
|
-
terminalOutcome: terminal.outcome,
|
|
717
|
-
...terminal.failureMessage ? { terminalFailureMessage: terminal.failureMessage } : {},
|
|
718
|
-
callSpanIds: measurements.callSpanIds,
|
|
719
|
-
costMeasurement: measurements.cost,
|
|
720
|
-
...measurements.aggregate ? { aggregateMeasurement: measurements.aggregate } : {}
|
|
721
|
-
};
|
|
722
|
-
}
|
|
723
|
-
function terminalEvidenceFromRoots(spans) {
|
|
724
|
-
const roots = spans.filter((span) => span.parent_span_id === null && isTerminalRootCandidate(span));
|
|
725
|
-
if (roots.length !== 1) return { outcome: "unknown" };
|
|
726
|
-
const root = roots[0];
|
|
727
|
-
if (root.status === "ERROR") return {
|
|
728
|
-
outcome: "failed",
|
|
729
|
-
failureMessage: (root.status_message ?? `${root.name} — STATUS_CODE_ERROR`).slice(0, 500)
|
|
730
|
-
};
|
|
731
|
-
if (root.status === "OK") return { outcome: "succeeded" };
|
|
732
|
-
return { outcome: "unknown" };
|
|
733
|
-
}
|
|
734
|
-
function isTerminalRootCandidate(span) {
|
|
735
|
-
const role = errorRoleForProjectedSpan(span);
|
|
736
|
-
return role !== "LLM" && role !== "TOOL" && role !== "EVALUATOR" && role !== "GUARDRAIL";
|
|
737
|
-
}
|
|
738
|
-
function errorRoleForProjectedSpan(span) {
|
|
739
|
-
return classifyOtlpSpanRole({
|
|
740
|
-
kind: span.kind,
|
|
741
|
-
name: span.name,
|
|
742
|
-
attributes: span.attributes
|
|
743
|
-
});
|
|
744
|
-
}
|
|
745
|
-
function resolveScore(opts, traceId, agg) {
|
|
746
|
-
const supplied = opts.scoreForTrace?.(traceId, agg);
|
|
747
|
-
if (supplied !== void 0) {
|
|
748
|
-
if (!Number.isFinite(supplied)) throw new Error(`otlpToRunRecords: scoreForTrace('${traceId}') returned non-finite ${supplied}`);
|
|
749
|
-
return supplied;
|
|
750
|
-
}
|
|
751
|
-
}
|
|
752
|
-
function resolveCost(opts, agg) {
|
|
753
|
-
const observedCost = agg.costMeasurement;
|
|
754
|
-
if (observedCost.complete && observedCost.value !== void 0) return {
|
|
755
|
-
costUsd: observedCost.value,
|
|
756
|
-
costProvenance: {
|
|
757
|
-
kind: "observed",
|
|
758
|
-
usd: observedCost.value
|
|
759
|
-
}
|
|
760
|
-
};
|
|
761
|
-
if (agg.aggregateMeasurement?.costUsd !== void 0) return {
|
|
762
|
-
costUsd: agg.aggregateMeasurement.costUsd,
|
|
763
|
-
costProvenance: {
|
|
764
|
-
kind: "observed",
|
|
765
|
-
usd: agg.aggregateMeasurement.costUsd
|
|
766
|
-
}
|
|
767
|
-
};
|
|
768
|
-
if (opts.priceUsdPerToken !== void 0) {
|
|
769
|
-
const costUsd = (agg.tokenUsage.input + agg.tokenUsage.output) * opts.priceUsdPerToken;
|
|
770
|
-
return {
|
|
771
|
-
costUsd,
|
|
772
|
-
costProvenance: {
|
|
773
|
-
kind: "estimated",
|
|
774
|
-
usd: costUsd
|
|
775
|
-
}
|
|
776
|
-
};
|
|
777
|
-
}
|
|
778
|
-
return {
|
|
779
|
-
costUsd: null,
|
|
780
|
-
costProvenance: {
|
|
781
|
-
kind: "uncaptured",
|
|
782
|
-
usd: null
|
|
783
|
-
}
|
|
784
|
-
};
|
|
785
|
-
}
|
|
786
|
-
function extractPromptCompletion(spans, callSpanIds) {
|
|
787
|
-
const callIds = new Set(callSpanIds);
|
|
788
|
-
const measuredCalls = spans.filter((span) => callIds.has(span.span_id));
|
|
789
|
-
const llm = (measuredCalls.length > 0 ? measuredCalls : spans.filter((s) => s.kind === "LLM")).sort((a, b) => compareSpanTime(a.start_time, b.start_time) || a.span_id.localeCompare(b.span_id));
|
|
790
|
-
if (llm.length === 0) return {};
|
|
791
|
-
const promptText = firstStringAttr(llm[0].attributes, [
|
|
792
|
-
"input.value",
|
|
793
|
-
"llm.input_messages",
|
|
794
|
-
"gen_ai.prompt"
|
|
795
|
-
]) ?? void 0;
|
|
796
|
-
const last = llm[llm.length - 1];
|
|
797
|
-
const completionText = firstStringAttr(last.attributes, [
|
|
798
|
-
"output.value",
|
|
799
|
-
"llm.output_messages",
|
|
800
|
-
"gen_ai.completion"
|
|
801
|
-
]) ?? void 0;
|
|
802
|
-
return {
|
|
803
|
-
...promptText !== void 0 ? { promptText } : {},
|
|
804
|
-
...completionText !== void 0 ? { completionText } : {}
|
|
805
|
-
};
|
|
806
|
-
}
|
|
807
|
-
function topVote(votes) {
|
|
808
|
-
let best = null;
|
|
809
|
-
let bestN = 0;
|
|
810
|
-
for (const [k, n] of votes) if (n > bestN || n === bestN && best !== null && k < best) {
|
|
811
|
-
best = k;
|
|
812
|
-
bestN = n;
|
|
813
|
-
}
|
|
814
|
-
return best;
|
|
815
|
-
}
|
|
816
|
-
function firstModelAttr(spans) {
|
|
817
|
-
for (const s of spans) {
|
|
818
|
-
const m = firstStringAttr(s.attributes, LLM_MODEL_ATTR_KEYS) ?? s.model_name;
|
|
819
|
-
if (m) return m;
|
|
820
|
-
}
|
|
821
|
-
return null;
|
|
822
|
-
}
|
|
823
|
-
/**
|
|
824
|
-
* `validateRunRecord` rejects bare model aliases (`gpt-4o`) that remap
|
|
825
|
-
* silently. AppWorld/HALO traces frequently carry such bare ids (or a null
|
|
826
|
-
* model). When the model already encodes a snapshot we keep it; otherwise we
|
|
827
|
-
* append the fallback snapshot token so the row is admissible without lying
|
|
828
|
-
* about the model — the bare base name is preserved verbatim before `@`.
|
|
829
|
-
*/
|
|
830
|
-
function ensureSnapshot(model, fallbackModel) {
|
|
831
|
-
if (modelHasSnapshot(model)) return model;
|
|
832
|
-
return `${model}${fallbackModel.includes("@") ? fallbackModel.slice(fallbackModel.indexOf("@")) : "@otlp"}`;
|
|
833
|
-
}
|
|
834
|
-
//#endregion
|
|
835
|
-
//#region src/trace-analyst/store-tool-spans.ts
|
|
836
|
-
/** Missing tool spans cannot distinguish a tool-free run from broken capture. */
|
|
837
|
-
var ToolTraceMissingError = class extends CaptureIntegrityError {
|
|
838
|
-
constructor() {
|
|
839
|
-
super("toolSpansToTraceAnalysisStore: no tool spans supplied; trace evidence is missing");
|
|
840
|
-
}
|
|
841
|
-
};
|
|
842
|
-
/**
|
|
843
|
-
* Snapshot canonical tool spans into the read interface used by trace analysts.
|
|
844
|
-
* One run becomes one trace while arguments, results, errors, and timing remain searchable.
|
|
845
|
-
*/
|
|
846
|
-
function toolSpansToTraceAnalysisStore(spans, options = {}) {
|
|
847
|
-
if (!spans || spans.length === 0) throw new ToolTraceMissingError();
|
|
848
|
-
const seen = /* @__PURE__ */ new Set();
|
|
849
|
-
const lines = spans.map((span, index) => {
|
|
850
|
-
assertToolSpanIdentity(span, index);
|
|
851
|
-
const identity = `${span.runId}\u0000${span.spanId}`;
|
|
852
|
-
if (seen.has(identity)) throw new CaptureIntegrityError(`toolSpansToTraceAnalysisStore: duplicate span '${span.spanId}' in run '${span.runId}'`);
|
|
853
|
-
seen.add(identity);
|
|
854
|
-
const attributes = { ...span.attributes ?? {} };
|
|
855
|
-
applyToolSpanOtlpAttributes(attributes, span);
|
|
856
|
-
attributes[OPENINFERENCE_SPAN_KIND] = "TOOL";
|
|
857
|
-
const endedAt = span.endedAt ?? span.startedAt + (span.latencyMs ?? 0);
|
|
858
|
-
const line = createOtlpFlatLine({
|
|
859
|
-
traceId: span.runId,
|
|
860
|
-
spanId: span.spanId,
|
|
861
|
-
parentSpanId: span.parentSpanId ?? null,
|
|
862
|
-
name: span.name,
|
|
863
|
-
kind: "SPAN_KIND_INTERNAL",
|
|
864
|
-
startTime: toolSpanTimeIso(span.startedAt, span.spanId, "startedAt"),
|
|
865
|
-
endTime: toolSpanTimeIso(endedAt, span.spanId, "endedAt"),
|
|
866
|
-
statusCode: spanStatusToOtlp(span.status, span.error, "STATUS_CODE_UNSET"),
|
|
867
|
-
statusMessage: span.error,
|
|
868
|
-
resource: { attributes: {} },
|
|
869
|
-
attributes
|
|
870
|
-
});
|
|
871
|
-
try {
|
|
872
|
-
return JSON.stringify(line);
|
|
873
|
-
} catch (cause) {
|
|
874
|
-
throw new CaptureIntegrityError(`toolSpansToTraceAnalysisStore: span '${span.spanId}' in run '${span.runId}' is not JSON-serializable`, { cause });
|
|
875
|
-
}
|
|
876
|
-
});
|
|
877
|
-
return createOtlpBufferTraceStore(Buffer.from(`${lines.join("\n")}\n`, "utf8"), options);
|
|
878
|
-
}
|
|
879
|
-
function assertToolSpanIdentity(span, index) {
|
|
880
|
-
if (span.kind !== "tool") throw new CaptureIntegrityError(`toolSpansToTraceAnalysisStore: span at index ${index} has kind '${String(span.kind)}', not 'tool'`);
|
|
881
|
-
if (!span.runId || !span.spanId || !span.name || !span.toolName) throw new CaptureIntegrityError(`toolSpansToTraceAnalysisStore: span at index ${index} is missing runId, spanId, name, or toolName`);
|
|
882
|
-
if (!Number.isFinite(span.startedAt)) throw new CaptureIntegrityError(`toolSpansToTraceAnalysisStore: span '${span.spanId}' has invalid startedAt`);
|
|
883
|
-
if (span.endedAt !== void 0 && !Number.isFinite(span.endedAt)) throw new CaptureIntegrityError(`toolSpansToTraceAnalysisStore: span '${span.spanId}' has invalid endedAt`);
|
|
884
|
-
if (span.endedAt !== void 0 && span.endedAt < span.startedAt) throw new CaptureIntegrityError(`toolSpansToTraceAnalysisStore: span '${span.spanId}' ends before it starts`);
|
|
885
|
-
if (span.latencyMs !== void 0 && (!Number.isFinite(span.latencyMs) || span.latencyMs < 0)) throw new CaptureIntegrityError(`toolSpansToTraceAnalysisStore: span '${span.spanId}' has invalid latencyMs`);
|
|
886
|
-
}
|
|
887
|
-
function toolSpanTimeIso(value, spanId, field) {
|
|
888
|
-
const iso = epochMillisToIso(value);
|
|
889
|
-
if (iso) return iso;
|
|
890
|
-
throw new CaptureIntegrityError(`toolSpansToTraceAnalysisStore: span '${spanId}' has invalid ${field}`);
|
|
891
|
-
}
|
|
892
|
-
//#endregion
|
|
893
|
-
//#region src/trace/capture-fetch.ts
|
|
894
|
-
/**
|
|
895
|
-
* Wrap a provider `fetch` and record request, response, and error events.
|
|
896
|
-
*
|
|
897
|
-
* The returned value is a plain `typeof fetch`. Capture is best-effort by
|
|
898
|
-
* default; set `failClosed` when telemetry loss must stop the provider call.
|
|
899
|
-
*/
|
|
900
|
-
const DEFAULT_BODY_CAP = 2 * 1024 * 1024;
|
|
901
|
-
function headersToRecord(headers) {
|
|
902
|
-
if (!headers) return void 0;
|
|
903
|
-
const out = {};
|
|
904
|
-
headers.forEach((value, key) => {
|
|
905
|
-
out[key.toLowerCase()] = value;
|
|
906
|
-
});
|
|
907
|
-
return Object.keys(out).length > 0 ? out : void 0;
|
|
908
|
-
}
|
|
909
|
-
function parseMaybeJson(text) {
|
|
910
|
-
if (text.length === 0) return void 0;
|
|
911
|
-
try {
|
|
912
|
-
return JSON.parse(text);
|
|
913
|
-
} catch {
|
|
914
|
-
return text;
|
|
915
|
-
}
|
|
916
|
-
}
|
|
917
|
-
/** Best-effort request-body read across the `fetch` input forms. */
|
|
918
|
-
async function readRequestBody(input, init) {
|
|
919
|
-
if (typeof init?.body === "string") return parseMaybeJson(init.body);
|
|
920
|
-
if (init?.body != null) return void 0;
|
|
921
|
-
if (input instanceof Request) try {
|
|
922
|
-
return parseMaybeJson(await input.clone().text());
|
|
923
|
-
} catch {
|
|
924
|
-
return;
|
|
925
|
-
}
|
|
926
|
-
}
|
|
927
|
-
function endpointFromUrl(url, baseUrl) {
|
|
928
|
-
const normalisedBase = baseUrl.replace(/\/+$/, "");
|
|
929
|
-
if (url.startsWith(normalisedBase)) return url.slice(normalisedBase.length) || "/";
|
|
930
|
-
try {
|
|
931
|
-
return new URL(url).pathname;
|
|
932
|
-
} catch {
|
|
933
|
-
return url;
|
|
934
|
-
}
|
|
935
|
-
}
|
|
936
|
-
function captureFetchToRawSink(fetch, sink, ctx, opts = {}) {
|
|
937
|
-
const provider = ctx.provider ?? providerFromBaseUrl(ctx.baseUrl);
|
|
938
|
-
const redactor = opts.redactor ?? defaultProviderRedactor;
|
|
939
|
-
const bodyCap = opts.responseBodyByteCap ?? DEFAULT_BODY_CAP;
|
|
940
|
-
let warned = false;
|
|
941
|
-
const baseEvent = (direction, endpoint) => ({
|
|
942
|
-
eventId: crypto.randomUUID(),
|
|
943
|
-
runId: ctx.runId,
|
|
944
|
-
spanId: ctx.spanId,
|
|
945
|
-
provider,
|
|
946
|
-
model: ctx.model,
|
|
947
|
-
endpoint,
|
|
948
|
-
baseUrl: ctx.baseUrl,
|
|
949
|
-
attemptIndex: 0,
|
|
950
|
-
direction,
|
|
951
|
-
timestamp: Date.now(),
|
|
952
|
-
redactedFields: []
|
|
953
|
-
});
|
|
954
|
-
const record = async (event) => {
|
|
955
|
-
try {
|
|
956
|
-
await sink.record(redactor(event));
|
|
957
|
-
} catch (err) {
|
|
958
|
-
if (opts.failClosed) throw err;
|
|
959
|
-
if (!warned) {
|
|
960
|
-
warned = true;
|
|
961
|
-
console.warn(`captureFetchToRawSink: sink.record failed (capture is best-effort) — ${err instanceof Error ? err.message : String(err)}`);
|
|
962
|
-
}
|
|
963
|
-
}
|
|
964
|
-
};
|
|
965
|
-
return async (input, init) => {
|
|
966
|
-
const url = typeof input === "string" ? input : input instanceof URL ? input.toString() : input.url;
|
|
967
|
-
const method = (init?.method ?? (input instanceof Request ? input.method : "GET")).toUpperCase();
|
|
968
|
-
const endpoint = endpointFromUrl(url, ctx.baseUrl);
|
|
969
|
-
const reqHeaders = new Headers(init?.headers ?? (input instanceof Request ? input.headers : void 0));
|
|
970
|
-
await record({
|
|
971
|
-
...baseEvent("request", endpoint),
|
|
972
|
-
requestHeaders: {
|
|
973
|
-
...headersToRecord(reqHeaders),
|
|
974
|
-
"x-http-method": method
|
|
975
|
-
},
|
|
976
|
-
requestBody: await readRequestBody(input, init)
|
|
977
|
-
});
|
|
978
|
-
const start = Date.now();
|
|
979
|
-
let response;
|
|
980
|
-
try {
|
|
981
|
-
response = await fetch(input, init);
|
|
982
|
-
} catch (err) {
|
|
983
|
-
await record({
|
|
984
|
-
...baseEvent("error", endpoint),
|
|
985
|
-
durationMs: Date.now() - start,
|
|
986
|
-
errorMessage: err instanceof Error ? err.message : String(err)
|
|
987
|
-
});
|
|
988
|
-
throw err;
|
|
989
|
-
}
|
|
990
|
-
let responseBody;
|
|
991
|
-
let rawText;
|
|
992
|
-
const redactedFields = [];
|
|
993
|
-
try {
|
|
994
|
-
rawText = await response.clone().text();
|
|
995
|
-
if (rawText.length > bodyCap) {
|
|
996
|
-
responseBody = rawText.slice(0, bodyCap);
|
|
997
|
-
redactedFields.push("body_truncated");
|
|
998
|
-
} else responseBody = parseMaybeJson(rawText);
|
|
999
|
-
} catch {
|
|
1000
|
-
responseBody = void 0;
|
|
1001
|
-
}
|
|
1002
|
-
if (opts.onUsage && rawText !== void 0) try {
|
|
1003
|
-
const parsedForUsage = parseMaybeJson(rawText);
|
|
1004
|
-
const usage = extractUsage(parsedForUsage) ?? (typeof parsedForUsage === "string" ? extractUsageFromSse(rawText, { mode: opts.sseUsageMode }) : null);
|
|
1005
|
-
if (usage) opts.onUsage(usage, ctx);
|
|
1006
|
-
} catch (err) {
|
|
1007
|
-
if (opts.failClosed) throw err;
|
|
1008
|
-
}
|
|
1009
|
-
await record({
|
|
1010
|
-
...baseEvent("response", endpoint),
|
|
1011
|
-
durationMs: Date.now() - start,
|
|
1012
|
-
statusCode: response.status,
|
|
1013
|
-
responseHeaders: headersToRecord(response.headers),
|
|
1014
|
-
responseBody,
|
|
1015
|
-
redactedFields
|
|
1016
|
-
});
|
|
1017
|
-
return response;
|
|
1018
|
-
};
|
|
1019
|
-
}
|
|
1020
|
-
//#endregion
|
|
1021
|
-
//#region src/trace/wire-ids.ts
|
|
1022
|
-
/**
|
|
1023
|
-
* The ONE mapping from agent-eval's human-readable ids (run ids, span labels)
|
|
1024
|
-
* to W3C/OTLP wire ids. Every exporter in this package MUST route through
|
|
1025
|
-
* these two functions — two exporters with private paddings once produced
|
|
1026
|
-
* DIFFERENT trace ids for the same run, and one emitted invalid hex embedding
|
|
1027
|
-
* the raw run id in the wire id (tangle-network/agent-runtime#694).
|
|
1028
|
-
*
|
|
1029
|
-
* Semantics:
|
|
1030
|
-
* - an id that is ALREADY a valid W3C id passes through unchanged, so a
|
|
1031
|
-
* trace id received from an inbound `traceparent` survives the round-trip
|
|
1032
|
-
* and cross-process correlation is preserved;
|
|
1033
|
-
* - anything else is derived with the contract's `deriveHexId`, the only
|
|
1034
|
-
* legal derivation — deterministic, so every process that derives from the
|
|
1035
|
-
* same human id mints the SAME wire id.
|
|
1036
|
-
*/
|
|
1037
|
-
/** 32-hex W3C trace id for any id string. */
|
|
1038
|
-
function traceIdForWire(id) {
|
|
1039
|
-
return isW3CTraceId(id) ? id : deriveHexId(id, 16);
|
|
1040
|
-
}
|
|
1041
|
-
/** 16-hex W3C span id for any id string. */
|
|
1042
|
-
function spanIdForWire(id) {
|
|
1043
|
-
return isW3CSpanId(id) ? id : deriveHexId(id, 8);
|
|
1044
|
-
}
|
|
1045
|
-
//#endregion
|
|
1046
|
-
//#region src/trace/otel.ts
|
|
1047
|
-
/**
|
|
1048
|
-
* OpenTelemetry JSON export — maps TraceSchema v1 to OTLP/JSON so
|
|
1049
|
-
* traces render natively in Jaeger / Honeycomb / Langfuse / Grafana.
|
|
1050
|
-
*
|
|
1051
|
-
* Wire format only. We do NOT depend on the @opentelemetry SDK — that
|
|
1052
|
-
* would drag in polyfills incompatible with Workers/Edge. Consumers
|
|
1053
|
-
* push the JSON to their collector of choice via HTTP.
|
|
1054
|
-
*
|
|
1055
|
-
* Reference: OTLP 1.3.2 (ResourceSpans / ScopeSpans / Span).
|
|
1056
|
-
*/
|
|
1057
|
-
const OTEL_AGENT_EVAL_SCOPE = {
|
|
1058
|
-
name: "@tangle-network/agent-eval",
|
|
1059
|
-
version: "0.3.0"
|
|
1060
|
-
};
|
|
1061
|
-
/** Export a single run's spans + events in OTLP/JSON. */
|
|
1062
|
-
async function exportRunAsOtlp(store, runId, resourceAttrs = {}) {
|
|
1063
|
-
const run = await store.getRun(runId);
|
|
1064
|
-
if (!run) throw new Error(`run ${runId} not found`);
|
|
1065
|
-
const spans = await store.spans({ runId });
|
|
1066
|
-
const events = await store.events({ runId });
|
|
1067
|
-
const eventsBySpan = /* @__PURE__ */ new Map();
|
|
1068
|
-
for (const e of events) {
|
|
1069
|
-
if (!e.spanId) continue;
|
|
1070
|
-
const arr = eventsBySpan.get(e.spanId) ?? [];
|
|
1071
|
-
arr.push(e);
|
|
1072
|
-
eventsBySpan.set(e.spanId, arr);
|
|
1073
|
-
}
|
|
1074
|
-
const traceId = runToTraceId(run);
|
|
1075
|
-
const otlpSpans = spans.map((s) => spanToOtlp(s, traceId, eventsBySpan.get(s.spanId) ?? []));
|
|
1076
|
-
return { resourceSpans: [{
|
|
1077
|
-
resource: { attributes: toAttributes$1({
|
|
1078
|
-
"service.name": "agent-eval",
|
|
1079
|
-
"run.id": run.runId,
|
|
1080
|
-
"run.scenario_id": run.scenarioId,
|
|
1081
|
-
"run.variant_id": run.variantId ?? "",
|
|
1082
|
-
"run.dataset_version": run.datasetVersion ?? "",
|
|
1083
|
-
"run.code_sha": run.codeSha ?? "",
|
|
1084
|
-
"run.model_fingerprint": run.modelFingerprint ?? "",
|
|
1085
|
-
...resourceAttrs
|
|
1086
|
-
}) },
|
|
1087
|
-
scopeSpans: [{
|
|
1088
|
-
scope: OTEL_AGENT_EVAL_SCOPE,
|
|
1089
|
-
spans: otlpSpans
|
|
1090
|
-
}]
|
|
1091
|
-
}] };
|
|
1092
|
-
}
|
|
1093
|
-
function spanToOtlp(span, traceId, events) {
|
|
1094
|
-
const endedAt = span.endedAt ?? span.startedAt;
|
|
1095
|
-
return {
|
|
1096
|
-
traceId,
|
|
1097
|
-
spanId: spanIdForWire(span.spanId),
|
|
1098
|
-
parentSpanId: span.parentSpanId ? spanIdForWire(span.parentSpanId) : void 0,
|
|
1099
|
-
name: span.name,
|
|
1100
|
-
kind: 1,
|
|
1101
|
-
startTimeUnixNano: msToNs$1(span.startedAt),
|
|
1102
|
-
endTimeUnixNano: msToNs$1(endedAt),
|
|
1103
|
-
attributes: toAttributes$1(flattenSpanAttributes(span)),
|
|
1104
|
-
events: events.map((e) => ({
|
|
1105
|
-
timeUnixNano: msToNs$1(e.timestamp),
|
|
1106
|
-
name: e.kind,
|
|
1107
|
-
attributes: toAttributes$1(flattenPayload(e.payload))
|
|
1108
|
-
})),
|
|
1109
|
-
status: span.status === "error" ? {
|
|
1110
|
-
code: 2,
|
|
1111
|
-
message: span.error
|
|
1112
|
-
} : { code: 1 }
|
|
1113
|
-
};
|
|
1114
|
-
}
|
|
1115
|
-
function flattenSpanAttributes(span) {
|
|
1116
|
-
const base = {};
|
|
1117
|
-
if (span.attributes) {
|
|
1118
|
-
for (const [k, v] of Object.entries(span.attributes)) if (typeof v === "string" || typeof v === "number" || typeof v === "boolean") base[k] = v;
|
|
1119
|
-
}
|
|
1120
|
-
base[OPENINFERENCE_SPAN_KIND] = traceSpanKindToOpenInferenceKind(span.kind);
|
|
1121
|
-
if (span.kind === "llm") applyLlmSpanOtlpAttributes(base, span);
|
|
1122
|
-
else if (span.kind === "tool") applyToolSpanOtlpAttributes(base, span);
|
|
1123
|
-
else if (span.kind === "retrieval") {
|
|
1124
|
-
base["retrieval.query"] = span.query;
|
|
1125
|
-
base["retrieval.hits"] = span.hits.length;
|
|
1126
|
-
} else if (span.kind === "judge") {
|
|
1127
|
-
base["judge.id"] = span.judgeId;
|
|
1128
|
-
base["judge.dimension"] = span.dimension;
|
|
1129
|
-
base["judge.score"] = span.score;
|
|
1130
|
-
base["judge.target_span_id"] = span.targetSpanId;
|
|
1131
|
-
} else if (span.kind === "sandbox") {
|
|
1132
|
-
if (span.image) base["sandbox.image"] = span.image;
|
|
1133
|
-
if (span.exitCode !== void 0) base["sandbox.exit_code"] = span.exitCode;
|
|
1134
|
-
if (span.testsPassed !== void 0) base["sandbox.tests_passed"] = span.testsPassed;
|
|
1135
|
-
if (span.testsTotal !== void 0) base["sandbox.tests_total"] = span.testsTotal;
|
|
1136
|
-
}
|
|
1137
|
-
return base;
|
|
1138
|
-
}
|
|
1139
|
-
function flattenPayload(payload) {
|
|
1140
|
-
const out = {};
|
|
1141
|
-
for (const [k, v] of Object.entries(payload)) if (typeof v === "string" || typeof v === "number" || typeof v === "boolean") out[k] = v;
|
|
1142
|
-
else out[k] = JSON.stringify(v);
|
|
1143
|
-
return out;
|
|
1144
|
-
}
|
|
1145
|
-
function toAttributes$1(record) {
|
|
1146
|
-
return Object.entries(record).map(([key, value]) => ({
|
|
1147
|
-
key,
|
|
1148
|
-
value: typeof value === "number" ? Number.isInteger(value) ? { intValue: value.toString() } : { doubleValue: value } : typeof value === "boolean" ? { boolValue: value } : { stringValue: value }
|
|
1149
|
-
}));
|
|
1150
|
-
}
|
|
1151
|
-
function msToNs$1(ms) {
|
|
1152
|
-
return (BigInt(Math.floor(ms)) * 1000000n).toString();
|
|
1153
|
-
}
|
|
1154
|
-
function runToTraceId(run) {
|
|
1155
|
-
return traceIdForWire(run.runId);
|
|
1156
|
-
}
|
|
1157
|
-
//#endregion
|
|
1158
|
-
//#region src/trace/otel-bridge.ts
|
|
1159
|
-
/**
|
|
1160
|
-
* Create a RunCompleteHook that exports all spans from the completed run
|
|
1161
|
-
* to the OTEL exporter, then flushes.
|
|
1162
|
-
*/
|
|
1163
|
-
function otelRunCompleteHook(exporter) {
|
|
1164
|
-
return async (ctx) => {
|
|
1165
|
-
const spans = await ctx.store.spans({ runId: ctx.runId });
|
|
1166
|
-
for (const span of spans) if (span.endedAt) exporter.exportSpan(storeSpanToExportable(span, ctx.runId));
|
|
1167
|
-
await exporter.flush();
|
|
1168
|
-
};
|
|
1169
|
-
}
|
|
1170
|
-
/**
|
|
1171
|
-
* Create an auto-exporting TraceStore wrapper that intercepts updateSpan
|
|
1172
|
-
* calls. When a span gets an endedAt, it's exported immediately. This
|
|
1173
|
-
* gives real-time streaming instead of batch-at-end.
|
|
1174
|
-
*
|
|
1175
|
-
* This is the preferred integration path: wrap the store before
|
|
1176
|
-
* constructing the TraceEmitter.
|
|
1177
|
-
*/
|
|
1178
|
-
function createOtelTracingStore(inner, exporter, traceId) {
|
|
1179
|
-
return {
|
|
1180
|
-
async appendRun(run) {
|
|
1181
|
-
return inner.appendRun(run);
|
|
1182
|
-
},
|
|
1183
|
-
async updateRun(runId, patch) {
|
|
1184
|
-
return inner.updateRun(runId, patch);
|
|
1185
|
-
},
|
|
1186
|
-
async appendSpan(span) {
|
|
1187
|
-
if (span.endedAt) exporter.exportSpan(storeSpanToExportable(span, traceId));
|
|
1188
|
-
return inner.appendSpan(span);
|
|
1189
|
-
},
|
|
1190
|
-
async updateSpan(spanId, patch) {
|
|
1191
|
-
await inner.updateSpan(spanId, patch);
|
|
1192
|
-
if (patch.endedAt) {
|
|
1193
|
-
const found = (await inner.spans({ runId: traceId })).find((s) => s.spanId === spanId);
|
|
1194
|
-
if (found) exporter.exportSpan(storeSpanToExportable(found, traceId));
|
|
1195
|
-
}
|
|
1196
|
-
},
|
|
1197
|
-
async appendEvent(event) {
|
|
1198
|
-
return inner.appendEvent(event);
|
|
1199
|
-
},
|
|
1200
|
-
async appendBudgetEntry(entry) {
|
|
1201
|
-
return inner.appendBudgetEntry(entry);
|
|
1202
|
-
},
|
|
1203
|
-
async appendArtifact(artifact) {
|
|
1204
|
-
return inner.appendArtifact(artifact);
|
|
1205
|
-
},
|
|
1206
|
-
getRun: inner.getRun.bind(inner),
|
|
1207
|
-
listRuns: inner.listRuns.bind(inner),
|
|
1208
|
-
spans: inner.spans.bind(inner),
|
|
1209
|
-
events: inner.events.bind(inner),
|
|
1210
|
-
budget: inner.budget.bind(inner),
|
|
1211
|
-
artifacts: inner.artifacts.bind(inner)
|
|
1212
|
-
};
|
|
1213
|
-
}
|
|
1214
|
-
function storeSpanToExportable(span, traceId) {
|
|
1215
|
-
const llm = span.kind === "llm" ? span : void 0;
|
|
1216
|
-
const tool = span.kind === "tool" ? span : void 0;
|
|
1217
|
-
return {
|
|
1218
|
-
traceId,
|
|
1219
|
-
spanId: span.spanId,
|
|
1220
|
-
parentSpanId: span.parentSpanId,
|
|
1221
|
-
name: span.name,
|
|
1222
|
-
kind: span.kind,
|
|
1223
|
-
startedAt: span.startedAt,
|
|
1224
|
-
endedAt: span.endedAt,
|
|
1225
|
-
status: span.status,
|
|
1226
|
-
error: span.error,
|
|
1227
|
-
model: llm?.model,
|
|
1228
|
-
inputTokens: llm?.inputTokens,
|
|
1229
|
-
outputTokens: llm?.outputTokens,
|
|
1230
|
-
reasoningTokens: llm?.reasoningTokens,
|
|
1231
|
-
cachedTokens: llm?.cachedTokens,
|
|
1232
|
-
cacheWriteTokens: llm?.cacheWriteTokens,
|
|
1233
|
-
costUsd: llm?.costUsd,
|
|
1234
|
-
tool: tool ? {
|
|
1235
|
-
toolName: tool.toolName,
|
|
1236
|
-
args: tool.args,
|
|
1237
|
-
argsCaptured: tool.argsCaptured,
|
|
1238
|
-
result: tool.result,
|
|
1239
|
-
latencyMs: tool.latencyMs
|
|
1240
|
-
} : void 0,
|
|
1241
|
-
attributes: span.attributes
|
|
1242
|
-
};
|
|
1243
|
-
}
|
|
1244
|
-
//#endregion
|
|
1245
|
-
//#region src/trace/otel-export.ts
|
|
1246
|
-
/**
|
|
1247
|
-
* OTEL span exporter — streams spans to an OTLP/HTTP collector.
|
|
1248
|
-
*
|
|
1249
|
-
* Reads OTEL_EXPORTER_OTLP_ENDPOINT + OTEL_EXPORTER_OTLP_HEADERS from env
|
|
1250
|
-
* when no explicit config is given. Batches spans and flushes periodically
|
|
1251
|
-
* or when the batch fills. No @opentelemetry SDK dependency — minimal
|
|
1252
|
-
* OTLP/JSON serializer (~120 LOC) using the existing otel.ts helpers.
|
|
1253
|
-
*/
|
|
1254
|
-
/**
|
|
1255
|
-
* Create an OTEL exporter. Returns undefined when no endpoint is configured
|
|
1256
|
-
* (neither via config nor env) — callers should check before attaching.
|
|
1257
|
-
*/
|
|
1258
|
-
function createOtelExporter(config) {
|
|
1259
|
-
const resolvedEndpoint = config?.endpoint ?? (typeof process !== "undefined" ? process.env.OTEL_EXPORTER_OTLP_ENDPOINT : void 0);
|
|
1260
|
-
if (!resolvedEndpoint) return void 0;
|
|
1261
|
-
const endpoint = resolvedEndpoint;
|
|
1262
|
-
const headers = config?.headers ?? parseHeadersFromEnv();
|
|
1263
|
-
const batchSize = config?.batchSize ?? 64;
|
|
1264
|
-
const flushIntervalMs = config?.flushIntervalMs ?? 5e3;
|
|
1265
|
-
const serviceName = config?.serviceName ?? "agent-eval";
|
|
1266
|
-
const resourceAttrs = config?.resourceAttributes ?? {};
|
|
1267
|
-
const pending = [];
|
|
1268
|
-
let timer;
|
|
1269
|
-
let stopped = false;
|
|
1270
|
-
const exporter = {
|
|
1271
|
-
exportSpan(span) {
|
|
1272
|
-
if (stopped) return;
|
|
1273
|
-
pending.push(toOtlpSpan(span));
|
|
1274
|
-
if (pending.length >= batchSize) doFlush();
|
|
1275
|
-
},
|
|
1276
|
-
async flush() {
|
|
1277
|
-
await doFlush();
|
|
1278
|
-
},
|
|
1279
|
-
async shutdown() {
|
|
1280
|
-
stopped = true;
|
|
1281
|
-
if (timer !== void 0) {
|
|
1282
|
-
clearInterval(timer);
|
|
1283
|
-
timer = void 0;
|
|
1284
|
-
}
|
|
1285
|
-
await doFlush();
|
|
1286
|
-
}
|
|
1287
|
-
};
|
|
1288
|
-
timer = setInterval(() => {
|
|
1289
|
-
if (pending.length > 0) doFlush();
|
|
1290
|
-
}, flushIntervalMs);
|
|
1291
|
-
if (typeof timer === "object" && "unref" in timer) timer.unref();
|
|
1292
|
-
async function doFlush() {
|
|
1293
|
-
if (pending.length === 0) return;
|
|
1294
|
-
const batch = pending.splice(0);
|
|
1295
|
-
const body = { resourceSpans: [{
|
|
1296
|
-
resource: { attributes: toAttributes({
|
|
1297
|
-
"service.name": serviceName,
|
|
1298
|
-
...resourceAttrs
|
|
1299
|
-
}) },
|
|
1300
|
-
scopeSpans: [{
|
|
1301
|
-
scope: OTEL_AGENT_EVAL_SCOPE,
|
|
1302
|
-
spans: batch
|
|
1303
|
-
}]
|
|
1304
|
-
}] };
|
|
1305
|
-
const url = `${endpoint.replace(/\/+$/, "")}/v1/traces`;
|
|
1306
|
-
try {
|
|
1307
|
-
await fetch(url, {
|
|
1308
|
-
method: "POST",
|
|
1309
|
-
headers: {
|
|
1310
|
-
"content-type": "application/json",
|
|
1311
|
-
...headers
|
|
1312
|
-
},
|
|
1313
|
-
body: JSON.stringify(body)
|
|
1314
|
-
});
|
|
1315
|
-
} catch {}
|
|
1316
|
-
}
|
|
1317
|
-
return exporter;
|
|
1318
|
-
}
|
|
1319
|
-
function parseHeadersFromEnv() {
|
|
1320
|
-
if (typeof process === "undefined") return {};
|
|
1321
|
-
const raw = process.env.OTEL_EXPORTER_OTLP_HEADERS;
|
|
1322
|
-
if (!raw) return {};
|
|
1323
|
-
const out = {};
|
|
1324
|
-
for (const pair of raw.split(",")) {
|
|
1325
|
-
const eq = pair.indexOf("=");
|
|
1326
|
-
if (eq < 0) continue;
|
|
1327
|
-
const key = pair.slice(0, eq).trim();
|
|
1328
|
-
const value = pair.slice(eq + 1).trim();
|
|
1329
|
-
if (key) out[key] = value;
|
|
1330
|
-
}
|
|
1331
|
-
return out;
|
|
1332
|
-
}
|
|
1333
|
-
function toOtlpSpan(span) {
|
|
1334
|
-
const endedAt = span.endedAt ?? span.startedAt;
|
|
1335
|
-
const attrs = {};
|
|
1336
|
-
if (span.attributes) {
|
|
1337
|
-
for (const [k, v] of Object.entries(span.attributes)) if (typeof v === "string" || typeof v === "number" || typeof v === "boolean") attrs[k] = v;
|
|
1338
|
-
}
|
|
1339
|
-
attrs[OPENINFERENCE_SPAN_KIND] = traceSpanKindToOpenInferenceKind(span.kind);
|
|
1340
|
-
applyLlmSpanOtlpAttributes(attrs, span);
|
|
1341
|
-
if (span.tool) applyToolSpanOtlpAttributes(attrs, span.tool);
|
|
1342
|
-
return {
|
|
1343
|
-
traceId: traceIdForWire(span.traceId),
|
|
1344
|
-
spanId: spanIdForWire(span.spanId),
|
|
1345
|
-
parentSpanId: span.parentSpanId ? spanIdForWire(span.parentSpanId) : void 0,
|
|
1346
|
-
name: span.name,
|
|
1347
|
-
kind: 1,
|
|
1348
|
-
startTimeUnixNano: msToNs(span.startedAt),
|
|
1349
|
-
endTimeUnixNano: msToNs(endedAt),
|
|
1350
|
-
attributes: toAttributes(attrs),
|
|
1351
|
-
status: span.status === "error" ? {
|
|
1352
|
-
code: 2,
|
|
1353
|
-
message: span.error
|
|
1354
|
-
} : { code: 1 }
|
|
1355
|
-
};
|
|
1356
|
-
}
|
|
1357
|
-
function toAttributes(record) {
|
|
1358
|
-
return Object.entries(record).map(([key, value]) => ({
|
|
1359
|
-
key,
|
|
1360
|
-
value: typeof value === "number" ? Number.isInteger(value) ? { intValue: value.toString() } : { doubleValue: value } : typeof value === "boolean" ? { boolValue: value } : { stringValue: value }
|
|
1361
|
-
}));
|
|
1362
|
-
}
|
|
1363
|
-
function msToNs(ms) {
|
|
1364
|
-
return (BigInt(Math.floor(ms)) * 1000000n).toString();
|
|
1365
|
-
}
|
|
1366
|
-
//#endregion
|
|
1367
|
-
//#region src/trace/store-to-otlp.ts
|
|
1368
|
-
/**
|
|
1369
|
-
* Convert agent-eval's internal trace shape (`FileSystemTraceStore` → `Run`,
|
|
1370
|
-
* `Span`, `TraceEvent`) into the OTLP-flat JSONL the trace analyst
|
|
1371
|
-
* (`analyzeTraces` + `OtlpFileTraceStore`) reads.
|
|
1372
|
-
*
|
|
1373
|
-
* Eval harnesses shard a `FileSystemTraceStore` per cell (persona / variant)
|
|
1374
|
-
* under a run directory. The analyst consumes a single OTLP-NDJSON file keyed
|
|
1375
|
-
* on `trace_id` + `span_id` with `start_time`/`end_time` in ISO-8601 and
|
|
1376
|
-
* resource + `attributes` rolled up per-span. This module walks every shard,
|
|
1377
|
-
* projects each `Span` (plus events pinned to it) into the flat OTLP shape,
|
|
1378
|
-
* and emits one NDJSON file.
|
|
1379
|
-
*
|
|
1380
|
-
* Generic OTLP/OpenInference fields are always emitted (`service.name`,
|
|
1381
|
-
* `agent.name`, `run.id`/`run.status`, `openinference.span.kind`,
|
|
1382
|
-
* `llm.model_name`, …). Domain attributes (`legal.*`, `tax.*`, …) are injected
|
|
1383
|
-
* per-run via {@link TraceStoreToOtlpOptions.resourceAttributes} /
|
|
1384
|
-
* {@link TraceStoreToOtlpOptions.runAttributes} so consumers don't re-roll the
|
|
1385
|
-
* walker.
|
|
1386
|
-
*/
|
|
1387
|
-
/**
|
|
1388
|
-
* Read every per-cell shard under each source root and write a flat OTLP-JSONL
|
|
1389
|
-
* view of the corpus to `outPath`. Each cell directory is a
|
|
1390
|
-
* `FileSystemTraceStore` — NDJSON append-only with size-based rotation;
|
|
1391
|
-
* `updateRun`/`updateSpan` append `{ id, ...patch, _update: true }` rows
|
|
1392
|
-
* rather than rewriting, so readers must merge those patches in (done here).
|
|
1393
|
-
*
|
|
1394
|
-
* A `string` source is treated as a celled root.
|
|
1395
|
-
*/
|
|
1396
|
-
function convertTraceStoresToOtlp(source, outPath, opts = {}) {
|
|
1397
|
-
const sources = Array.isArray(source) ? [...source] : typeof source === "string" ? [{
|
|
1398
|
-
root: source,
|
|
1399
|
-
layout: "celled"
|
|
1400
|
-
}] : [source];
|
|
1401
|
-
const defaultServiceName = opts.serviceName ?? "agent-eval";
|
|
1402
|
-
const resourceAttributes = opts.resourceAttributes ?? (() => ({}));
|
|
1403
|
-
const runAttributes = opts.runAttributes ?? (() => ({}));
|
|
1404
|
-
const lines = [];
|
|
1405
|
-
let spanCount = 0;
|
|
1406
|
-
let runCount = 0;
|
|
1407
|
-
let cellCount = 0;
|
|
1408
|
-
let cellErrorCount = 0;
|
|
1409
|
-
for (const src of sources) {
|
|
1410
|
-
const serviceName = src.serviceName ?? defaultServiceName;
|
|
1411
|
-
const cellDirs = src.layout === "flat" ? [{
|
|
1412
|
-
label: "<root>",
|
|
1413
|
-
dir: src.root
|
|
1414
|
-
}] : listCells(src.root).map((name) => ({
|
|
1415
|
-
label: name,
|
|
1416
|
-
dir: join(src.root, name)
|
|
1417
|
-
}));
|
|
1418
|
-
for (const cell of cellDirs) try {
|
|
1419
|
-
const result = projectCell({
|
|
1420
|
-
cellDir: cell.dir,
|
|
1421
|
-
serviceName,
|
|
1422
|
-
resourceAttributes,
|
|
1423
|
-
runAttributes
|
|
1424
|
-
});
|
|
1425
|
-
for (const line of result.lines) lines.push(line);
|
|
1426
|
-
spanCount += result.spanCount;
|
|
1427
|
-
runCount += result.runCount;
|
|
1428
|
-
cellCount += 1;
|
|
1429
|
-
} catch (err) {
|
|
1430
|
-
console.warn(`[traces-to-otlp] cell ${cell.label} (${cell.dir}) skipped: ${err instanceof Error ? err.message : String(err)}`);
|
|
1431
|
-
cellErrorCount += 1;
|
|
1432
|
-
}
|
|
1433
|
-
}
|
|
1434
|
-
writeFileSync(outPath, lines.join("\n") + (lines.length > 0 ? "\n" : ""));
|
|
1435
|
-
return {
|
|
1436
|
-
spanCount,
|
|
1437
|
-
runCount,
|
|
1438
|
-
cellCount,
|
|
1439
|
-
cellErrorCount
|
|
1440
|
-
};
|
|
1441
|
-
}
|
|
1442
|
-
function projectCell(args) {
|
|
1443
|
-
const { cellDir, serviceName, resourceAttributes, runAttributes } = args;
|
|
1444
|
-
const lines = [];
|
|
1445
|
-
let runCount = 0;
|
|
1446
|
-
let spanCount = 0;
|
|
1447
|
-
const runs = readMergedShards(cellDir, "runs", "runId");
|
|
1448
|
-
const spans = readMergedShards(cellDir, "spans", "spanId");
|
|
1449
|
-
const events = readShards(cellDir, "events");
|
|
1450
|
-
const runByRunId = /* @__PURE__ */ new Map();
|
|
1451
|
-
for (const r of runs) runByRunId.set(r.runId, r);
|
|
1452
|
-
const spanBySpanId = /* @__PURE__ */ new Map();
|
|
1453
|
-
for (const s of spans) spanBySpanId.set(s.spanId, s);
|
|
1454
|
-
const eventsBySpanId = /* @__PURE__ */ new Map();
|
|
1455
|
-
for (const e of events) {
|
|
1456
|
-
if (e.kind === "state_mutation" && e.payload && typeof e.payload === "object") {
|
|
1457
|
-
const entity = e.payload.entity;
|
|
1458
|
-
if (entity === "run") {
|
|
1459
|
-
const run = e.payload.run;
|
|
1460
|
-
if (run?.runId) runByRunId.set(run.runId, run);
|
|
1461
|
-
continue;
|
|
1462
|
-
}
|
|
1463
|
-
if (entity === "run.update") {
|
|
1464
|
-
const patch = e.payload.patch;
|
|
1465
|
-
if (patch && e.runId) {
|
|
1466
|
-
const prior = runByRunId.get(e.runId);
|
|
1467
|
-
if (prior) runByRunId.set(e.runId, {
|
|
1468
|
-
...prior,
|
|
1469
|
-
...patch
|
|
1470
|
-
});
|
|
1471
|
-
}
|
|
1472
|
-
continue;
|
|
1473
|
-
}
|
|
1474
|
-
if (entity === "span") {
|
|
1475
|
-
const span = e.payload.span;
|
|
1476
|
-
if (span?.spanId) spanBySpanId.set(span.spanId, span);
|
|
1477
|
-
continue;
|
|
1478
|
-
}
|
|
1479
|
-
if (entity === "span.update") {
|
|
1480
|
-
const spanId = e.payload.spanId;
|
|
1481
|
-
const patch = e.payload.patch;
|
|
1482
|
-
if (spanId && patch) {
|
|
1483
|
-
const prior = spanBySpanId.get(spanId);
|
|
1484
|
-
if (prior) spanBySpanId.set(spanId, {
|
|
1485
|
-
...prior,
|
|
1486
|
-
...patch
|
|
1487
|
-
});
|
|
1488
|
-
}
|
|
1489
|
-
continue;
|
|
1490
|
-
}
|
|
1491
|
-
}
|
|
1492
|
-
if (!e.spanId) continue;
|
|
1493
|
-
const arr = eventsBySpanId.get(e.spanId) ?? [];
|
|
1494
|
-
arr.push(e);
|
|
1495
|
-
eventsBySpanId.set(e.spanId, arr);
|
|
1496
|
-
}
|
|
1497
|
-
for (const run of runByRunId.values()) {
|
|
1498
|
-
const traceId = traceIdForWire(run.runId);
|
|
1499
|
-
const agentName = run.variantId ?? run.scenarioId;
|
|
1500
|
-
const sharedResource = { attributes: {
|
|
1501
|
-
"service.name": serviceName,
|
|
1502
|
-
"agent.name": agentName,
|
|
1503
|
-
"run.id": run.runId,
|
|
1504
|
-
"run.status": run.status,
|
|
1505
|
-
...resourceAttributes(run)
|
|
1506
|
-
} };
|
|
1507
|
-
const runSpanId = spanIdForWire(`run-${run.runId}`);
|
|
1508
|
-
const runStart = msToIso(run.startedAt);
|
|
1509
|
-
const runEnd = msToIso(run.endedAt ?? run.startedAt);
|
|
1510
|
-
const runStatus = run.outcome?.failureClass && run.outcome.failureClass !== "success" ? "STATUS_CODE_ERROR" : "STATUS_CODE_OK";
|
|
1511
|
-
const runAttrs = {
|
|
1512
|
-
[OPENINFERENCE_SPAN_KIND]: "AGENT",
|
|
1513
|
-
"agent.name": agentName,
|
|
1514
|
-
"agent.workflow.name": serviceName,
|
|
1515
|
-
...runAttributes(run)
|
|
1516
|
-
};
|
|
1517
|
-
lines.push(JSON.stringify(toLine({
|
|
1518
|
-
traceId,
|
|
1519
|
-
spanId: runSpanId,
|
|
1520
|
-
parentSpanId: "",
|
|
1521
|
-
name: `run.${agentName}`,
|
|
1522
|
-
kind: "SPAN_KIND_INTERNAL",
|
|
1523
|
-
startTime: runStart,
|
|
1524
|
-
endTime: runEnd,
|
|
1525
|
-
statusCode: runStatus,
|
|
1526
|
-
statusMessage: run.outcome?.notes ?? "",
|
|
1527
|
-
resource: sharedResource,
|
|
1528
|
-
attributes: runAttrs
|
|
1529
|
-
})));
|
|
1530
|
-
runCount += 1;
|
|
1531
|
-
for (const span of spanBySpanId.values()) {
|
|
1532
|
-
if (span.runId !== run.runId) continue;
|
|
1533
|
-
const spanAttrs = spanToAttributes(span, eventsBySpanId.get(span.spanId) ?? []);
|
|
1534
|
-
const statusCode = spanStatusToOtlp(span.status, span.error, "STATUS_CODE_OK");
|
|
1535
|
-
lines.push(JSON.stringify(toLine({
|
|
1536
|
-
traceId,
|
|
1537
|
-
spanId: spanIdForWire(span.spanId),
|
|
1538
|
-
parentSpanId: span.parentSpanId ? spanIdForWire(span.parentSpanId) : runSpanId,
|
|
1539
|
-
name: span.name,
|
|
1540
|
-
kind: spanKindToOtlpKind(span.kind),
|
|
1541
|
-
startTime: msToIso(span.startedAt),
|
|
1542
|
-
endTime: msToIso(span.endedAt ?? span.startedAt),
|
|
1543
|
-
statusCode,
|
|
1544
|
-
statusMessage: span.error ?? "",
|
|
1545
|
-
resource: sharedResource,
|
|
1546
|
-
attributes: spanAttrs
|
|
1547
|
-
})));
|
|
1548
|
-
spanCount += 1;
|
|
1549
|
-
}
|
|
1550
|
-
}
|
|
1551
|
-
return {
|
|
1552
|
-
lines,
|
|
1553
|
-
runCount,
|
|
1554
|
-
spanCount
|
|
1555
|
-
};
|
|
1556
|
-
}
|
|
1557
|
-
function listCells(root) {
|
|
1558
|
-
try {
|
|
1559
|
-
return readdirSync(root, { withFileTypes: true }).filter((d) => d.isDirectory()).map((d) => d.name).sort();
|
|
1560
|
-
} catch {
|
|
1561
|
-
return [];
|
|
1562
|
-
}
|
|
1563
|
-
}
|
|
1564
|
-
/**
|
|
1565
|
-
* Read every NDJSON shard for `name` under `cellDir`, ordered by mtime so
|
|
1566
|
-
* rotated files apply before the active one. Yields raw rows including any
|
|
1567
|
-
* `_update: true` patches.
|
|
1568
|
-
*/
|
|
1569
|
-
function readShards(cellDir, name) {
|
|
1570
|
-
let entries;
|
|
1571
|
-
try {
|
|
1572
|
-
entries = readdirSync(cellDir);
|
|
1573
|
-
} catch {
|
|
1574
|
-
return [];
|
|
1575
|
-
}
|
|
1576
|
-
const shards = entries.filter((f) => (f === `${name}.ndjson` || f.startsWith(`${name}.`)) && f.endsWith(".ndjson")).map((f) => ({
|
|
1577
|
-
file: f,
|
|
1578
|
-
path: join(cellDir, f)
|
|
1579
|
-
})).map((s) => {
|
|
1580
|
-
let mtime = 0;
|
|
1581
|
-
try {
|
|
1582
|
-
mtime = statSync(s.path).mtimeMs;
|
|
1583
|
-
} catch {}
|
|
1584
|
-
return {
|
|
1585
|
-
...s,
|
|
1586
|
-
mtime
|
|
1587
|
-
};
|
|
1588
|
-
}).sort((a, b) => a.mtime - b.mtime || a.file.localeCompare(b.file));
|
|
1589
|
-
const rows = [];
|
|
1590
|
-
for (const shard of shards) {
|
|
1591
|
-
let text;
|
|
1592
|
-
try {
|
|
1593
|
-
text = readFileSync(shard.path, "utf-8");
|
|
1594
|
-
} catch {
|
|
1595
|
-
continue;
|
|
1596
|
-
}
|
|
1597
|
-
for (const line of text.split("\n")) {
|
|
1598
|
-
const trimmed = line.trim();
|
|
1599
|
-
if (!trimmed) continue;
|
|
1600
|
-
try {
|
|
1601
|
-
rows.push(JSON.parse(trimmed));
|
|
1602
|
-
} catch {}
|
|
1603
|
-
}
|
|
1604
|
-
}
|
|
1605
|
-
return rows;
|
|
1606
|
-
}
|
|
1607
|
-
/**
|
|
1608
|
-
* Read NDJSON shards and merge `{ ...patch, _update: true }` rows into the
|
|
1609
|
-
* prior record keyed on `idKey` — mirrors the in-memory merge
|
|
1610
|
-
* `FileSystemTraceStore` keeps but doesn't replay on cross-process load.
|
|
1611
|
-
*/
|
|
1612
|
-
function readMergedShards(cellDir, name, idKey) {
|
|
1613
|
-
const rows = readShards(cellDir, name);
|
|
1614
|
-
const byId = /* @__PURE__ */ new Map();
|
|
1615
|
-
for (const row of rows) {
|
|
1616
|
-
const id = row[idKey];
|
|
1617
|
-
if (!id) continue;
|
|
1618
|
-
const prior = byId.get(id);
|
|
1619
|
-
if (prior && row._update) byId.set(id, {
|
|
1620
|
-
...prior,
|
|
1621
|
-
...row,
|
|
1622
|
-
_update: void 0
|
|
1623
|
-
});
|
|
1624
|
-
else byId.set(id, row);
|
|
1625
|
-
}
|
|
1626
|
-
return [...byId.values()];
|
|
1627
|
-
}
|
|
1628
|
-
function spanToAttributes(span, events) {
|
|
1629
|
-
const attrs = { [OPENINFERENCE_SPAN_KIND]: traceSpanKindToOpenInferenceKind(span.kind) };
|
|
1630
|
-
if (span.kind === "llm") {
|
|
1631
|
-
applyLlmSpanOtlpAttributes(attrs, span);
|
|
1632
|
-
if (Array.isArray(span.messages)) attrs["llm.input_messages"] = JSON.stringify(span.messages.slice(-6));
|
|
1633
|
-
if (typeof span.output === "string") attrs["llm.output_messages"] = JSON.stringify([{
|
|
1634
|
-
role: "assistant",
|
|
1635
|
-
content: span.output
|
|
1636
|
-
}]);
|
|
1637
|
-
} else if (span.kind === "tool") applyToolSpanOtlpAttributes(attrs, span);
|
|
1638
|
-
else if (span.kind === "judge") {
|
|
1639
|
-
attrs["judge.id"] = span.judgeId;
|
|
1640
|
-
attrs["judge.dimension"] = span.dimension;
|
|
1641
|
-
attrs["judge.score"] = span.score;
|
|
1642
|
-
attrs["judge.target_span_id"] = span.targetSpanId;
|
|
1643
|
-
}
|
|
1644
|
-
if (span.attributes) for (const [k, v] of Object.entries(span.attributes)) attrs[`agent_eval.${k}`] = v;
|
|
1645
|
-
if (events.length > 0) {
|
|
1646
|
-
attrs["agent_eval.event_count"] = events.length;
|
|
1647
|
-
attrs["agent_eval.event_kinds"] = JSON.stringify(events.map((e) => e.kind));
|
|
1648
|
-
}
|
|
1649
|
-
return attrs;
|
|
1650
|
-
}
|
|
1651
|
-
function spanKindToOtlpKind(kind) {
|
|
1652
|
-
switch (kind) {
|
|
1653
|
-
case "llm": return "SPAN_KIND_CLIENT";
|
|
1654
|
-
case "retrieval": return "SPAN_KIND_CLIENT";
|
|
1655
|
-
default: return "SPAN_KIND_INTERNAL";
|
|
1656
|
-
}
|
|
1657
|
-
}
|
|
1658
|
-
const toLine = createOtlpFlatLine;
|
|
1659
|
-
function msToIso(ms) {
|
|
1660
|
-
if (ms <= 0) return (/* @__PURE__ */ new Date(0)).toISOString();
|
|
1661
|
-
return epochMillisToIso(ms) ?? (/* @__PURE__ */ new Date(0)).toISOString();
|
|
1662
|
-
}
|
|
1663
|
-
//#endregion
|
|
1664
|
-
//#region src/replay.ts
|
|
1665
|
-
/**
|
|
1666
|
-
* Replay-from-raw-events — turn every captured campaign run into a
|
|
1667
|
-
* re-runnable artifact.
|
|
1668
|
-
*
|
|
1669
|
-
* `RawProviderSink` captures every provider HTTP envelope; `runEvalCampaign`
|
|
1670
|
-
* makes that capture the default. Together they make every past run a
|
|
1671
|
-
* complete fingerprint of what happened on the wire — enough to replay
|
|
1672
|
-
* the run without burning new LLM cost.
|
|
1673
|
-
*
|
|
1674
|
-
* Three use cases this primitive enables:
|
|
1675
|
-
*
|
|
1676
|
-
* 1. **Post-hoc judging** — apply a new judge / rubric / scoring callback
|
|
1677
|
-
* to last week's runs without re-calling any LLM. The cost of trying
|
|
1678
|
-
* a new rubric drops from "another full sweep" to a CPU-bound replay.
|
|
1679
|
-
* 2. **Determinism audits** — replay the same campaign and verify the
|
|
1680
|
-
* raw responses match byte-for-byte. Any drift is a non-determinism
|
|
1681
|
-
* bug (in the harness, the prompt builder, the sandbox, …).
|
|
1682
|
-
* 3. **Free judge calibration** — run two judges on identical responses
|
|
1683
|
-
* and measure inter-judge agreement without doubling LLM spend.
|
|
1684
|
-
*
|
|
1685
|
-
* The interface is deliberately fetch-shaped. Inject `createReplayFetch`
|
|
1686
|
-
* into `LlmClientOptions.fetch` and every `callLlm` transparently reads
|
|
1687
|
-
* from the cache instead of calling the network. No new code path through
|
|
1688
|
-
* the LLM client is needed; the cache hit is invisible to the runner.
|
|
1689
|
-
*/
|
|
1690
|
-
var ReplayCacheMissError = class extends ReplayError {
|
|
1691
|
-
url;
|
|
1692
|
-
requestKey;
|
|
1693
|
-
constructor(url, requestKey, message) {
|
|
1694
|
-
super(message ?? `replay cache miss for ${url} (key=${requestKey})`);
|
|
1695
|
-
this.url = url;
|
|
1696
|
-
this.requestKey = requestKey;
|
|
1697
|
-
}
|
|
1698
|
-
};
|
|
1699
|
-
/**
|
|
1700
|
-
* In-memory deterministic cache of (request → response) keyed on a stable
|
|
1701
|
-
* hash of the request body. Built from a `RawProviderSink` containing
|
|
1702
|
-
* paired `request` and `response` events from a previous run.
|
|
1703
|
-
*
|
|
1704
|
-
* The cache is the source of truth for replay; `createReplayFetch` is a
|
|
1705
|
-
* thin wrapper that reads from it.
|
|
1706
|
-
*/
|
|
1707
|
-
var ReplayCache = class ReplayCache {
|
|
1708
|
-
byKey = /* @__PURE__ */ new Map();
|
|
1709
|
-
orphans = 0;
|
|
1710
|
-
byProvider = {};
|
|
1711
|
-
byModel = {};
|
|
1712
|
-
/**
|
|
1713
|
-
* Build a cache from a sink's events. The sink must implement `list()`.
|
|
1714
|
-
* Filter by `runId` / `spanId` to scope to a specific replay.
|
|
1715
|
-
*/
|
|
1716
|
-
static async fromSink(sink, filter = {}) {
|
|
1717
|
-
if (!sink.list) throw new ReplayError("ReplayCache.fromSink: sink must implement list() to be replayable.");
|
|
1718
|
-
const events = await sink.list(filter);
|
|
1719
|
-
return ReplayCache.fromEvents(events);
|
|
1720
|
-
}
|
|
1721
|
-
/** Build a cache from an in-memory event list. */
|
|
1722
|
-
static async fromEvents(events) {
|
|
1723
|
-
const cache = new ReplayCache();
|
|
1724
|
-
const groups = /* @__PURE__ */ new Map();
|
|
1725
|
-
for (const e of events) {
|
|
1726
|
-
const k = `${e.runId ?? ""}::${e.spanId ?? ""}::${e.attemptIndex}`;
|
|
1727
|
-
const g = groups.get(k) ?? {};
|
|
1728
|
-
if (e.direction === "request") g.req = e;
|
|
1729
|
-
else g.res = e;
|
|
1730
|
-
groups.set(k, g);
|
|
1731
|
-
}
|
|
1732
|
-
for (const g of groups.values()) {
|
|
1733
|
-
if (!g.req) continue;
|
|
1734
|
-
if (!g.res) {
|
|
1735
|
-
cache.orphans += 1;
|
|
1736
|
-
continue;
|
|
1737
|
-
}
|
|
1738
|
-
const key = await requestKey(g.req);
|
|
1739
|
-
cache.byKey.set(key, {
|
|
1740
|
-
request: g.req,
|
|
1741
|
-
response: g.res
|
|
1742
|
-
});
|
|
1743
|
-
cache.byProvider[g.req.provider] = (cache.byProvider[g.req.provider] ?? 0) + 1;
|
|
1744
|
-
cache.byModel[g.req.model] = (cache.byModel[g.req.model] ?? 0) + 1;
|
|
1745
|
-
}
|
|
1746
|
-
return cache;
|
|
1747
|
-
}
|
|
1748
|
-
/** Number of cacheable (request, response) pairs in the cache. */
|
|
1749
|
-
size() {
|
|
1750
|
-
return this.byKey.size;
|
|
1751
|
-
}
|
|
1752
|
-
stats() {
|
|
1753
|
-
return {
|
|
1754
|
-
total: this.byKey.size,
|
|
1755
|
-
byProvider: { ...this.byProvider },
|
|
1756
|
-
byModel: { ...this.byModel },
|
|
1757
|
-
orphanRequests: this.orphans
|
|
1758
|
-
};
|
|
1759
|
-
}
|
|
1760
|
-
/** Iterate every cached `(request, response)` pair in insertion order. */
|
|
1761
|
-
*entries() {
|
|
1762
|
-
for (const entry of this.byKey.values()) yield entry;
|
|
1763
|
-
}
|
|
1764
|
-
/**
|
|
1765
|
-
* Look up a cached response by hashing the (model, messages, temperature,
|
|
1766
|
-
* maxTokens, response_format) shape. Returns `undefined` on miss; the
|
|
1767
|
-
* caller decides whether to throw, fall back to the network, or skip.
|
|
1768
|
-
*/
|
|
1769
|
-
async lookup(requestBody) {
|
|
1770
|
-
const key = await keyFromBody(requestBody);
|
|
1771
|
-
return this.byKey.get(key);
|
|
1772
|
-
}
|
|
1773
|
-
};
|
|
1774
|
-
/**
|
|
1775
|
-
* Build a `fetch`-shaped function that serves cached responses out of a
|
|
1776
|
-
* `ReplayCache` for any URL ending in `/chat/completions`. Pass through
|
|
1777
|
-
* `LlmClientOptions.fetch` and `callLlm` becomes free.
|
|
1778
|
-
*
|
|
1779
|
-
* Non-`/chat/completions` URLs are passed straight to the fallback fetch
|
|
1780
|
-
* (default: `globalThis.fetch`). This matters because non-LLM HTTP work
|
|
1781
|
-
* (judge HTTP servers, sandbox callbacks) sometimes flows through the same
|
|
1782
|
-
* `fetch` and shouldn't be intercepted.
|
|
1783
|
-
*/
|
|
1784
|
-
function createReplayFetch(cache, opts = {}) {
|
|
1785
|
-
const onMiss = opts.onMiss ?? "throw";
|
|
1786
|
-
const fallback = opts.fallbackFetch ?? globalThis.fetch?.bind(globalThis);
|
|
1787
|
-
return (async (input, init) => {
|
|
1788
|
-
const url = typeof input === "string" ? input : input instanceof URL ? input.toString() : input.url;
|
|
1789
|
-
if (!/\/chat\/completions(?:[?#].*)?$/.test(url)) {
|
|
1790
|
-
if (!fallback) throw new ReplayError(`replay fetch: non-completions URL ${url} but no fallbackFetch configured`);
|
|
1791
|
-
return fallback(input, init);
|
|
1792
|
-
}
|
|
1793
|
-
let bodyParsed;
|
|
1794
|
-
if (init?.body && typeof init.body === "string") try {
|
|
1795
|
-
bodyParsed = JSON.parse(init.body);
|
|
1796
|
-
} catch {}
|
|
1797
|
-
const hit = bodyParsed === void 0 ? void 0 : await cache.lookup(bodyParsed);
|
|
1798
|
-
if (hit) {
|
|
1799
|
-
opts.onHit?.({
|
|
1800
|
-
url,
|
|
1801
|
-
provider: hit.request.provider,
|
|
1802
|
-
model: hit.request.model
|
|
1803
|
-
});
|
|
1804
|
-
const status = hit.response.statusCode ?? 200;
|
|
1805
|
-
const headers = new Headers(Object.entries(hit.response.responseHeaders ?? { "Content-Type": "application/json" }));
|
|
1806
|
-
const bodyText = typeof hit.response.responseBody === "string" ? hit.response.responseBody : JSON.stringify(hit.response.responseBody ?? {});
|
|
1807
|
-
return new Response(bodyText, {
|
|
1808
|
-
status,
|
|
1809
|
-
headers
|
|
1810
|
-
});
|
|
1811
|
-
}
|
|
1812
|
-
opts.onMissNotify?.({
|
|
1813
|
-
url,
|
|
1814
|
-
requestBody: bodyParsed
|
|
1815
|
-
});
|
|
1816
|
-
if (onMiss === "throw") throw new ReplayCacheMissError(url, bodyParsed === void 0 ? "<unparseable>" : await keyFromBody(bodyParsed));
|
|
1817
|
-
if (onMiss === "fail-closed") return new Response(JSON.stringify({ error: "replay_cache_miss" }), { status: 599 });
|
|
1818
|
-
if (!fallback) throw new ReplayError("replay fetch: onMiss=fallback but no fallbackFetch configured");
|
|
1819
|
-
return fallback(input, init);
|
|
1820
|
-
});
|
|
1821
|
-
}
|
|
1822
|
-
/**
|
|
1823
|
-
* Convenience iterator over `(request, response)` pairs in a sink — for
|
|
1824
|
-
* post-hoc scoring that doesn't need a `fetch` shim. The judge or scorer
|
|
1825
|
-
* runs purely in-process over cached LLM outputs.
|
|
1826
|
-
*/
|
|
1827
|
-
async function* iterateRawCalls(sink, filter = {}) {
|
|
1828
|
-
if (!sink.list) throw new ReplayError("iterateRawCalls: sink must implement list().");
|
|
1829
|
-
const events = await sink.list(filter);
|
|
1830
|
-
const cache = await ReplayCache.fromEvents(events);
|
|
1831
|
-
for (const entry of cache.entries()) yield entry;
|
|
1832
|
-
}
|
|
1833
|
-
/**
|
|
1834
|
-
* Canonical request key.
|
|
1835
|
-
*
|
|
1836
|
-
* `model + messages + temperature + max_tokens|max_completion_tokens +
|
|
1837
|
-
* response_format` are the dimensions that affect the response shape.
|
|
1838
|
-
* Other fields (timestamp headers, provider-specific metadata) are
|
|
1839
|
-
* intentionally excluded so a request hashes the same across re-runs.
|
|
1840
|
-
*/
|
|
1841
|
-
async function requestKey(event) {
|
|
1842
|
-
return keyFromBody(event.requestBody);
|
|
1843
|
-
}
|
|
1844
|
-
async function keyFromBody(body) {
|
|
1845
|
-
if (body == null || typeof body !== "object") return hashJson({ raw: String(body) });
|
|
1846
|
-
const b = body;
|
|
1847
|
-
return hashJson(canonicalize({
|
|
1848
|
-
model: b.model ?? null,
|
|
1849
|
-
messages: b.messages ?? null,
|
|
1850
|
-
temperature: b.temperature ?? null,
|
|
1851
|
-
max_tokens: b.max_tokens ?? null,
|
|
1852
|
-
max_completion_tokens: b.max_completion_tokens ?? null,
|
|
1853
|
-
response_format: b.response_format ?? null
|
|
1854
|
-
}));
|
|
1855
|
-
}
|
|
1856
|
-
//#endregion
|
|
1857
|
-
export { TRACE_ANALYST_ACTOR_DESCRIPTION as A, domainEvidencePattern as C, tokenizeDomainWords as D, scoreTraceInsightReadiness as E, traceAnalystOnRunComplete as O, describeTraceInsightScope as S, planTraceInsightQuestions as T, otlpToTraceRunRecords as _, convertTraceStoresToOtlp as a, buildTraceInsightPrompt as b, otelRunCompleteHook as c, captureFetchToRawSink as d, ToolTraceMissingError as f, otlpToRunRecords as g, otlpRowsToTraceRunRecords as h, iterateRawCalls as i, TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION as j, analyzeTraces as k, OTEL_AGENT_EVAL_SCOPE as l, otlpRowsToRunRecords as m, ReplayCacheMissError as n, createOtelExporter as o, toolSpansToTraceAnalysisStore as p, createReplayFetch as r, createOtelTracingStore as s, ReplayCache as t, exportRunAsOtlp as u, flattenOtlpExportToNdjson as v, inferDomainKeywords as w, defaultTraceInsightPanel as x, buildTraceInsightContext as y };
|
|
1858
|
-
|
|
1859
|
-
//# sourceMappingURL=replay-CohS93nE.js.map
|