@tangle-network/agent-eval 0.144.11 → 0.144.13
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +11 -0
- package/dist/agent-profile-B9_GGsG8.d.ts +84 -0
- package/dist/agent-profile-B9_GGsG8.d.ts.map +1 -0
- package/dist/{agent-profile-cell-BOP-iA9Q.d.ts → agent-profile-cell-BkcRDikH.d.ts} +2 -2
- package/dist/{agent-profile-cell-BOP-iA9Q.d.ts.map → agent-profile-cell-BkcRDikH.d.ts.map} +1 -1
- package/dist/analyst/index.d.ts +134 -16
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +364 -10
- package/dist/analyst/index.js.map +1 -1
- package/dist/backend-integrity-CsVin_Wb.d.ts +280 -0
- package/dist/backend-integrity-CsVin_Wb.d.ts.map +1 -0
- package/dist/{baseline-C-GocmIW.js → baseline-BhPRQBVn.js} +3 -2
- package/dist/{baseline-C-GocmIW.js.map → baseline-BhPRQBVn.js.map} +1 -1
- package/dist/{benchmark-CWeqGl7x.js → benchmark-BhT16ep9.js} +2 -2
- package/dist/{benchmark-CWeqGl7x.js.map → benchmark-BhT16ep9.js.map} +1 -1
- package/dist/{agent-profile-DPi7IZg7.d.ts → benchmark-Ceoan7vk.d.ts} +4 -91
- package/dist/benchmark-Ceoan7vk.d.ts.map +1 -0
- package/dist/{benchmark-command-BteMFN62.js → benchmark-command-CY6cbhOB.js} +36 -79
- package/dist/benchmark-command-CY6cbhOB.js.map +1 -0
- package/dist/benchmarks/index.d.ts +244 -2
- package/dist/benchmarks/index.d.ts.map +1 -0
- package/dist/benchmarks/index.js +733 -1
- package/dist/benchmarks/index.js.map +1 -0
- package/dist/builder-eval/index.d.ts +23 -2
- package/dist/builder-eval/index.d.ts.map +1 -1
- package/dist/builder-eval/index.js +227 -3
- package/dist/builder-eval/index.js.map +1 -1
- package/dist/campaign/index.d.ts +10 -8
- package/dist/campaign/index.js +9 -6
- package/dist/{campaign-C2TTzQII.js → campaign-BYjBAypg.js} +21 -634
- package/dist/campaign-BYjBAypg.js.map +1 -0
- package/dist/{canonical-D011XM8r.js → canonical-D-XsTQ6_.js} +2 -2
- package/dist/{canonical-D011XM8r.js.map → canonical-D-XsTQ6_.js.map} +1 -1
- package/dist/{index-CQTZ-4XN.d.ts → capture-fetch-DDvpjVRU.d.ts} +3 -3
- package/dist/capture-fetch-DDvpjVRU.d.ts.map +1 -0
- package/dist/{default-registry-BmktKy8r.js → chat-client-Bp4Ebuuc.js} +1276 -1276
- package/dist/chat-client-Bp4Ebuuc.js.map +1 -0
- package/dist/cli.js +2 -2
- package/dist/{client-C9gzZE59.d.ts → client-KUbGulm_.d.ts} +4 -4
- package/dist/{client-C9gzZE59.d.ts.map → client-KUbGulm_.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +13 -390
- package/dist/contract/index.d.ts.map +1 -1
- package/dist/contract/index.js +18 -542
- package/dist/contract/index.js.map +1 -1
- package/dist/{cost-ledger-DMFxsLKr.js → cost-ledger-BSe92yAV.js} +3 -3
- package/dist/{cost-ledger-DMFxsLKr.js.map → cost-ledger-BSe92yAV.js.map} +1 -1
- package/dist/{cost-ledger-Bv_e8XHY.d.ts → cost-ledger-DbQdN3nO.d.ts} +2 -2
- package/dist/{cost-ledger-Bv_e8XHY.d.ts.map → cost-ledger-DbQdN3nO.d.ts.map} +1 -1
- package/dist/{counterfactual-CxmxAONP.d.ts → counterfactual--bpysZF0.d.ts} +2 -14
- package/dist/{counterfactual-CxmxAONP.d.ts.map → counterfactual--bpysZF0.d.ts.map} +1 -1
- package/dist/{counterfactual-CWPTrMH7.js → counterfactual-lDfCx0Uz.js} +3 -31
- package/dist/{counterfactual-CWPTrMH7.js.map → counterfactual-lDfCx0Uz.js.map} +1 -1
- package/dist/{dataset-C8xaLXdY.d.ts → dataset-CJjKqQfA.d.ts} +2 -8
- package/dist/dataset-CJjKqQfA.d.ts.map +1 -0
- package/dist/{default-registry-Bf8Woqmq.d.ts → default-registry-Di6HP6pG.d.ts} +7 -12
- package/dist/default-registry-Di6HP6pG.d.ts.map +1 -0
- package/dist/{analyze-runs-C30yljDJ.js → define-agent-eval-iqjT--ZZ.js} +542 -130
- package/dist/define-agent-eval-iqjT--ZZ.js.map +1 -0
- package/dist/define-agent-eval-sH24zBfM.d.ts +388 -0
- package/dist/define-agent-eval-sH24zBfM.d.ts.map +1 -0
- package/dist/descriptive-B5MwKfbf.js +144 -0
- package/dist/descriptive-B5MwKfbf.js.map +1 -0
- package/dist/{dspy-rlm-engine-DbTk4JdR.js → dspy-rlm-engine-B_qhSc21.js} +4 -4
- package/dist/{dspy-rlm-engine-DbTk4JdR.js.map → dspy-rlm-engine-B_qhSc21.js.map} +1 -1
- package/dist/effect-sizes-DiH8MGOH.js +82 -0
- package/dist/effect-sizes-DiH8MGOH.js.map +1 -0
- package/dist/{engine-3hL-XqwJ.d.ts → engine-otFpE2gF.d.ts} +10 -38
- package/dist/engine-otFpE2gF.d.ts.map +1 -0
- package/dist/{errors-CKPfb2aH.d.ts → errors-DEE6u6ot.d.ts} +2 -14
- package/dist/{errors-CKPfb2aH.d.ts.map → errors-DEE6u6ot.d.ts.map} +1 -1
- package/dist/{errors-D-LKuDhb.js → errors-Dngq5h35.js} +2 -8
- package/dist/{errors-D-LKuDhb.js.map → errors-Dngq5h35.js.map} +1 -1
- package/dist/{eval-campaign-B_7wcnav.js → eval-campaign-UB-usSQ2.js} +6 -6
- package/dist/{eval-campaign-B_7wcnav.js.map → eval-campaign-UB-usSQ2.js.map} +1 -1
- package/dist/{exact-types-DSFFpLLI.d.ts → exact-types-BH1twmAJ.d.ts} +2 -2
- package/dist/{exact-types-DSFFpLLI.d.ts.map → exact-types-BH1twmAJ.d.ts.map} +1 -1
- package/dist/experiment/index.d.ts +9 -6
- package/dist/experiment/index.d.ts.map +1 -1
- package/dist/experiment/index.js +11 -7
- package/dist/experiment/index.js.map +1 -1
- package/dist/experiment-tracker-C29gXM4B.js +269 -0
- package/dist/experiment-tracker-C29gXM4B.js.map +1 -0
- package/dist/{experiment-tracker-IMntXr6J.d.ts → experiment-tracker-DWHZBAYL.d.ts} +77 -121
- package/dist/experiment-tracker-DWHZBAYL.d.ts.map +1 -0
- package/dist/{external-optimizer-contracts-CZuJNcT5.d.ts → external-optimizer-contracts-lixrOZdX.d.ts} +3 -3
- package/dist/{external-optimizer-contracts-CZuJNcT5.d.ts.map → external-optimizer-contracts-lixrOZdX.d.ts.map} +1 -1
- package/dist/external-optimizer-process-BTiNB-RH.js +301 -0
- package/dist/external-optimizer-process-BTiNB-RH.js.map +1 -0
- package/dist/{single-run-lock-DFWHEB09.js → external-optimizer-subprocess-DrJ9hR8u.js} +144 -438
- package/dist/external-optimizer-subprocess-DrJ9hR8u.js.map +1 -0
- package/dist/extract-usage-BrQ8mCLX.js +155 -0
- package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
- package/dist/{failure-cluster-CqcvCcdR.d.ts → failure-cluster-BLURuWG4.d.ts} +2 -3
- package/dist/failure-cluster-BLURuWG4.d.ts.map +1 -0
- package/dist/{feedback-trajectory-CacHpxsp.d.ts → feedback-trajectory-juOozjAc.d.ts} +5 -36
- package/dist/feedback-trajectory-juOozjAc.d.ts.map +1 -0
- package/dist/fuzz.d.ts +2 -2
- package/dist/fuzz.js +2 -2
- package/dist/hosted/index.d.ts +3 -3
- package/dist/{index-CvSN3IG1.d.ts → index-B8Ui1mr1.d.ts} +2 -2
- package/dist/{index-CvSN3IG1.d.ts.map → index-B8Ui1mr1.d.ts.map} +1 -1
- package/dist/{skill-usage-CO9OLRBx.d.ts → index-BWDrSVfw.d.ts} +11 -136
- package/dist/index-BWDrSVfw.d.ts.map +1 -0
- package/dist/index-Ba3YrbAL.d.ts +1 -0
- package/dist/{index-BnEuDAK2.d.ts → index-COQYtuRF.d.ts} +3 -3
- package/dist/{index-BnEuDAK2.d.ts.map → index-COQYtuRF.d.ts.map} +1 -1
- package/dist/{index-C3ssXVLv.d.ts → index-CvjYbU0D.d.ts} +2 -2
- package/dist/{index-C3ssXVLv.d.ts.map → index-CvjYbU0D.d.ts.map} +1 -1
- package/dist/{index-DPPGNJ_R.d.ts → index-DSmEylT9.d.ts} +17 -115
- package/dist/index-DSmEylT9.d.ts.map +1 -0
- package/dist/index.d.ts +2397 -5308
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +5914 -10496
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-CgX_s0Ez.d.ts → insight-report-CRi-Ufrj.d.ts} +4 -4
- package/dist/{insight-report-CgX_s0Ez.d.ts.map → insight-report-CRi-Ufrj.d.ts.map} +1 -1
- package/dist/{integrity-MLzHOfV9.js → integrity-Cy9WHAtb.js} +2 -2
- package/dist/{integrity-MLzHOfV9.js.map → integrity-Cy9WHAtb.js.map} +1 -1
- package/dist/{integrity-DY6tIbl0.js → integrity-DysDBWDu.js} +2 -2
- package/dist/{integrity-DY6tIbl0.js.map → integrity-DysDBWDu.js.map} +1 -1
- package/dist/{integrity-BuqEKu-x.d.ts → integrity-OrcI9Nau.d.ts} +3 -3
- package/dist/{integrity-BuqEKu-x.d.ts.map → integrity-OrcI9Nau.d.ts.map} +1 -1
- package/dist/internal-BDHPCnjk.js +230 -0
- package/dist/internal-BDHPCnjk.js.map +1 -0
- package/dist/judge-calibration-C5CbMYce.d.ts +117 -0
- package/dist/judge-calibration-C5CbMYce.d.ts.map +1 -0
- package/dist/judge-calibration-DZkWrm5H.js +317 -0
- package/dist/judge-calibration-DZkWrm5H.js.map +1 -0
- package/dist/{kind-factory-B8-r8-y8.js → kind-factory-CPmSd58s.js} +208 -208
- package/dist/{kind-factory-B8-r8-y8.js.map → kind-factory-CPmSd58s.js.map} +1 -1
- package/dist/ledger-core/index.d.ts +1 -1
- package/dist/ledger-core/index.js +2 -2
- package/dist/{ledger-core-DXZIqu17.js → ledger-core-BmZt19oQ.js} +110 -110
- package/dist/{ledger-core-DXZIqu17.js.map → ledger-core-BmZt19oQ.js.map} +1 -1
- package/dist/{llm-client-DzvMUsS_.js → llm-client-d0-2TT1g.js} +4 -64
- package/dist/{llm-client-DzvMUsS_.js.map → llm-client-d0-2TT1g.js.map} +1 -1
- package/dist/{skillopt-optimization-method-CQdVeM8k.js → llm-judge-dZ8P6nGI.js} +3349 -5401
- package/dist/llm-judge-dZ8P6nGI.js.map +1 -0
- package/dist/matrix/index.d.ts +2 -2
- package/dist/meta-eval/index.d.ts +3 -3
- package/dist/meta-eval/index.js +3 -3
- package/dist/{metrics-C9YY1OcL.js → metrics-Cl0L1KUy.js} +2 -108
- package/dist/{metrics-C9YY1OcL.js.map → metrics-Cl0L1KUy.js.map} +1 -1
- package/dist/{mint-B2O60ACG.js → mint-Dj9Ww_3I.js} +4 -4
- package/dist/{mint-B2O60ACG.js.map → mint-Dj9Ww_3I.js.map} +1 -1
- package/dist/{multi-layer-verifier-DnAqwl0h.d.ts → multi-layer-verifier-DIguZc8Z.d.ts} +3 -3
- package/dist/{multi-layer-verifier-DnAqwl0h.d.ts.map → multi-layer-verifier-DIguZc8Z.d.ts.map} +1 -1
- package/dist/multiplicity-DIWHvysC.d.ts +43 -0
- package/dist/multiplicity-DIWHvysC.d.ts.map +1 -0
- package/dist/multishot/index.d.ts +3 -3
- package/dist/multishot/index.js +1 -1
- package/dist/openapi.json +1 -1
- package/dist/{opencode-sqlite-8r6WUfHc.js → opencode-sqlite-DJWAXLms.js} +2 -2
- package/dist/{opencode-sqlite-8r6WUfHc.js.map → opencode-sqlite-DJWAXLms.js.map} +1 -1
- package/dist/package-version-D7lQHt_-.js +34 -0
- package/dist/package-version-D7lQHt_-.js.map +1 -0
- package/dist/paired-arms-D-XRF_fy.js +1045 -0
- package/dist/paired-arms-D-XRF_fy.js.map +1 -0
- package/dist/paired-promotion-decision-CGzg0cI_.d.ts +251 -0
- package/dist/paired-promotion-decision-CGzg0cI_.d.ts.map +1 -0
- package/dist/paired-tests-BHIhYVdu.js +213 -0
- package/dist/paired-tests-BHIhYVdu.js.map +1 -0
- package/dist/pareto-BqNW3LJR.d.ts +117 -0
- package/dist/pareto-BqNW3LJR.d.ts.map +1 -0
- package/dist/pipelines/index.d.ts +3 -64
- package/dist/pipelines/index.d.ts.map +1 -1
- package/dist/pipelines/index.js +4 -284
- package/dist/pipelines/index.js.map +1 -1
- package/dist/power-and-mde-CHIrXJll.js +195 -0
- package/dist/power-and-mde-CHIrXJll.js.map +1 -0
- package/dist/{promotion-policy-CrLrmys8.js → power-preflight-DEw-uC7q.js} +4 -184
- package/dist/power-preflight-DEw-uC7q.js.map +1 -0
- package/dist/pre-registration-CZwSFQS4.d.ts +577 -0
- package/dist/pre-registration-CZwSFQS4.d.ts.map +1 -0
- package/dist/{prime-protocol-BfSalTfR.js → prime-protocol-6tZTVsWm.js} +72 -21
- package/dist/prime-protocol-6tZTVsWm.js.map +1 -0
- package/dist/produced-state-DU79a81m.js +586 -0
- package/dist/produced-state-DU79a81m.js.map +1 -0
- package/dist/profile-cell.d.ts +1 -1
- package/dist/profile-cell.js +1 -1
- package/dist/promotion-policy-D0nPhkSy.d.ts +134 -0
- package/dist/promotion-policy-D0nPhkSy.d.ts.map +1 -0
- package/dist/promotion-policy-xzA40Evo.js +186 -0
- package/dist/promotion-policy-xzA40Evo.js.map +1 -0
- package/dist/{skillopt-optimization-method-USDKhxSA.d.ts → provenance-CCdxgLDT.d.ts} +57 -478
- package/dist/provenance-CCdxgLDT.d.ts.map +1 -0
- package/dist/registry-oJeeI4-a.d.ts +178 -0
- package/dist/registry-oJeeI4-a.d.ts.map +1 -0
- package/dist/{release-report-DA2BCu5a.d.ts → release-confidence-BFRE5WSp.d.ts} +180 -112
- package/dist/release-confidence-BFRE5WSp.d.ts.map +1 -0
- package/dist/{release-report-BUYmoKo2.js → release-confidence-CxDuiAev.js} +125 -217
- package/dist/release-confidence-CxDuiAev.js.map +1 -0
- package/dist/reporting.d.ts +6 -5
- package/dist/reporting.js +7 -5
- package/dist/{researcher-BchpD55R.d.ts → researcher-DaN4GST-.d.ts} +7 -40
- package/dist/{researcher-BchpD55R.d.ts.map → researcher-DaN4GST-.d.ts.map} +1 -1
- package/dist/{reward-hacking-BDToousL.js → reward-hacking-DNgjilrV.js} +3 -3
- package/dist/reward-hacking-DNgjilrV.js.map +1 -0
- package/dist/{reward-hacking-BI0OMAlo.d.ts → reward-hacking-DSSTuI9r.d.ts} +5 -5
- package/dist/{reward-hacking-BI0OMAlo.d.ts.map → reward-hacking-DSSTuI9r.d.ts.map} +1 -1
- package/dist/rl.d.ts +7 -7
- package/dist/rl.js +11 -10
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +2 -2
- package/dist/rollout/index.js +4 -4
- package/dist/{rollout-DRrksrcV.js → rollout-BWtw0I_6.js} +3 -3
- package/dist/{rollout-DRrksrcV.js.map → rollout-BWtw0I_6.js.map} +1 -1
- package/dist/{rubric-predictive-validity-DvkjPYCe.d.ts → rubric-predictive-validity-BgxtKe4G.d.ts} +2 -2
- package/dist/{rubric-predictive-validity-DvkjPYCe.d.ts.map → rubric-predictive-validity-BgxtKe4G.d.ts.map} +1 -1
- package/dist/{rubric-predictive-validity-BRR632r1.js → rubric-predictive-validity-Cwwyd7ah.js} +2 -2
- package/dist/{rubric-predictive-validity-BRR632r1.js.map → rubric-predictive-validity-Cwwyd7ah.js.map} +1 -1
- package/dist/{run-record-BmSPWXJR.js → run-record-BvHPVS-i.js} +3 -3
- package/dist/{run-record-BmSPWXJR.js.map → run-record-BvHPVS-i.js.map} +1 -1
- package/dist/{run-record-CF4Dwpxr.d.ts → run-record-CKiihE6f.d.ts} +4 -4
- package/dist/{run-record-CF4Dwpxr.d.ts.map → run-record-CKiihE6f.d.ts.map} +1 -1
- package/dist/{proposal-findings-2GIUo1et.js → run-score-lDzV0X8j.js} +31 -31
- package/dist/run-score-lDzV0X8j.js.map +1 -0
- package/dist/sandbox-harness-BlSOu4LX.d.ts +70 -0
- package/dist/sandbox-harness-BlSOu4LX.d.ts.map +1 -0
- package/dist/{schema-Cef2cFmb2.d.ts → schema-Cef2cFmb.d.ts} +1 -1
- package/dist/schema-Cef2cFmb.d.ts.map +1 -0
- package/dist/semantic-concept-judge-BiJxScqe.js +406 -0
- package/dist/semantic-concept-judge-BiJxScqe.js.map +1 -0
- package/dist/{sequential-D-BLJBKU.js → sequential-C458DXNf.js} +4 -3
- package/dist/{sequential-D-BLJBKU.js.map → sequential-C458DXNf.js.map} +1 -1
- package/dist/sequential-eprocess-CbUt2htw.js +83 -0
- package/dist/sequential-eprocess-CbUt2htw.js.map +1 -0
- package/dist/series-convergence-D1cL1f-4.d.ts +129 -0
- package/dist/series-convergence-D1cL1f-4.d.ts.map +1 -0
- package/dist/{server-iu0ede49.js → server-ulsOdrTI.js} +5 -21
- package/dist/server-ulsOdrTI.js.map +1 -0
- package/dist/skillopt-optimization-method-C5cotF4E.d.ts +427 -0
- package/dist/skillopt-optimization-method-C5cotF4E.d.ts.map +1 -0
- package/dist/skillopt-optimization-method-jjdnc3YK.js +1999 -0
- package/dist/skillopt-optimization-method-jjdnc3YK.js.map +1 -0
- package/dist/{statistical-heldout-TQ-4CYiN.d.ts → statistical-heldout-DoFF5KJX.d.ts} +5 -278
- package/dist/statistical-heldout-DoFF5KJX.d.ts.map +1 -0
- package/dist/{store-otlp-Dw8PPIlL.js → store-otlp-CsptLYpN.js} +3 -3
- package/dist/{store-otlp-Dw8PPIlL.js.map → store-otlp-CsptLYpN.js.map} +1 -1
- package/dist/store-tool-spans-Cq9mFd-q.js +667 -0
- package/dist/store-tool-spans-Cq9mFd-q.js.map +1 -0
- package/dist/store-tool-spans-DbDLOBOb.d.ts +342 -0
- package/dist/store-tool-spans-DbDLOBOb.d.ts.map +1 -0
- package/dist/student-t-CvBq2mve.js +38 -0
- package/dist/student-t-CvBq2mve.js.map +1 -0
- package/dist/{summary-report-Lf-5I7xh.js → summary-report-Blysd6Z2.js} +6 -3
- package/dist/{summary-report-Lf-5I7xh.js.map → summary-report-Blysd6Z2.js.map} +1 -1
- package/dist/{summary-report-BNR7DWTj.d.ts → summary-report-CaL-Hnxt.d.ts} +5 -5
- package/dist/{summary-report-BNR7DWTj.d.ts.map → summary-report-CaL-Hnxt.d.ts.map} +1 -1
- package/dist/supervisor-run/index.d.ts +391 -3
- package/dist/supervisor-run/index.d.ts.map +1 -0
- package/dist/supervisor-run/index.js +1689 -2
- package/dist/{supervisor-run-D_sokXcO.js.map → supervisor-run/index.js.map} +1 -1
- package/dist/{extract-usage-CdZdoj1s.js → task-failure-attributes-CpQ4y5RD.js} +5 -157
- package/dist/task-failure-attributes-CpQ4y5RD.js.map +1 -0
- package/dist/{tool-groups-CZPGGlHf.d.ts → tool-groups-Cteb03Ps.d.ts} +3 -3
- package/dist/tool-groups-Cteb03Ps.d.ts.map +1 -0
- package/dist/tool-waste-BDdBZG1F.js +803 -0
- package/dist/tool-waste-BDdBZG1F.js.map +1 -0
- package/dist/tool-waste-DjRDEsuI.d.ts +128 -0
- package/dist/tool-waste-DjRDEsuI.d.ts.map +1 -0
- package/dist/trace-repair/index.d.ts +14 -5
- package/dist/trace-repair/index.d.ts.map +1 -1
- package/dist/trace-repair/index.js +35 -7
- package/dist/trace-repair/index.js.map +1 -1
- package/dist/traces.d.ts +406 -7
- package/dist/traces.d.ts.map +1 -0
- package/dist/traces.js +1011 -10
- package/dist/traces.js.map +1 -0
- package/dist/trajectory-replay/index.d.ts +16 -3
- package/dist/trajectory-replay/index.d.ts.map +1 -1
- package/dist/trajectory-replay/index.js +52 -5
- package/dist/trajectory-replay/index.js.map +1 -1
- package/dist/types-BEPZc6eo.d.ts +93 -0
- package/dist/types-BEPZc6eo.d.ts.map +1 -0
- package/dist/{usage-receipt-t7vAzCRQ.js → types-BI4fT3HN.js} +67 -67
- package/dist/types-BI4fT3HN.js.map +1 -0
- package/dist/{types-BnjdJ70P.d.ts → types-BZ59Ahr8.d.ts} +5 -5
- package/dist/{types-BnjdJ70P.d.ts.map → types-BZ59Ahr8.d.ts.map} +1 -1
- package/dist/{types-D216SgwM.d.ts → types-Cx3YUh2r.d.ts} +4 -241
- package/dist/types-Cx3YUh2r.d.ts.map +1 -0
- package/dist/{types-BvDKaULh.d.ts → types-DLQx4mKU.d.ts} +4 -4
- package/dist/{types-BvDKaULh.d.ts.map → types-DLQx4mKU.d.ts.map} +1 -1
- package/dist/{types-D4mog56g.d.ts → types-yLK8gXE9.d.ts} +2 -2
- package/dist/{types-D4mog56g.d.ts.map → types-yLK8gXE9.d.ts.map} +1 -1
- package/dist/verdict-BndeTAh_.js +61 -0
- package/dist/verdict-BndeTAh_.js.map +1 -0
- package/dist/verdict-E4eRNf7-.d.ts +392 -0
- package/dist/verdict-E4eRNf7-.d.ts.map +1 -0
- package/dist/{verdict-cache-BCcOh0kF.js → verdict-cache-mZf5FEiY.js} +3 -55
- package/dist/{verdict-cache-BCcOh0kF.js.map → verdict-cache-mZf5FEiY.js.map} +1 -1
- package/dist/wire/index.d.ts +3 -3
- package/dist/wire/index.d.ts.map +1 -1
- package/dist/wire/index.js +1 -1
- package/docs/charter.md +3 -3
- package/docs/control-runtime.md +3 -42
- package/docs/experiment.md +0 -1
- package/docs/feature-guide.md +2 -2
- package/docs/trace-repair-grader.md +1 -0
- package/docs/trajectory-replay.md +1 -0
- package/docs/verdicts.md +43 -0
- package/docs/verification-strategies.md +3 -2
- package/package.json +6 -11
- package/dist/agent-profile-DPi7IZg7.d.ts.map +0 -1
- package/dist/analyze-runs-C30yljDJ.js.map +0 -1
- package/dist/baseline-CavEbRyH.d.ts +0 -136
- package/dist/baseline-CavEbRyH.d.ts.map +0 -1
- package/dist/benchmark-command-BteMFN62.js.map +0 -1
- package/dist/benchmarks-Dzs8CKb1.js +0 -755
- package/dist/benchmarks-Dzs8CKb1.js.map +0 -1
- package/dist/campaign-C2TTzQII.js.map +0 -1
- package/dist/completion-verifier-DJA5BhPb.d.ts +0 -414
- package/dist/completion-verifier-DJA5BhPb.d.ts.map +0 -1
- package/dist/control.d.ts +0 -3
- package/dist/control.js +0 -2
- package/dist/dataset-C8xaLXdY.d.ts.map +0 -1
- package/dist/default-registry-Bf8Woqmq.d.ts.map +0 -1
- package/dist/default-registry-BmktKy8r.js.map +0 -1
- package/dist/engine-3hL-XqwJ.d.ts.map +0 -1
- package/dist/experiment-tracker-CnRICnMl.js +0 -500
- package/dist/experiment-tracker-CnRICnMl.js.map +0 -1
- package/dist/experiment-tracker-IMntXr6J.d.ts.map +0 -1
- package/dist/extract-usage-CdZdoj1s.js.map +0 -1
- package/dist/failure-cluster-CqcvCcdR.d.ts.map +0 -1
- package/dist/feedback-trajectory-CacHpxsp.d.ts.map +0 -1
- package/dist/index-BZ3-y4YL.d.ts +0 -391
- package/dist/index-BZ3-y4YL.d.ts.map +0 -1
- package/dist/index-CQTZ-4XN.d.ts.map +0 -1
- package/dist/index-DPPGNJ_R.d.ts.map +0 -1
- package/dist/index-YE4KdKbO2.d.ts +0 -335
- package/dist/index-YE4KdKbO2.d.ts.map +0 -1
- package/dist/paired-arms-iZ08VFMN.js +0 -260
- package/dist/paired-arms-iZ08VFMN.js.map +0 -1
- package/dist/paired-promotion-decision-B6zJ3gYM.d.ts +0 -114
- package/dist/paired-promotion-decision-B6zJ3gYM.d.ts.map +0 -1
- package/dist/prime-protocol-BfSalTfR.js.map +0 -1
- package/dist/promotion-policy-Ckjhzg_4.d.ts +0 -289
- package/dist/promotion-policy-Ckjhzg_4.d.ts.map +0 -1
- package/dist/promotion-policy-CrLrmys8.js.map +0 -1
- package/dist/proposal-findings-2GIUo1et.js.map +0 -1
- package/dist/propose-review-control-dSNPjFUH.js +0 -1458
- package/dist/propose-review-control-dSNPjFUH.js.map +0 -1
- package/dist/release-report-BUYmoKo2.js.map +0 -1
- package/dist/release-report-DA2BCu5a.d.ts.map +0 -1
- package/dist/replay-CohS93nE.js +0 -1859
- package/dist/replay-CohS93nE.js.map +0 -1
- package/dist/replay-DbhZ4Ked.d.ts +0 -834
- package/dist/replay-DbhZ4Ked.d.ts.map +0 -1
- package/dist/reward-hacking-BDToousL.js.map +0 -1
- package/dist/run-evidence-Dhi3C81V.d.ts +0 -225
- package/dist/run-evidence-Dhi3C81V.d.ts.map +0 -1
- package/dist/schema-Cef2cFmb2.d.ts.map +0 -1
- package/dist/semantic-concept-judge-D1z-KepS.js +0 -767
- package/dist/semantic-concept-judge-D1z-KepS.js.map +0 -1
- package/dist/series-convergence-ofsqPWhs.d.ts +0 -35
- package/dist/series-convergence-ofsqPWhs.d.ts.map +0 -1
- package/dist/server-iu0ede49.js.map +0 -1
- package/dist/single-run-lock-DFWHEB09.js.map +0 -1
- package/dist/skill-usage-CO9OLRBx.d.ts.map +0 -1
- package/dist/skillopt-optimization-method-CQdVeM8k.js.map +0 -1
- package/dist/skillopt-optimization-method-USDKhxSA.d.ts.map +0 -1
- package/dist/statistical-heldout-TQ-4CYiN.d.ts.map +0 -1
- package/dist/statistics-ByxzSiOM.js +0 -2212
- package/dist/statistics-ByxzSiOM.js.map +0 -1
- package/dist/statistics-D6Uebe_4.d.ts +0 -968
- package/dist/statistics-D6Uebe_4.d.ts.map +0 -1
- package/dist/supervisor-run-D_sokXcO.js +0 -1690
- package/dist/test-graded-scenario-D1TaI2va.d.ts +0 -141
- package/dist/test-graded-scenario-D1TaI2va.d.ts.map +0 -1
- package/dist/test-graded-scenario-JHcKQNpq.js +0 -318
- package/dist/test-graded-scenario-JHcKQNpq.js.map +0 -1
- package/dist/tool-groups-CZPGGlHf.d.ts.map +0 -1
- package/dist/tool-use-metrics-DEGMKycK.js +0 -370
- package/dist/tool-use-metrics-DEGMKycK.js.map +0 -1
- package/dist/types-D216SgwM.d.ts.map +0 -1
- package/dist/usage-receipt-t7vAzCRQ.js.map +0 -1
- package/dist/verdict-DExhxfgR.d.ts +0 -201
- package/dist/verdict-DExhxfgR.d.ts.map +0 -1
|
@@ -0,0 +1,803 @@
|
|
|
1
|
+
import { s as ValidationError } from "./errors-Dngq5h35.js";
|
|
2
|
+
import { r as pearsonR } from "./descriptive-B5MwKfbf.js";
|
|
3
|
+
import { r as continuousAgreement } from "./judge-calibration-DZkWrm5H.js";
|
|
4
|
+
import { i as hasCapturedToolArgs, l as toolSpans, n as argHash, o as llmSpans, r as groupBy } from "./query-Di7eEQ79.js";
|
|
5
|
+
//#region src/statistics/agreement-irr.ts
|
|
6
|
+
/**
|
|
7
|
+
* Inter-rater reliability — Krippendorff's α under the squared-difference
|
|
8
|
+
* metric, pooled across dimensions.
|
|
9
|
+
*
|
|
10
|
+
* Each inner array is one judge's scores. Items are matched by position
|
|
11
|
+
* WITHIN a dimension: the k-th score a judge supplies carrying dimension
|
|
12
|
+
* `d` is item k of `d`, and the ratings compared against each other are
|
|
13
|
+
* the ones different judges gave to the same item. Every judge that scores
|
|
14
|
+
* a dimension at all must supply the same number of scores for it —
|
|
15
|
+
* ragged input cannot be aligned into items and throws rather than
|
|
16
|
+
* comparing mismatched items.
|
|
17
|
+
*
|
|
18
|
+
* α = 1 − D_observed / D_expected: D_observed averages the squared
|
|
19
|
+
* difference over within-item judge pairs, D_expected over every pair of
|
|
20
|
+
* ratings irrespective of item. α = 1 is perfect agreement, 0 is chance,
|
|
21
|
+
* negative is systematic disagreement.
|
|
22
|
+
*/
|
|
23
|
+
function interRaterReliability(judgeScores) {
|
|
24
|
+
if (judgeScores.length < 2) return 1;
|
|
25
|
+
const perDimension = /* @__PURE__ */ new Map();
|
|
26
|
+
for (let judgeIndex = 0; judgeIndex < judgeScores.length; judgeIndex++) for (const s of judgeScores[judgeIndex]) {
|
|
27
|
+
let byJudge = perDimension.get(s.dimension);
|
|
28
|
+
if (byJudge === void 0) {
|
|
29
|
+
byJudge = Array.from({ length: judgeScores.length }, () => []);
|
|
30
|
+
perDimension.set(s.dimension, byJudge);
|
|
31
|
+
}
|
|
32
|
+
byJudge[judgeIndex].push(s.score);
|
|
33
|
+
}
|
|
34
|
+
const allValues = [];
|
|
35
|
+
const pairDiffs = [];
|
|
36
|
+
for (const [dimension, byJudge] of perDimension) {
|
|
37
|
+
const scoring = byJudge.filter((scores) => scores.length > 0);
|
|
38
|
+
if (scoring.length < 2) continue;
|
|
39
|
+
const itemCount = scoring[0].length;
|
|
40
|
+
if (scoring.some((scores) => scores.length !== itemCount)) throw new ValidationError(`interRaterReliability: dimension '${dimension}' has judges supplying ${scoring.map((scores) => scores.length).join("/")} scores — items cannot be aligned`);
|
|
41
|
+
for (let item = 0; item < itemCount; item++) {
|
|
42
|
+
const ratings = scoring.map((scores) => scores[item]);
|
|
43
|
+
for (const v of ratings) allValues.push(v);
|
|
44
|
+
for (let i = 0; i < ratings.length; i++) for (let j = i + 1; j < ratings.length; j++) pairDiffs.push((ratings[i] - ratings[j]) ** 2);
|
|
45
|
+
}
|
|
46
|
+
}
|
|
47
|
+
if (pairDiffs.length === 0 || allValues.length < 2) return 1;
|
|
48
|
+
const observedDisagreement = pairDiffs.reduce((a, b) => a + b, 0) / pairDiffs.length;
|
|
49
|
+
let expectedDisagreement = 0;
|
|
50
|
+
let expectedCount = 0;
|
|
51
|
+
for (let i = 0; i < allValues.length; i++) for (let j = i + 1; j < allValues.length; j++) {
|
|
52
|
+
expectedDisagreement += (allValues[i] - allValues[j]) ** 2;
|
|
53
|
+
expectedCount++;
|
|
54
|
+
}
|
|
55
|
+
expectedDisagreement = expectedCount > 0 ? expectedDisagreement / expectedCount : 0;
|
|
56
|
+
if (expectedDisagreement === 0) return 1;
|
|
57
|
+
return 1 - observedDisagreement / expectedDisagreement;
|
|
58
|
+
}
|
|
59
|
+
/**
|
|
60
|
+
* Corpus-wide inter-rater agreement across N items × M judges × D dimensions.
|
|
61
|
+
*
|
|
62
|
+
* For each dimension, builds the [n_items][n_judges] matrix of scores
|
|
63
|
+
* (keeping only items every judge rated on that dimension), then runs
|
|
64
|
+
* `continuousAgreement` to get ICC(2,1), κ_w, Pearson, Spearman, and
|
|
65
|
+
* bootstrap CIs. Reports a pooled mean across dimensions as a single
|
|
66
|
+
* "is this judge panel reliable on this corpus?" number.
|
|
67
|
+
*
|
|
68
|
+
* Fail-loud contract:
|
|
69
|
+
* - Empty input throws.
|
|
70
|
+
* - Fewer than 2 judges or fewer than 2 items per dimension throws.
|
|
71
|
+
* - A judge present in some dimensions but with zero scored items on
|
|
72
|
+
* another dimension throws (would silently shrink the matrix).
|
|
73
|
+
* - Duplicate (itemId, judgeName, dimension) records throw.
|
|
74
|
+
*/
|
|
75
|
+
function corpusInterRaterAgreement(records, opts = {}) {
|
|
76
|
+
if (records.length === 0) throw new ValidationError("corpusInterRaterAgreement: no score records supplied");
|
|
77
|
+
const judgesSeen = /* @__PURE__ */ new Set();
|
|
78
|
+
const dimsSeen = /* @__PURE__ */ new Set();
|
|
79
|
+
const grid = /* @__PURE__ */ new Map();
|
|
80
|
+
for (const r of records) {
|
|
81
|
+
if (!Number.isFinite(r.score)) throw new ValidationError(`corpusInterRaterAgreement: non-finite score for (item=${r.itemId}, judge=${r.judgeName}, dim=${r.dimension})`);
|
|
82
|
+
judgesSeen.add(r.judgeName);
|
|
83
|
+
dimsSeen.add(r.dimension);
|
|
84
|
+
const byJudge = grid.get(r.dimension) ?? /* @__PURE__ */ new Map();
|
|
85
|
+
const byItem = byJudge.get(r.judgeName) ?? /* @__PURE__ */ new Map();
|
|
86
|
+
if (byItem.has(r.itemId)) throw new ValidationError(`corpusInterRaterAgreement: duplicate record for (item=${r.itemId}, judge=${r.judgeName}, dim=${r.dimension})`);
|
|
87
|
+
byItem.set(r.itemId, r.score);
|
|
88
|
+
byJudge.set(r.judgeName, byItem);
|
|
89
|
+
grid.set(r.dimension, byJudge);
|
|
90
|
+
}
|
|
91
|
+
const targetDims = opts.dimensions ?? [...dimsSeen].sort();
|
|
92
|
+
for (const d of targetDims) if (!dimsSeen.has(d)) throw new ValidationError(`corpusInterRaterAgreement: dimension '${d}' was requested but no records carry it`);
|
|
93
|
+
const targetJudges = opts.judges ? [...opts.judges] : [...judgesSeen].sort();
|
|
94
|
+
for (const j of targetJudges) if (!judgesSeen.has(j)) throw new ValidationError(`corpusInterRaterAgreement: judge '${j}' was requested but produced no records`);
|
|
95
|
+
if (targetJudges.length < 2) throw new ValidationError(`corpusInterRaterAgreement: need ≥2 judges, got ${targetJudges.length}`);
|
|
96
|
+
const perDimension = [];
|
|
97
|
+
const iccs = [];
|
|
98
|
+
const kappas = [];
|
|
99
|
+
for (const dim of targetDims) {
|
|
100
|
+
const byJudge = grid.get(dim);
|
|
101
|
+
const judgeItemCounts = {};
|
|
102
|
+
for (const j of targetJudges) judgeItemCounts[j] = byJudge.get(j)?.size ?? 0;
|
|
103
|
+
const emptyJudges = targetJudges.filter((j) => judgeItemCounts[j] === 0);
|
|
104
|
+
if (emptyJudges.length > 0) throw new ValidationError(`corpusInterRaterAgreement: dimension '${dim}' has no scores from judge(s) ${emptyJudges.join(", ")} (counts: ${JSON.stringify(judgeItemCounts)})`);
|
|
105
|
+
let commonItems = null;
|
|
106
|
+
for (const j of targetJudges) {
|
|
107
|
+
const ids = new Set(byJudge.get(j).keys());
|
|
108
|
+
if (commonItems === null) commonItems = ids;
|
|
109
|
+
else commonItems = new Set([...commonItems].filter((x) => ids.has(x)));
|
|
110
|
+
}
|
|
111
|
+
const sortedItems = [...commonItems ?? /* @__PURE__ */ new Set()].sort();
|
|
112
|
+
if (sortedItems.length < 2) throw new ValidationError(`corpusInterRaterAgreement: dimension '${dim}' has ${sortedItems.length} item(s) rated by all ${targetJudges.length} judges (need ≥2)`);
|
|
113
|
+
const agreement = continuousAgreement(sortedItems.map((itemId) => targetJudges.map((j) => byJudge.get(j).get(itemId))), opts);
|
|
114
|
+
perDimension.push({
|
|
115
|
+
...agreement,
|
|
116
|
+
dimension: dim,
|
|
117
|
+
itemIds: sortedItems,
|
|
118
|
+
judgeIds: [...targetJudges]
|
|
119
|
+
});
|
|
120
|
+
if (Number.isFinite(agreement.icc)) iccs.push(agreement.icc);
|
|
121
|
+
if (Number.isFinite(agreement.weightedKappa)) kappas.push(agreement.weightedKappa);
|
|
122
|
+
}
|
|
123
|
+
const mean = (xs) => xs.length === 0 ? NaN : xs.reduce((a, b) => a + b, 0) / xs.length;
|
|
124
|
+
return {
|
|
125
|
+
perDimension,
|
|
126
|
+
overallIcc: mean(iccs),
|
|
127
|
+
overallWeightedKappa: mean(kappas),
|
|
128
|
+
dimensions: targetDims,
|
|
129
|
+
judgeIds: targetJudges
|
|
130
|
+
};
|
|
131
|
+
}
|
|
132
|
+
/**
|
|
133
|
+
* Convenience adapter for `JudgeScore[]` data keyed externally by item.
|
|
134
|
+
*
|
|
135
|
+
* Use when you have per-item arrays of `JudgeScore[]` (e.g. one
|
|
136
|
+
* `ScenarioResult.judgeScores` per scenario) and want corpus-wide
|
|
137
|
+
* agreement without manually flattening. `itemId` must be unique per
|
|
138
|
+
* row of `itemsScores`.
|
|
139
|
+
*/
|
|
140
|
+
function corpusInterRaterAgreementFromJudgeScores(itemsScores, opts = {}) {
|
|
141
|
+
const records = [];
|
|
142
|
+
const seen = /* @__PURE__ */ new Set();
|
|
143
|
+
for (const { itemId, scores } of itemsScores) {
|
|
144
|
+
if (seen.has(itemId)) throw new ValidationError(`corpusInterRaterAgreementFromJudgeScores: duplicate itemId '${itemId}'`);
|
|
145
|
+
seen.add(itemId);
|
|
146
|
+
for (const s of scores) records.push({
|
|
147
|
+
itemId,
|
|
148
|
+
judgeName: s.judgeName,
|
|
149
|
+
dimension: s.dimension,
|
|
150
|
+
score: s.score
|
|
151
|
+
});
|
|
152
|
+
}
|
|
153
|
+
return corpusInterRaterAgreement(records, opts);
|
|
154
|
+
}
|
|
155
|
+
//#endregion
|
|
156
|
+
//#region src/failure-taxonomy.ts
|
|
157
|
+
const DEFAULT_RULES = [
|
|
158
|
+
{
|
|
159
|
+
id: "explicit-outcome",
|
|
160
|
+
match: ({ run }) => {
|
|
161
|
+
const fc = run.outcome?.failureClass;
|
|
162
|
+
if (fc && fc !== "unknown") return {
|
|
163
|
+
failureClass: fc,
|
|
164
|
+
reason: "outcome.failureClass set explicitly"
|
|
165
|
+
};
|
|
166
|
+
return null;
|
|
167
|
+
}
|
|
168
|
+
},
|
|
169
|
+
{
|
|
170
|
+
id: "knowledge-readiness-blocked",
|
|
171
|
+
match: ({ events }) => {
|
|
172
|
+
const event = events.find((e) => e.kind === "custom" && e.payload.kind === "readiness_scored" && e.payload.passed === false);
|
|
173
|
+
return event ? {
|
|
174
|
+
failureClass: "knowledge_readiness_blocked",
|
|
175
|
+
reason: "knowledge readiness report blocked execution",
|
|
176
|
+
triggerEventId: event.eventId
|
|
177
|
+
} : null;
|
|
178
|
+
}
|
|
179
|
+
},
|
|
180
|
+
{
|
|
181
|
+
id: "bad-integration-manifest",
|
|
182
|
+
match: ({ events }) => {
|
|
183
|
+
const event = events.find((e) => e.kind === "custom" && (e.payload.kind === "integration_manifest_validated" && e.payload.valid === false || e.payload.kind === "integration_invoke_failed" && e.payload.code === "manifest_invalid"));
|
|
184
|
+
return event ? {
|
|
185
|
+
failureClass: "bad_integration_manifest",
|
|
186
|
+
reason: "integration manifest validation failed before launch",
|
|
187
|
+
triggerEventId: event.eventId
|
|
188
|
+
} : null;
|
|
189
|
+
}
|
|
190
|
+
},
|
|
191
|
+
{
|
|
192
|
+
id: "missing-integration-connection",
|
|
193
|
+
match: ({ events }) => {
|
|
194
|
+
const event = events.find((e) => e.kind === "custom" && e.payload.kind === "integration_manifest_resolved" && hasResolutionStatus(e.payload, "missing_connection"));
|
|
195
|
+
return event ? {
|
|
196
|
+
failureClass: "missing_integration_connection",
|
|
197
|
+
reason: "required integration connection was missing",
|
|
198
|
+
triggerEventId: event.eventId
|
|
199
|
+
} : null;
|
|
200
|
+
}
|
|
201
|
+
},
|
|
202
|
+
{
|
|
203
|
+
id: "missing-integration-scope",
|
|
204
|
+
match: ({ events }) => {
|
|
205
|
+
const event = events.find((e) => e.kind === "custom" && (e.payload.kind === "integration_manifest_resolved" && hasMissingScopes(e.payload) || e.payload.kind === "integration_invoke_failed" && e.payload.code === "scope_denied"));
|
|
206
|
+
return event ? {
|
|
207
|
+
failureClass: "missing_integration_scope",
|
|
208
|
+
reason: "integration grant or connection lacks required scopes",
|
|
209
|
+
triggerEventId: event.eventId
|
|
210
|
+
} : null;
|
|
211
|
+
}
|
|
212
|
+
},
|
|
213
|
+
{
|
|
214
|
+
id: "integration-approval-required",
|
|
215
|
+
match: ({ events }) => {
|
|
216
|
+
const event = events.find((e) => e.kind === "custom" && (e.payload.kind === "integration_invoke" && e.payload.status === "approval_required" || e.payload.kind === "integration_invoke_failed" && e.payload.code === "approval_required" || e.payload.kind === "integration_approval_required"));
|
|
217
|
+
return event ? {
|
|
218
|
+
failureClass: "integration_approval_required",
|
|
219
|
+
reason: "integration write paused for user approval",
|
|
220
|
+
triggerEventId: event.eventId
|
|
221
|
+
} : null;
|
|
222
|
+
}
|
|
223
|
+
},
|
|
224
|
+
{
|
|
225
|
+
id: "integration-auth-expired",
|
|
226
|
+
match: ({ events }) => {
|
|
227
|
+
const event = events.find((e) => e.kind === "custom" && e.payload.kind === "integration_invoke_failed" && (e.payload.code === "auth_expired" || e.payload.code === "connection_not_active" || e.payload.code === "capability_expired" || e.payload.status === "expired"));
|
|
228
|
+
return event ? {
|
|
229
|
+
failureClass: "integration_auth_expired",
|
|
230
|
+
reason: "integration connection or capability expired",
|
|
231
|
+
triggerEventId: event.eventId
|
|
232
|
+
} : null;
|
|
233
|
+
}
|
|
234
|
+
},
|
|
235
|
+
{
|
|
236
|
+
id: "unsafe-integration-write-denied",
|
|
237
|
+
match: ({ events }) => {
|
|
238
|
+
const event = events.find((e) => e.kind === "custom" && e.payload.kind === "integration_invoke_failed" && (e.payload.code === "unsafe_write_denied" || e.payload.code === "policy_denied" || e.payload.code === "action_denied"));
|
|
239
|
+
return event ? {
|
|
240
|
+
failureClass: "unsafe_integration_write_denied",
|
|
241
|
+
reason: "integration write was denied by policy or capability scope",
|
|
242
|
+
triggerEventId: event.eventId
|
|
243
|
+
} : null;
|
|
244
|
+
}
|
|
245
|
+
},
|
|
246
|
+
{
|
|
247
|
+
id: "integration-provider-failure",
|
|
248
|
+
match: ({ events }) => {
|
|
249
|
+
const event = events.find((e) => e.kind === "custom" && e.payload.kind === "integration_invoke_failed" && ![
|
|
250
|
+
"scope_denied",
|
|
251
|
+
"approval_required",
|
|
252
|
+
"auth_expired",
|
|
253
|
+
"connection_not_active",
|
|
254
|
+
"capability_expired",
|
|
255
|
+
"unsafe_write_denied",
|
|
256
|
+
"policy_denied",
|
|
257
|
+
"action_denied",
|
|
258
|
+
"manifest_invalid"
|
|
259
|
+
].includes(String(e.payload.code)));
|
|
260
|
+
return event ? {
|
|
261
|
+
failureClass: "integration_provider_failure",
|
|
262
|
+
reason: "integration provider invocation failed",
|
|
263
|
+
triggerEventId: event.eventId
|
|
264
|
+
} : null;
|
|
265
|
+
}
|
|
266
|
+
},
|
|
267
|
+
{
|
|
268
|
+
id: "missing-credentials",
|
|
269
|
+
match: ({ events }) => {
|
|
270
|
+
const event = events.find((e) => e.kind === "custom" && e.payload.kind === "knowledge_gap" && e.payload.category === "credential_or_secret");
|
|
271
|
+
return event ? {
|
|
272
|
+
failureClass: "missing_credentials",
|
|
273
|
+
reason: "required credential or secret was missing",
|
|
274
|
+
triggerEventId: event.eventId
|
|
275
|
+
} : null;
|
|
276
|
+
}
|
|
277
|
+
},
|
|
278
|
+
{
|
|
279
|
+
id: "bad-retrieval",
|
|
280
|
+
match: ({ run, spans }) => {
|
|
281
|
+
if (run.outcome?.pass !== false) return null;
|
|
282
|
+
const retrieval = spans.find((s) => s.kind === "retrieval" && (s.hits.length === 0 || s.hits.every((hit) => hit.score <= 0)));
|
|
283
|
+
return retrieval ? {
|
|
284
|
+
failureClass: "bad_retrieval",
|
|
285
|
+
reason: "retrieval returned no useful hits for a failed run",
|
|
286
|
+
triggerSpanId: retrieval.spanId
|
|
287
|
+
} : null;
|
|
288
|
+
}
|
|
289
|
+
},
|
|
290
|
+
{
|
|
291
|
+
id: "insufficient-evidence",
|
|
292
|
+
match: ({ events }) => {
|
|
293
|
+
const event = events.find((e) => e.kind === "custom" && e.payload.kind === "knowledge_gap" && e.payload.reason === "insufficient_evidence");
|
|
294
|
+
return event ? {
|
|
295
|
+
failureClass: "insufficient_evidence",
|
|
296
|
+
reason: "task proceeded with insufficient supporting evidence",
|
|
297
|
+
triggerEventId: event.eventId
|
|
298
|
+
} : null;
|
|
299
|
+
}
|
|
300
|
+
},
|
|
301
|
+
{
|
|
302
|
+
id: "contradictory-evidence",
|
|
303
|
+
match: ({ events }) => {
|
|
304
|
+
const event = events.find((e) => e.kind === "custom" && e.payload.kind === "knowledge_gap" && e.payload.reason === "contradictory_evidence");
|
|
305
|
+
return event ? {
|
|
306
|
+
failureClass: "contradictory_evidence",
|
|
307
|
+
reason: "supporting evidence contradicted itself",
|
|
308
|
+
triggerEventId: event.eventId
|
|
309
|
+
} : null;
|
|
310
|
+
}
|
|
311
|
+
},
|
|
312
|
+
{
|
|
313
|
+
id: "budget-breach",
|
|
314
|
+
match: ({ events }) => {
|
|
315
|
+
const breach = events.find((e) => e.kind === "budget_breach");
|
|
316
|
+
return breach ? {
|
|
317
|
+
failureClass: "budget_exceeded",
|
|
318
|
+
reason: `budget breached on ${breach.payload.dimension ?? "unknown dimension"}`,
|
|
319
|
+
triggerEventId: breach.eventId
|
|
320
|
+
} : null;
|
|
321
|
+
}
|
|
322
|
+
},
|
|
323
|
+
{
|
|
324
|
+
id: "policy-violation",
|
|
325
|
+
match: ({ events }) => {
|
|
326
|
+
const e = events.find((x) => x.kind === "policy_violation");
|
|
327
|
+
return e ? {
|
|
328
|
+
failureClass: "policy_violation",
|
|
329
|
+
reason: "policy_violation event emitted",
|
|
330
|
+
triggerEventId: e.eventId
|
|
331
|
+
} : null;
|
|
332
|
+
}
|
|
333
|
+
},
|
|
334
|
+
{
|
|
335
|
+
id: "sandbox-failure",
|
|
336
|
+
match: ({ spans }) => {
|
|
337
|
+
const s = spans.find((x) => x.kind === "sandbox" && typeof x.exitCode === "number" && x.exitCode !== 0);
|
|
338
|
+
if (!s) return null;
|
|
339
|
+
return {
|
|
340
|
+
failureClass: "sandbox_failure",
|
|
341
|
+
reason: `sandbox exited ${s.exitCode}`,
|
|
342
|
+
triggerSpanId: s.spanId
|
|
343
|
+
};
|
|
344
|
+
}
|
|
345
|
+
},
|
|
346
|
+
{
|
|
347
|
+
id: "timeout",
|
|
348
|
+
match: ({ run, events }) => {
|
|
349
|
+
if (run.status !== "aborted") return null;
|
|
350
|
+
const hasTimeout = events.some((e) => e.kind === "error" && String(e.payload.reason ?? "").toLowerCase().includes("timeout"));
|
|
351
|
+
const note = (run.outcome?.notes ?? "").toLowerCase();
|
|
352
|
+
if (hasTimeout || note.includes("timeout") || note.includes("deadline")) return {
|
|
353
|
+
failureClass: "timeout",
|
|
354
|
+
reason: "timeout signal observed"
|
|
355
|
+
};
|
|
356
|
+
return null;
|
|
357
|
+
}
|
|
358
|
+
},
|
|
359
|
+
{
|
|
360
|
+
id: "tool-recovery-failure",
|
|
361
|
+
match: ({ spans }) => {
|
|
362
|
+
const tools = spans.filter((s) => s.kind === "tool");
|
|
363
|
+
const byTool = /* @__PURE__ */ new Map();
|
|
364
|
+
for (const t of tools) {
|
|
365
|
+
const name = t.toolName;
|
|
366
|
+
const arr = byTool.get(name) ?? [];
|
|
367
|
+
arr.push(t);
|
|
368
|
+
byTool.set(name, arr);
|
|
369
|
+
}
|
|
370
|
+
for (const [name, arr] of byTool) {
|
|
371
|
+
const errs = arr.filter((s) => s.status === "error");
|
|
372
|
+
if (errs.length >= 3 && errs.length === arr.length) return {
|
|
373
|
+
failureClass: "tool_recovery_failure",
|
|
374
|
+
reason: `${errs.length} consecutive errors on tool "${name}"`,
|
|
375
|
+
triggerSpanId: errs[errs.length - 1].spanId
|
|
376
|
+
};
|
|
377
|
+
}
|
|
378
|
+
return null;
|
|
379
|
+
}
|
|
380
|
+
},
|
|
381
|
+
{
|
|
382
|
+
id: "tool-selection-error",
|
|
383
|
+
match: ({ run, spans }) => {
|
|
384
|
+
if (run.outcome?.pass !== false) return null;
|
|
385
|
+
const hasToolsAvailable = spans.some((s) => s.kind === "agent" && s.attributes?.toolsAvailable !== void 0 && s.attributes?.toolsAvailable > 0);
|
|
386
|
+
const tools = spans.filter((s) => s.kind === "tool");
|
|
387
|
+
if (hasToolsAvailable && tools.length === 0) return {
|
|
388
|
+
failureClass: "tool_selection_error",
|
|
389
|
+
reason: "tools were available but none were called"
|
|
390
|
+
};
|
|
391
|
+
return null;
|
|
392
|
+
}
|
|
393
|
+
},
|
|
394
|
+
{
|
|
395
|
+
id: "format-drift",
|
|
396
|
+
match: ({ spans }) => {
|
|
397
|
+
const judge = spans.find((s) => s.kind === "judge" && s.dimension === "format" && s.score < .5);
|
|
398
|
+
return judge ? {
|
|
399
|
+
failureClass: "format_drift",
|
|
400
|
+
reason: "format judge scored below 0.5",
|
|
401
|
+
triggerSpanId: judge.spanId
|
|
402
|
+
} : null;
|
|
403
|
+
}
|
|
404
|
+
}
|
|
405
|
+
];
|
|
406
|
+
function hasResolutionStatus(payload, status) {
|
|
407
|
+
if (status === "missing_connection" && stringArray(payload.missingConnections).length > 0) return true;
|
|
408
|
+
return resolutionItems(payload).some((item) => item.status === status);
|
|
409
|
+
}
|
|
410
|
+
function hasMissingScopes(payload) {
|
|
411
|
+
if (stringArray(payload.missingScopes).length > 0) return true;
|
|
412
|
+
return resolutionItems(payload).some((item) => Array.isArray(item.missingScopes) && item.missingScopes.length > 0);
|
|
413
|
+
}
|
|
414
|
+
function resolutionItems(payload) {
|
|
415
|
+
return [
|
|
416
|
+
...records(payload.missing),
|
|
417
|
+
...records(payload.optionalMissing),
|
|
418
|
+
...records(payload.ready)
|
|
419
|
+
];
|
|
420
|
+
}
|
|
421
|
+
function records(value) {
|
|
422
|
+
if (!Array.isArray(value)) return [];
|
|
423
|
+
return value.filter((item) => Boolean(item) && typeof item === "object" && !Array.isArray(item));
|
|
424
|
+
}
|
|
425
|
+
function stringArray(value) {
|
|
426
|
+
return Array.isArray(value) ? value.filter((item) => typeof item === "string") : [];
|
|
427
|
+
}
|
|
428
|
+
/** Classify the failure mode of a run using an ordered rule list. */
|
|
429
|
+
function classifyFailure(ctx, rules = DEFAULT_RULES) {
|
|
430
|
+
if (ctx.run.outcome?.pass !== false && ctx.run.status === "completed") return {
|
|
431
|
+
failureClass: "success",
|
|
432
|
+
reason: "run completed with pass=true (or no explicit fail)"
|
|
433
|
+
};
|
|
434
|
+
for (const rule of rules) {
|
|
435
|
+
const hit = rule.match(ctx);
|
|
436
|
+
if (hit) return hit;
|
|
437
|
+
}
|
|
438
|
+
return {
|
|
439
|
+
failureClass: "unknown",
|
|
440
|
+
reason: "no rule matched; run failed for unclassified reason"
|
|
441
|
+
};
|
|
442
|
+
}
|
|
443
|
+
//#endregion
|
|
444
|
+
//#region src/pipelines/budget-breach.ts
|
|
445
|
+
async function budgetBreachView(store, options = {}) {
|
|
446
|
+
const runs = await store.listRuns({
|
|
447
|
+
scenarioId: options.scenarioId,
|
|
448
|
+
variantId: options.variantId
|
|
449
|
+
});
|
|
450
|
+
const findings = [];
|
|
451
|
+
const byDimension = {};
|
|
452
|
+
const byScenario = {};
|
|
453
|
+
const byVariant = {};
|
|
454
|
+
for (const run of runs) {
|
|
455
|
+
const entries = await store.budget(run.runId);
|
|
456
|
+
for (const e of entries) {
|
|
457
|
+
if (!e.breached) continue;
|
|
458
|
+
const excessRatio = e.limit > 0 ? e.consumed / e.limit : Infinity;
|
|
459
|
+
findings.push({
|
|
460
|
+
runId: run.runId,
|
|
461
|
+
scenarioId: run.scenarioId,
|
|
462
|
+
variantId: run.variantId,
|
|
463
|
+
dimension: e.dimension,
|
|
464
|
+
limit: e.limit,
|
|
465
|
+
consumed: e.consumed,
|
|
466
|
+
excessRatio,
|
|
467
|
+
timestamp: e.timestamp
|
|
468
|
+
});
|
|
469
|
+
byDimension[e.dimension] = (byDimension[e.dimension] ?? 0) + 1;
|
|
470
|
+
byScenario[run.scenarioId] = (byScenario[run.scenarioId] ?? 0) + 1;
|
|
471
|
+
if (run.variantId) byVariant[run.variantId] = (byVariant[run.variantId] ?? 0) + 1;
|
|
472
|
+
}
|
|
473
|
+
}
|
|
474
|
+
const breachedRuns = new Set(findings.map((f) => f.runId));
|
|
475
|
+
return {
|
|
476
|
+
findings,
|
|
477
|
+
byDimension,
|
|
478
|
+
byScenario,
|
|
479
|
+
byVariant,
|
|
480
|
+
totalRuns: runs.length,
|
|
481
|
+
breachedRunRatio: runs.length > 0 ? breachedRuns.size / runs.length : 0
|
|
482
|
+
};
|
|
483
|
+
}
|
|
484
|
+
//#endregion
|
|
485
|
+
//#region src/pipelines/failure-cluster.ts
|
|
486
|
+
/**
|
|
487
|
+
* FailureClusterView — groups failed runs by (failureClass, triggerTool,
|
|
488
|
+
* argHash-prefix) so weekly reviews can prioritize the top-N clusters.
|
|
489
|
+
*
|
|
490
|
+
* Each cluster includes: N runs, scenarios affected, representative
|
|
491
|
+
* error message, a proposed mitigation hint (rule → action table).
|
|
492
|
+
*/
|
|
493
|
+
async function failureClusterView(store, options = {}) {
|
|
494
|
+
const rules = options.rules ?? DEFAULT_RULES;
|
|
495
|
+
const minSize = options.minClusterSize ?? 1;
|
|
496
|
+
const runs = await store.listRuns();
|
|
497
|
+
const clusters = /* @__PURE__ */ new Map();
|
|
498
|
+
let totalFailures = 0;
|
|
499
|
+
for (const run of runs) {
|
|
500
|
+
if (run.status === "completed" && run.outcome?.pass !== false) continue;
|
|
501
|
+
totalFailures++;
|
|
502
|
+
const spans = await store.spans({ runId: run.runId });
|
|
503
|
+
const cls = classifyFailure({
|
|
504
|
+
run,
|
|
505
|
+
spans,
|
|
506
|
+
events: await store.events({ runId: run.runId })
|
|
507
|
+
}, rules);
|
|
508
|
+
let toolName;
|
|
509
|
+
let argPrefix;
|
|
510
|
+
let dimension;
|
|
511
|
+
if (cls.triggerSpanId) {
|
|
512
|
+
const trig = spans.find((s) => s.spanId === cls.triggerSpanId);
|
|
513
|
+
if (trig?.kind === "tool") {
|
|
514
|
+
toolName = trig.toolName;
|
|
515
|
+
if (hasCapturedToolArgs(trig)) argPrefix = argHash(trig.args).slice(0, 16);
|
|
516
|
+
} else if (trig?.kind === "judge") dimension = trig.dimension;
|
|
517
|
+
}
|
|
518
|
+
if (!toolName) {
|
|
519
|
+
const errored = (await toolSpans(store, run.runId)).filter((t) => t.status === "error").pop();
|
|
520
|
+
if (errored) {
|
|
521
|
+
toolName = errored.toolName;
|
|
522
|
+
if (hasCapturedToolArgs(errored)) argPrefix = argHash(errored.args).slice(0, 16);
|
|
523
|
+
}
|
|
524
|
+
}
|
|
525
|
+
if (!dimension) {
|
|
526
|
+
const judge = spans.find((s) => s.kind === "judge" && typeof s.dimension === "string");
|
|
527
|
+
if (judge?.kind === "judge") dimension = judge.dimension;
|
|
528
|
+
}
|
|
529
|
+
const key = `${cls.failureClass}|${toolName ?? ""}|${argPrefix ?? ""}|${dimension ?? ""}`;
|
|
530
|
+
let cluster = clusters.get(key);
|
|
531
|
+
if (!cluster) {
|
|
532
|
+
cluster = {
|
|
533
|
+
failureClass: cls.failureClass,
|
|
534
|
+
toolName,
|
|
535
|
+
argPrefix,
|
|
536
|
+
dimension,
|
|
537
|
+
runCount: 0,
|
|
538
|
+
scenarioIds: [],
|
|
539
|
+
exampleRunId: run.runId,
|
|
540
|
+
exampleError: firstErrorMessage(spans) ?? cls.reason
|
|
541
|
+
};
|
|
542
|
+
clusters.set(key, cluster);
|
|
543
|
+
}
|
|
544
|
+
cluster.runCount++;
|
|
545
|
+
if (!cluster.scenarioIds.includes(run.scenarioId)) cluster.scenarioIds.push(run.scenarioId);
|
|
546
|
+
}
|
|
547
|
+
return {
|
|
548
|
+
clusters: [...clusters.values()].filter((c) => c.runCount >= minSize).sort((a, b) => b.runCount - a.runCount),
|
|
549
|
+
totalFailures,
|
|
550
|
+
totalRuns: runs.length
|
|
551
|
+
};
|
|
552
|
+
}
|
|
553
|
+
function firstErrorMessage(spans) {
|
|
554
|
+
return spans.find((s) => s.status === "error")?.error;
|
|
555
|
+
}
|
|
556
|
+
//#endregion
|
|
557
|
+
//#region src/pipelines/judge-agreement.ts
|
|
558
|
+
/**
|
|
559
|
+
* JudgeAgreementView — pairwise agreement between judges across the
|
|
560
|
+
* corpus, grouped by dimension.
|
|
561
|
+
*
|
|
562
|
+
* Output drives two workflows:
|
|
563
|
+
* - Judge robustness audit: "does Claude agree with GPT at κ ≥ 0.6?"
|
|
564
|
+
* - Calibration tracking: κ vs golden human labels over time (by
|
|
565
|
+
* providing a `humanGoldenJudgeId`).
|
|
566
|
+
*/
|
|
567
|
+
async function judgeAgreementView(store) {
|
|
568
|
+
const all = (await store.spans({ kind: "judge" })).filter((s) => s.kind === "judge");
|
|
569
|
+
if (all.length === 0) return {
|
|
570
|
+
pairs: [],
|
|
571
|
+
dimensions: [],
|
|
572
|
+
judgeIds: []
|
|
573
|
+
};
|
|
574
|
+
const byDimension = /* @__PURE__ */ new Map();
|
|
575
|
+
for (const s of all) {
|
|
576
|
+
const arr = byDimension.get(s.dimension) ?? [];
|
|
577
|
+
arr.push(s);
|
|
578
|
+
byDimension.set(s.dimension, arr);
|
|
579
|
+
}
|
|
580
|
+
const judgeIds = [...new Set(all.map((s) => s.judgeId))].sort();
|
|
581
|
+
const pairs = [];
|
|
582
|
+
for (const [dim, spans] of byDimension) {
|
|
583
|
+
const byJudge = /* @__PURE__ */ new Map();
|
|
584
|
+
for (const s of spans) {
|
|
585
|
+
const m = byJudge.get(s.judgeId) ?? /* @__PURE__ */ new Map();
|
|
586
|
+
m.set(s.targetSpanId, s.score);
|
|
587
|
+
byJudge.set(s.judgeId, m);
|
|
588
|
+
}
|
|
589
|
+
const judgesHere = [...byJudge.keys()];
|
|
590
|
+
for (let i = 0; i < judgesHere.length; i++) for (let j = i + 1; j < judgesHere.length; j++) {
|
|
591
|
+
const judgeI = judgesHere[i];
|
|
592
|
+
const judgeJ = judgesHere[j];
|
|
593
|
+
const a = byJudge.get(judgeI);
|
|
594
|
+
const b = byJudge.get(judgeJ);
|
|
595
|
+
const common = [];
|
|
596
|
+
for (const [target, scoreA] of a) {
|
|
597
|
+
const scoreB = b.get(target);
|
|
598
|
+
if (scoreB !== void 0) common.push([scoreA, scoreB]);
|
|
599
|
+
}
|
|
600
|
+
if (common.length < 2) continue;
|
|
601
|
+
const judgeScores = common.map(([scoreA, scoreB]) => [{
|
|
602
|
+
judgeName: judgeI,
|
|
603
|
+
dimension: dim,
|
|
604
|
+
score: scoreA,
|
|
605
|
+
reasoning: ""
|
|
606
|
+
}, {
|
|
607
|
+
judgeName: judgeJ,
|
|
608
|
+
dimension: dim,
|
|
609
|
+
score: scoreB,
|
|
610
|
+
reasoning: ""
|
|
611
|
+
}]);
|
|
612
|
+
const k = interRaterReliability(judgeScores[0].map((_, k2) => judgeScores.map((pair) => pair[k2])));
|
|
613
|
+
pairs.push({
|
|
614
|
+
judgeA: judgeI,
|
|
615
|
+
judgeB: judgeJ,
|
|
616
|
+
dimension: dim,
|
|
617
|
+
commonItems: common.length,
|
|
618
|
+
pearson: pearsonR(common.map((c) => c[0]), common.map((c) => c[1])),
|
|
619
|
+
krippendorff: k
|
|
620
|
+
});
|
|
621
|
+
}
|
|
622
|
+
}
|
|
623
|
+
return {
|
|
624
|
+
pairs: pairs.sort((a, b) => b.commonItems - a.commonItems),
|
|
625
|
+
dimensions: [...byDimension.keys()].sort(),
|
|
626
|
+
judgeIds
|
|
627
|
+
};
|
|
628
|
+
}
|
|
629
|
+
//#endregion
|
|
630
|
+
//#region src/tool-use-metrics.ts
|
|
631
|
+
/**
|
|
632
|
+
* Tool-use metrics — derived purely from trace data.
|
|
633
|
+
*
|
|
634
|
+
* No scoring assumptions: consumers supply optional ground-truth tool
|
|
635
|
+
* selections per turn + optional "information used downstream" signals.
|
|
636
|
+
* Without those, we still compute descriptive metrics (error rate,
|
|
637
|
+
* retry rate, duplicate-call rate) that are useful on their own.
|
|
638
|
+
*/
|
|
639
|
+
async function computeToolUseMetrics(store, runId, options = {}) {
|
|
640
|
+
const tools = await toolSpans(store, runId);
|
|
641
|
+
if (tools.length === 0) return {
|
|
642
|
+
runId,
|
|
643
|
+
totalCalls: 0,
|
|
644
|
+
callsWithCapturedArgs: 0,
|
|
645
|
+
byTool: {},
|
|
646
|
+
errorRate: 0,
|
|
647
|
+
duplicateRate: 0,
|
|
648
|
+
retryRate: 0
|
|
649
|
+
};
|
|
650
|
+
const byTool = {};
|
|
651
|
+
let totalErrors = 0;
|
|
652
|
+
let totalDuplicates = 0;
|
|
653
|
+
let callsWithCapturedArgs = 0;
|
|
654
|
+
const sortedTools = [...tools].sort((a, b) => a.startedAt - b.startedAt);
|
|
655
|
+
const seenSignatures = /* @__PURE__ */ new Set();
|
|
656
|
+
for (const t of sortedTools) {
|
|
657
|
+
byTool[t.toolName] ??= {
|
|
658
|
+
calls: 0,
|
|
659
|
+
callsWithCapturedArgs: 0,
|
|
660
|
+
errors: 0,
|
|
661
|
+
avgLatencyMs: 0,
|
|
662
|
+
duplicates: 0
|
|
663
|
+
};
|
|
664
|
+
const stat = byTool[t.toolName];
|
|
665
|
+
stat.calls += 1;
|
|
666
|
+
if (t.status === "error") {
|
|
667
|
+
stat.errors += 1;
|
|
668
|
+
totalErrors += 1;
|
|
669
|
+
}
|
|
670
|
+
if (typeof t.latencyMs === "number") stat.avgLatencyMs += t.latencyMs;
|
|
671
|
+
if (hasCapturedToolArgs(t)) {
|
|
672
|
+
callsWithCapturedArgs += 1;
|
|
673
|
+
stat.callsWithCapturedArgs += 1;
|
|
674
|
+
const sig = `${t.toolName}|${argHash(t.args)}`;
|
|
675
|
+
if (seenSignatures.has(sig)) {
|
|
676
|
+
stat.duplicates += 1;
|
|
677
|
+
totalDuplicates += 1;
|
|
678
|
+
}
|
|
679
|
+
seenSignatures.add(sig);
|
|
680
|
+
}
|
|
681
|
+
}
|
|
682
|
+
for (const stat of Object.values(byTool)) stat.avgLatencyMs = stat.calls > 0 ? stat.avgLatencyMs / stat.calls : 0;
|
|
683
|
+
let retryOpportunities = 0;
|
|
684
|
+
let retriesFollowed = 0;
|
|
685
|
+
for (const [, arr] of groupBy(sortedTools, (t) => t.toolName)) for (let i = 0; i < arr.length; i++) {
|
|
686
|
+
if (arr[i].status !== "error") continue;
|
|
687
|
+
retryOpportunities += 1;
|
|
688
|
+
if (arr[i + 1]) retriesFollowed += 1;
|
|
689
|
+
}
|
|
690
|
+
const retryRate = retryOpportunities > 0 ? retriesFollowed / retryOpportunities : 0;
|
|
691
|
+
let selectionAccuracy;
|
|
692
|
+
if (options.selectionLabels) {
|
|
693
|
+
const labeled = sortedTools.filter((t) => t.spanId in options.selectionLabels);
|
|
694
|
+
if (labeled.length > 0) selectionAccuracy = labeled.filter((t) => options.selectionLabels[t.spanId]).length / labeled.length;
|
|
695
|
+
}
|
|
696
|
+
return {
|
|
697
|
+
runId,
|
|
698
|
+
totalCalls: sortedTools.length,
|
|
699
|
+
callsWithCapturedArgs,
|
|
700
|
+
byTool,
|
|
701
|
+
errorRate: totalErrors / sortedTools.length,
|
|
702
|
+
duplicateRate: callsWithCapturedArgs > 0 ? totalDuplicates / callsWithCapturedArgs : 0,
|
|
703
|
+
retryRate,
|
|
704
|
+
selectionAccuracy
|
|
705
|
+
};
|
|
706
|
+
}
|
|
707
|
+
//#endregion
|
|
708
|
+
//#region src/pipelines/tool-waste.ts
|
|
709
|
+
/**
|
|
710
|
+
* ToolWasteView — fraction of tool calls whose results weren't used
|
|
711
|
+
* downstream. Without a "used" signal we fall back to structural
|
|
712
|
+
* proxies: error calls, duplicate calls, and tool calls followed by
|
|
713
|
+
* zero subsequent LLM spans are all considered waste.
|
|
714
|
+
*
|
|
715
|
+
* Consumers can pass a `usageOracle` that inspects a tool span and
|
|
716
|
+
* returns true iff the tool's result appears in a later LLM message,
|
|
717
|
+
* artifact, or state mutation — that's the canonical definition; the
|
|
718
|
+
* default heuristic is a reasonable fallback.
|
|
719
|
+
*/
|
|
720
|
+
async function toolWasteView(store, options = {}) {
|
|
721
|
+
const runs = options.runId ? [options.runId] : (await store.listRuns()).map((r) => r.runId);
|
|
722
|
+
const byRun = [];
|
|
723
|
+
let totalCalls = 0;
|
|
724
|
+
let totalWasted = 0;
|
|
725
|
+
for (const runId of runs) {
|
|
726
|
+
const tools = await toolSpans(store, runId);
|
|
727
|
+
if (tools.length === 0) {
|
|
728
|
+
byRun.push({
|
|
729
|
+
runId,
|
|
730
|
+
wastedCalls: 0,
|
|
731
|
+
totalCalls: 0,
|
|
732
|
+
wasteRate: 0
|
|
733
|
+
});
|
|
734
|
+
continue;
|
|
735
|
+
}
|
|
736
|
+
const sortedLlm = [...await llmSpans(store, runId)].sort((a, b) => a.startedAt - b.startedAt);
|
|
737
|
+
const startTimes = sortedLlm.map((l) => l.startedAt);
|
|
738
|
+
const suffixText = buildSuffixText(sortedLlm);
|
|
739
|
+
let wasted = 0;
|
|
740
|
+
for (const t of tools) {
|
|
741
|
+
if (t.status === "error") {
|
|
742
|
+
wasted++;
|
|
743
|
+
continue;
|
|
744
|
+
}
|
|
745
|
+
const cutoff = upperBound(startTimes, t.startedAt);
|
|
746
|
+
if (options.usageOracle) {
|
|
747
|
+
if (!options.usageOracle(t, { llm: sortedLlm.slice(cutoff) })) wasted++;
|
|
748
|
+
} else {
|
|
749
|
+
const resultStr = stringify(t.result);
|
|
750
|
+
if (resultStr === "") continue;
|
|
751
|
+
if (!(suffixText[cutoff] ?? "").includes(resultStr.slice(0, 120))) wasted++;
|
|
752
|
+
}
|
|
753
|
+
}
|
|
754
|
+
const wasteRate = wasted / tools.length;
|
|
755
|
+
byRun.push({
|
|
756
|
+
runId,
|
|
757
|
+
wastedCalls: wasted,
|
|
758
|
+
totalCalls: tools.length,
|
|
759
|
+
wasteRate
|
|
760
|
+
});
|
|
761
|
+
totalCalls += tools.length;
|
|
762
|
+
totalWasted += wasted;
|
|
763
|
+
}
|
|
764
|
+
return {
|
|
765
|
+
byRun,
|
|
766
|
+
overallWasteRate: totalCalls > 0 ? totalWasted / totalCalls : 0
|
|
767
|
+
};
|
|
768
|
+
}
|
|
769
|
+
/**
|
|
770
|
+
* Build per-position suffix haystacks: result[i] is the concatenation of every
|
|
771
|
+
* string message content in spans[i..end]. Built back-to-front so each entry
|
|
772
|
+
* reuses the next one — O(total message text) rather than O(spans²).
|
|
773
|
+
*/
|
|
774
|
+
function buildSuffixText(spans) {
|
|
775
|
+
const result = new Array(spans.length + 1);
|
|
776
|
+
result[spans.length] = "";
|
|
777
|
+
for (let i = spans.length - 1; i >= 0; i--) result[i] = `${spans[i].messages.map((m) => typeof m.content === "string" ? m.content : "").join("\n")}\n${result[i + 1]}`;
|
|
778
|
+
return result;
|
|
779
|
+
}
|
|
780
|
+
/** Index of the first element strictly greater than `target` in a sorted array. */
|
|
781
|
+
function upperBound(sorted, target) {
|
|
782
|
+
let lo = 0;
|
|
783
|
+
let hi = sorted.length;
|
|
784
|
+
while (lo < hi) {
|
|
785
|
+
const mid = lo + hi >>> 1;
|
|
786
|
+
if (sorted[mid] <= target) lo = mid + 1;
|
|
787
|
+
else hi = mid;
|
|
788
|
+
}
|
|
789
|
+
return lo;
|
|
790
|
+
}
|
|
791
|
+
function stringify(v) {
|
|
792
|
+
if (v === null || v === void 0) return "";
|
|
793
|
+
if (typeof v === "string") return v;
|
|
794
|
+
try {
|
|
795
|
+
return JSON.stringify(v);
|
|
796
|
+
} catch {
|
|
797
|
+
return String(v);
|
|
798
|
+
}
|
|
799
|
+
}
|
|
800
|
+
//#endregion
|
|
801
|
+
export { budgetBreachView as a, corpusInterRaterAgreementFromJudgeScores as c, failureClusterView as i, interRaterReliability as l, computeToolUseMetrics as n, classifyFailure as o, judgeAgreementView as r, corpusInterRaterAgreement as s, toolWasteView as t };
|
|
802
|
+
|
|
803
|
+
//# sourceMappingURL=tool-waste-BDdBZG1F.js.map
|