@tangle-network/agent-eval 0.144.11 → 0.144.13
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +11 -0
- package/dist/agent-profile-B9_GGsG8.d.ts +84 -0
- package/dist/agent-profile-B9_GGsG8.d.ts.map +1 -0
- package/dist/{agent-profile-cell-BOP-iA9Q.d.ts → agent-profile-cell-BkcRDikH.d.ts} +2 -2
- package/dist/{agent-profile-cell-BOP-iA9Q.d.ts.map → agent-profile-cell-BkcRDikH.d.ts.map} +1 -1
- package/dist/analyst/index.d.ts +134 -16
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +364 -10
- package/dist/analyst/index.js.map +1 -1
- package/dist/backend-integrity-CsVin_Wb.d.ts +280 -0
- package/dist/backend-integrity-CsVin_Wb.d.ts.map +1 -0
- package/dist/{baseline-C-GocmIW.js → baseline-BhPRQBVn.js} +3 -2
- package/dist/{baseline-C-GocmIW.js.map → baseline-BhPRQBVn.js.map} +1 -1
- package/dist/{benchmark-CWeqGl7x.js → benchmark-BhT16ep9.js} +2 -2
- package/dist/{benchmark-CWeqGl7x.js.map → benchmark-BhT16ep9.js.map} +1 -1
- package/dist/{agent-profile-DPi7IZg7.d.ts → benchmark-Ceoan7vk.d.ts} +4 -91
- package/dist/benchmark-Ceoan7vk.d.ts.map +1 -0
- package/dist/{benchmark-command-BteMFN62.js → benchmark-command-CY6cbhOB.js} +36 -79
- package/dist/benchmark-command-CY6cbhOB.js.map +1 -0
- package/dist/benchmarks/index.d.ts +244 -2
- package/dist/benchmarks/index.d.ts.map +1 -0
- package/dist/benchmarks/index.js +733 -1
- package/dist/benchmarks/index.js.map +1 -0
- package/dist/builder-eval/index.d.ts +23 -2
- package/dist/builder-eval/index.d.ts.map +1 -1
- package/dist/builder-eval/index.js +227 -3
- package/dist/builder-eval/index.js.map +1 -1
- package/dist/campaign/index.d.ts +10 -8
- package/dist/campaign/index.js +9 -6
- package/dist/{campaign-C2TTzQII.js → campaign-BYjBAypg.js} +21 -634
- package/dist/campaign-BYjBAypg.js.map +1 -0
- package/dist/{canonical-D011XM8r.js → canonical-D-XsTQ6_.js} +2 -2
- package/dist/{canonical-D011XM8r.js.map → canonical-D-XsTQ6_.js.map} +1 -1
- package/dist/{index-CQTZ-4XN.d.ts → capture-fetch-DDvpjVRU.d.ts} +3 -3
- package/dist/capture-fetch-DDvpjVRU.d.ts.map +1 -0
- package/dist/{default-registry-BmktKy8r.js → chat-client-Bp4Ebuuc.js} +1276 -1276
- package/dist/chat-client-Bp4Ebuuc.js.map +1 -0
- package/dist/cli.js +2 -2
- package/dist/{client-C9gzZE59.d.ts → client-KUbGulm_.d.ts} +4 -4
- package/dist/{client-C9gzZE59.d.ts.map → client-KUbGulm_.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +13 -390
- package/dist/contract/index.d.ts.map +1 -1
- package/dist/contract/index.js +18 -542
- package/dist/contract/index.js.map +1 -1
- package/dist/{cost-ledger-DMFxsLKr.js → cost-ledger-BSe92yAV.js} +3 -3
- package/dist/{cost-ledger-DMFxsLKr.js.map → cost-ledger-BSe92yAV.js.map} +1 -1
- package/dist/{cost-ledger-Bv_e8XHY.d.ts → cost-ledger-DbQdN3nO.d.ts} +2 -2
- package/dist/{cost-ledger-Bv_e8XHY.d.ts.map → cost-ledger-DbQdN3nO.d.ts.map} +1 -1
- package/dist/{counterfactual-CxmxAONP.d.ts → counterfactual--bpysZF0.d.ts} +2 -14
- package/dist/{counterfactual-CxmxAONP.d.ts.map → counterfactual--bpysZF0.d.ts.map} +1 -1
- package/dist/{counterfactual-CWPTrMH7.js → counterfactual-lDfCx0Uz.js} +3 -31
- package/dist/{counterfactual-CWPTrMH7.js.map → counterfactual-lDfCx0Uz.js.map} +1 -1
- package/dist/{dataset-C8xaLXdY.d.ts → dataset-CJjKqQfA.d.ts} +2 -8
- package/dist/dataset-CJjKqQfA.d.ts.map +1 -0
- package/dist/{default-registry-Bf8Woqmq.d.ts → default-registry-Di6HP6pG.d.ts} +7 -12
- package/dist/default-registry-Di6HP6pG.d.ts.map +1 -0
- package/dist/{analyze-runs-C30yljDJ.js → define-agent-eval-iqjT--ZZ.js} +542 -130
- package/dist/define-agent-eval-iqjT--ZZ.js.map +1 -0
- package/dist/define-agent-eval-sH24zBfM.d.ts +388 -0
- package/dist/define-agent-eval-sH24zBfM.d.ts.map +1 -0
- package/dist/descriptive-B5MwKfbf.js +144 -0
- package/dist/descriptive-B5MwKfbf.js.map +1 -0
- package/dist/{dspy-rlm-engine-DbTk4JdR.js → dspy-rlm-engine-B_qhSc21.js} +4 -4
- package/dist/{dspy-rlm-engine-DbTk4JdR.js.map → dspy-rlm-engine-B_qhSc21.js.map} +1 -1
- package/dist/effect-sizes-DiH8MGOH.js +82 -0
- package/dist/effect-sizes-DiH8MGOH.js.map +1 -0
- package/dist/{engine-3hL-XqwJ.d.ts → engine-otFpE2gF.d.ts} +10 -38
- package/dist/engine-otFpE2gF.d.ts.map +1 -0
- package/dist/{errors-CKPfb2aH.d.ts → errors-DEE6u6ot.d.ts} +2 -14
- package/dist/{errors-CKPfb2aH.d.ts.map → errors-DEE6u6ot.d.ts.map} +1 -1
- package/dist/{errors-D-LKuDhb.js → errors-Dngq5h35.js} +2 -8
- package/dist/{errors-D-LKuDhb.js.map → errors-Dngq5h35.js.map} +1 -1
- package/dist/{eval-campaign-B_7wcnav.js → eval-campaign-UB-usSQ2.js} +6 -6
- package/dist/{eval-campaign-B_7wcnav.js.map → eval-campaign-UB-usSQ2.js.map} +1 -1
- package/dist/{exact-types-DSFFpLLI.d.ts → exact-types-BH1twmAJ.d.ts} +2 -2
- package/dist/{exact-types-DSFFpLLI.d.ts.map → exact-types-BH1twmAJ.d.ts.map} +1 -1
- package/dist/experiment/index.d.ts +9 -6
- package/dist/experiment/index.d.ts.map +1 -1
- package/dist/experiment/index.js +11 -7
- package/dist/experiment/index.js.map +1 -1
- package/dist/experiment-tracker-C29gXM4B.js +269 -0
- package/dist/experiment-tracker-C29gXM4B.js.map +1 -0
- package/dist/{experiment-tracker-IMntXr6J.d.ts → experiment-tracker-DWHZBAYL.d.ts} +77 -121
- package/dist/experiment-tracker-DWHZBAYL.d.ts.map +1 -0
- package/dist/{external-optimizer-contracts-CZuJNcT5.d.ts → external-optimizer-contracts-lixrOZdX.d.ts} +3 -3
- package/dist/{external-optimizer-contracts-CZuJNcT5.d.ts.map → external-optimizer-contracts-lixrOZdX.d.ts.map} +1 -1
- package/dist/external-optimizer-process-BTiNB-RH.js +301 -0
- package/dist/external-optimizer-process-BTiNB-RH.js.map +1 -0
- package/dist/{single-run-lock-DFWHEB09.js → external-optimizer-subprocess-DrJ9hR8u.js} +144 -438
- package/dist/external-optimizer-subprocess-DrJ9hR8u.js.map +1 -0
- package/dist/extract-usage-BrQ8mCLX.js +155 -0
- package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
- package/dist/{failure-cluster-CqcvCcdR.d.ts → failure-cluster-BLURuWG4.d.ts} +2 -3
- package/dist/failure-cluster-BLURuWG4.d.ts.map +1 -0
- package/dist/{feedback-trajectory-CacHpxsp.d.ts → feedback-trajectory-juOozjAc.d.ts} +5 -36
- package/dist/feedback-trajectory-juOozjAc.d.ts.map +1 -0
- package/dist/fuzz.d.ts +2 -2
- package/dist/fuzz.js +2 -2
- package/dist/hosted/index.d.ts +3 -3
- package/dist/{index-CvSN3IG1.d.ts → index-B8Ui1mr1.d.ts} +2 -2
- package/dist/{index-CvSN3IG1.d.ts.map → index-B8Ui1mr1.d.ts.map} +1 -1
- package/dist/{skill-usage-CO9OLRBx.d.ts → index-BWDrSVfw.d.ts} +11 -136
- package/dist/index-BWDrSVfw.d.ts.map +1 -0
- package/dist/index-Ba3YrbAL.d.ts +1 -0
- package/dist/{index-BnEuDAK2.d.ts → index-COQYtuRF.d.ts} +3 -3
- package/dist/{index-BnEuDAK2.d.ts.map → index-COQYtuRF.d.ts.map} +1 -1
- package/dist/{index-C3ssXVLv.d.ts → index-CvjYbU0D.d.ts} +2 -2
- package/dist/{index-C3ssXVLv.d.ts.map → index-CvjYbU0D.d.ts.map} +1 -1
- package/dist/{index-DPPGNJ_R.d.ts → index-DSmEylT9.d.ts} +17 -115
- package/dist/index-DSmEylT9.d.ts.map +1 -0
- package/dist/index.d.ts +2397 -5308
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +5914 -10496
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-CgX_s0Ez.d.ts → insight-report-CRi-Ufrj.d.ts} +4 -4
- package/dist/{insight-report-CgX_s0Ez.d.ts.map → insight-report-CRi-Ufrj.d.ts.map} +1 -1
- package/dist/{integrity-MLzHOfV9.js → integrity-Cy9WHAtb.js} +2 -2
- package/dist/{integrity-MLzHOfV9.js.map → integrity-Cy9WHAtb.js.map} +1 -1
- package/dist/{integrity-DY6tIbl0.js → integrity-DysDBWDu.js} +2 -2
- package/dist/{integrity-DY6tIbl0.js.map → integrity-DysDBWDu.js.map} +1 -1
- package/dist/{integrity-BuqEKu-x.d.ts → integrity-OrcI9Nau.d.ts} +3 -3
- package/dist/{integrity-BuqEKu-x.d.ts.map → integrity-OrcI9Nau.d.ts.map} +1 -1
- package/dist/internal-BDHPCnjk.js +230 -0
- package/dist/internal-BDHPCnjk.js.map +1 -0
- package/dist/judge-calibration-C5CbMYce.d.ts +117 -0
- package/dist/judge-calibration-C5CbMYce.d.ts.map +1 -0
- package/dist/judge-calibration-DZkWrm5H.js +317 -0
- package/dist/judge-calibration-DZkWrm5H.js.map +1 -0
- package/dist/{kind-factory-B8-r8-y8.js → kind-factory-CPmSd58s.js} +208 -208
- package/dist/{kind-factory-B8-r8-y8.js.map → kind-factory-CPmSd58s.js.map} +1 -1
- package/dist/ledger-core/index.d.ts +1 -1
- package/dist/ledger-core/index.js +2 -2
- package/dist/{ledger-core-DXZIqu17.js → ledger-core-BmZt19oQ.js} +110 -110
- package/dist/{ledger-core-DXZIqu17.js.map → ledger-core-BmZt19oQ.js.map} +1 -1
- package/dist/{llm-client-DzvMUsS_.js → llm-client-d0-2TT1g.js} +4 -64
- package/dist/{llm-client-DzvMUsS_.js.map → llm-client-d0-2TT1g.js.map} +1 -1
- package/dist/{skillopt-optimization-method-CQdVeM8k.js → llm-judge-dZ8P6nGI.js} +3349 -5401
- package/dist/llm-judge-dZ8P6nGI.js.map +1 -0
- package/dist/matrix/index.d.ts +2 -2
- package/dist/meta-eval/index.d.ts +3 -3
- package/dist/meta-eval/index.js +3 -3
- package/dist/{metrics-C9YY1OcL.js → metrics-Cl0L1KUy.js} +2 -108
- package/dist/{metrics-C9YY1OcL.js.map → metrics-Cl0L1KUy.js.map} +1 -1
- package/dist/{mint-B2O60ACG.js → mint-Dj9Ww_3I.js} +4 -4
- package/dist/{mint-B2O60ACG.js.map → mint-Dj9Ww_3I.js.map} +1 -1
- package/dist/{multi-layer-verifier-DnAqwl0h.d.ts → multi-layer-verifier-DIguZc8Z.d.ts} +3 -3
- package/dist/{multi-layer-verifier-DnAqwl0h.d.ts.map → multi-layer-verifier-DIguZc8Z.d.ts.map} +1 -1
- package/dist/multiplicity-DIWHvysC.d.ts +43 -0
- package/dist/multiplicity-DIWHvysC.d.ts.map +1 -0
- package/dist/multishot/index.d.ts +3 -3
- package/dist/multishot/index.js +1 -1
- package/dist/openapi.json +1 -1
- package/dist/{opencode-sqlite-8r6WUfHc.js → opencode-sqlite-DJWAXLms.js} +2 -2
- package/dist/{opencode-sqlite-8r6WUfHc.js.map → opencode-sqlite-DJWAXLms.js.map} +1 -1
- package/dist/package-version-D7lQHt_-.js +34 -0
- package/dist/package-version-D7lQHt_-.js.map +1 -0
- package/dist/paired-arms-D-XRF_fy.js +1045 -0
- package/dist/paired-arms-D-XRF_fy.js.map +1 -0
- package/dist/paired-promotion-decision-CGzg0cI_.d.ts +251 -0
- package/dist/paired-promotion-decision-CGzg0cI_.d.ts.map +1 -0
- package/dist/paired-tests-BHIhYVdu.js +213 -0
- package/dist/paired-tests-BHIhYVdu.js.map +1 -0
- package/dist/pareto-BqNW3LJR.d.ts +117 -0
- package/dist/pareto-BqNW3LJR.d.ts.map +1 -0
- package/dist/pipelines/index.d.ts +3 -64
- package/dist/pipelines/index.d.ts.map +1 -1
- package/dist/pipelines/index.js +4 -284
- package/dist/pipelines/index.js.map +1 -1
- package/dist/power-and-mde-CHIrXJll.js +195 -0
- package/dist/power-and-mde-CHIrXJll.js.map +1 -0
- package/dist/{promotion-policy-CrLrmys8.js → power-preflight-DEw-uC7q.js} +4 -184
- package/dist/power-preflight-DEw-uC7q.js.map +1 -0
- package/dist/pre-registration-CZwSFQS4.d.ts +577 -0
- package/dist/pre-registration-CZwSFQS4.d.ts.map +1 -0
- package/dist/{prime-protocol-BfSalTfR.js → prime-protocol-6tZTVsWm.js} +72 -21
- package/dist/prime-protocol-6tZTVsWm.js.map +1 -0
- package/dist/produced-state-DU79a81m.js +586 -0
- package/dist/produced-state-DU79a81m.js.map +1 -0
- package/dist/profile-cell.d.ts +1 -1
- package/dist/profile-cell.js +1 -1
- package/dist/promotion-policy-D0nPhkSy.d.ts +134 -0
- package/dist/promotion-policy-D0nPhkSy.d.ts.map +1 -0
- package/dist/promotion-policy-xzA40Evo.js +186 -0
- package/dist/promotion-policy-xzA40Evo.js.map +1 -0
- package/dist/{skillopt-optimization-method-USDKhxSA.d.ts → provenance-CCdxgLDT.d.ts} +57 -478
- package/dist/provenance-CCdxgLDT.d.ts.map +1 -0
- package/dist/registry-oJeeI4-a.d.ts +178 -0
- package/dist/registry-oJeeI4-a.d.ts.map +1 -0
- package/dist/{release-report-DA2BCu5a.d.ts → release-confidence-BFRE5WSp.d.ts} +180 -112
- package/dist/release-confidence-BFRE5WSp.d.ts.map +1 -0
- package/dist/{release-report-BUYmoKo2.js → release-confidence-CxDuiAev.js} +125 -217
- package/dist/release-confidence-CxDuiAev.js.map +1 -0
- package/dist/reporting.d.ts +6 -5
- package/dist/reporting.js +7 -5
- package/dist/{researcher-BchpD55R.d.ts → researcher-DaN4GST-.d.ts} +7 -40
- package/dist/{researcher-BchpD55R.d.ts.map → researcher-DaN4GST-.d.ts.map} +1 -1
- package/dist/{reward-hacking-BDToousL.js → reward-hacking-DNgjilrV.js} +3 -3
- package/dist/reward-hacking-DNgjilrV.js.map +1 -0
- package/dist/{reward-hacking-BI0OMAlo.d.ts → reward-hacking-DSSTuI9r.d.ts} +5 -5
- package/dist/{reward-hacking-BI0OMAlo.d.ts.map → reward-hacking-DSSTuI9r.d.ts.map} +1 -1
- package/dist/rl.d.ts +7 -7
- package/dist/rl.js +11 -10
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +2 -2
- package/dist/rollout/index.js +4 -4
- package/dist/{rollout-DRrksrcV.js → rollout-BWtw0I_6.js} +3 -3
- package/dist/{rollout-DRrksrcV.js.map → rollout-BWtw0I_6.js.map} +1 -1
- package/dist/{rubric-predictive-validity-DvkjPYCe.d.ts → rubric-predictive-validity-BgxtKe4G.d.ts} +2 -2
- package/dist/{rubric-predictive-validity-DvkjPYCe.d.ts.map → rubric-predictive-validity-BgxtKe4G.d.ts.map} +1 -1
- package/dist/{rubric-predictive-validity-BRR632r1.js → rubric-predictive-validity-Cwwyd7ah.js} +2 -2
- package/dist/{rubric-predictive-validity-BRR632r1.js.map → rubric-predictive-validity-Cwwyd7ah.js.map} +1 -1
- package/dist/{run-record-BmSPWXJR.js → run-record-BvHPVS-i.js} +3 -3
- package/dist/{run-record-BmSPWXJR.js.map → run-record-BvHPVS-i.js.map} +1 -1
- package/dist/{run-record-CF4Dwpxr.d.ts → run-record-CKiihE6f.d.ts} +4 -4
- package/dist/{run-record-CF4Dwpxr.d.ts.map → run-record-CKiihE6f.d.ts.map} +1 -1
- package/dist/{proposal-findings-2GIUo1et.js → run-score-lDzV0X8j.js} +31 -31
- package/dist/run-score-lDzV0X8j.js.map +1 -0
- package/dist/sandbox-harness-BlSOu4LX.d.ts +70 -0
- package/dist/sandbox-harness-BlSOu4LX.d.ts.map +1 -0
- package/dist/{schema-Cef2cFmb2.d.ts → schema-Cef2cFmb.d.ts} +1 -1
- package/dist/schema-Cef2cFmb.d.ts.map +1 -0
- package/dist/semantic-concept-judge-BiJxScqe.js +406 -0
- package/dist/semantic-concept-judge-BiJxScqe.js.map +1 -0
- package/dist/{sequential-D-BLJBKU.js → sequential-C458DXNf.js} +4 -3
- package/dist/{sequential-D-BLJBKU.js.map → sequential-C458DXNf.js.map} +1 -1
- package/dist/sequential-eprocess-CbUt2htw.js +83 -0
- package/dist/sequential-eprocess-CbUt2htw.js.map +1 -0
- package/dist/series-convergence-D1cL1f-4.d.ts +129 -0
- package/dist/series-convergence-D1cL1f-4.d.ts.map +1 -0
- package/dist/{server-iu0ede49.js → server-ulsOdrTI.js} +5 -21
- package/dist/server-ulsOdrTI.js.map +1 -0
- package/dist/skillopt-optimization-method-C5cotF4E.d.ts +427 -0
- package/dist/skillopt-optimization-method-C5cotF4E.d.ts.map +1 -0
- package/dist/skillopt-optimization-method-jjdnc3YK.js +1999 -0
- package/dist/skillopt-optimization-method-jjdnc3YK.js.map +1 -0
- package/dist/{statistical-heldout-TQ-4CYiN.d.ts → statistical-heldout-DoFF5KJX.d.ts} +5 -278
- package/dist/statistical-heldout-DoFF5KJX.d.ts.map +1 -0
- package/dist/{store-otlp-Dw8PPIlL.js → store-otlp-CsptLYpN.js} +3 -3
- package/dist/{store-otlp-Dw8PPIlL.js.map → store-otlp-CsptLYpN.js.map} +1 -1
- package/dist/store-tool-spans-Cq9mFd-q.js +667 -0
- package/dist/store-tool-spans-Cq9mFd-q.js.map +1 -0
- package/dist/store-tool-spans-DbDLOBOb.d.ts +342 -0
- package/dist/store-tool-spans-DbDLOBOb.d.ts.map +1 -0
- package/dist/student-t-CvBq2mve.js +38 -0
- package/dist/student-t-CvBq2mve.js.map +1 -0
- package/dist/{summary-report-Lf-5I7xh.js → summary-report-Blysd6Z2.js} +6 -3
- package/dist/{summary-report-Lf-5I7xh.js.map → summary-report-Blysd6Z2.js.map} +1 -1
- package/dist/{summary-report-BNR7DWTj.d.ts → summary-report-CaL-Hnxt.d.ts} +5 -5
- package/dist/{summary-report-BNR7DWTj.d.ts.map → summary-report-CaL-Hnxt.d.ts.map} +1 -1
- package/dist/supervisor-run/index.d.ts +391 -3
- package/dist/supervisor-run/index.d.ts.map +1 -0
- package/dist/supervisor-run/index.js +1689 -2
- package/dist/{supervisor-run-D_sokXcO.js.map → supervisor-run/index.js.map} +1 -1
- package/dist/{extract-usage-CdZdoj1s.js → task-failure-attributes-CpQ4y5RD.js} +5 -157
- package/dist/task-failure-attributes-CpQ4y5RD.js.map +1 -0
- package/dist/{tool-groups-CZPGGlHf.d.ts → tool-groups-Cteb03Ps.d.ts} +3 -3
- package/dist/tool-groups-Cteb03Ps.d.ts.map +1 -0
- package/dist/tool-waste-BDdBZG1F.js +803 -0
- package/dist/tool-waste-BDdBZG1F.js.map +1 -0
- package/dist/tool-waste-DjRDEsuI.d.ts +128 -0
- package/dist/tool-waste-DjRDEsuI.d.ts.map +1 -0
- package/dist/trace-repair/index.d.ts +14 -5
- package/dist/trace-repair/index.d.ts.map +1 -1
- package/dist/trace-repair/index.js +35 -7
- package/dist/trace-repair/index.js.map +1 -1
- package/dist/traces.d.ts +406 -7
- package/dist/traces.d.ts.map +1 -0
- package/dist/traces.js +1011 -10
- package/dist/traces.js.map +1 -0
- package/dist/trajectory-replay/index.d.ts +16 -3
- package/dist/trajectory-replay/index.d.ts.map +1 -1
- package/dist/trajectory-replay/index.js +52 -5
- package/dist/trajectory-replay/index.js.map +1 -1
- package/dist/types-BEPZc6eo.d.ts +93 -0
- package/dist/types-BEPZc6eo.d.ts.map +1 -0
- package/dist/{usage-receipt-t7vAzCRQ.js → types-BI4fT3HN.js} +67 -67
- package/dist/types-BI4fT3HN.js.map +1 -0
- package/dist/{types-BnjdJ70P.d.ts → types-BZ59Ahr8.d.ts} +5 -5
- package/dist/{types-BnjdJ70P.d.ts.map → types-BZ59Ahr8.d.ts.map} +1 -1
- package/dist/{types-D216SgwM.d.ts → types-Cx3YUh2r.d.ts} +4 -241
- package/dist/types-Cx3YUh2r.d.ts.map +1 -0
- package/dist/{types-BvDKaULh.d.ts → types-DLQx4mKU.d.ts} +4 -4
- package/dist/{types-BvDKaULh.d.ts.map → types-DLQx4mKU.d.ts.map} +1 -1
- package/dist/{types-D4mog56g.d.ts → types-yLK8gXE9.d.ts} +2 -2
- package/dist/{types-D4mog56g.d.ts.map → types-yLK8gXE9.d.ts.map} +1 -1
- package/dist/verdict-BndeTAh_.js +61 -0
- package/dist/verdict-BndeTAh_.js.map +1 -0
- package/dist/verdict-E4eRNf7-.d.ts +392 -0
- package/dist/verdict-E4eRNf7-.d.ts.map +1 -0
- package/dist/{verdict-cache-BCcOh0kF.js → verdict-cache-mZf5FEiY.js} +3 -55
- package/dist/{verdict-cache-BCcOh0kF.js.map → verdict-cache-mZf5FEiY.js.map} +1 -1
- package/dist/wire/index.d.ts +3 -3
- package/dist/wire/index.d.ts.map +1 -1
- package/dist/wire/index.js +1 -1
- package/docs/charter.md +3 -3
- package/docs/control-runtime.md +3 -42
- package/docs/experiment.md +0 -1
- package/docs/feature-guide.md +2 -2
- package/docs/trace-repair-grader.md +1 -0
- package/docs/trajectory-replay.md +1 -0
- package/docs/verdicts.md +43 -0
- package/docs/verification-strategies.md +3 -2
- package/package.json +6 -11
- package/dist/agent-profile-DPi7IZg7.d.ts.map +0 -1
- package/dist/analyze-runs-C30yljDJ.js.map +0 -1
- package/dist/baseline-CavEbRyH.d.ts +0 -136
- package/dist/baseline-CavEbRyH.d.ts.map +0 -1
- package/dist/benchmark-command-BteMFN62.js.map +0 -1
- package/dist/benchmarks-Dzs8CKb1.js +0 -755
- package/dist/benchmarks-Dzs8CKb1.js.map +0 -1
- package/dist/campaign-C2TTzQII.js.map +0 -1
- package/dist/completion-verifier-DJA5BhPb.d.ts +0 -414
- package/dist/completion-verifier-DJA5BhPb.d.ts.map +0 -1
- package/dist/control.d.ts +0 -3
- package/dist/control.js +0 -2
- package/dist/dataset-C8xaLXdY.d.ts.map +0 -1
- package/dist/default-registry-Bf8Woqmq.d.ts.map +0 -1
- package/dist/default-registry-BmktKy8r.js.map +0 -1
- package/dist/engine-3hL-XqwJ.d.ts.map +0 -1
- package/dist/experiment-tracker-CnRICnMl.js +0 -500
- package/dist/experiment-tracker-CnRICnMl.js.map +0 -1
- package/dist/experiment-tracker-IMntXr6J.d.ts.map +0 -1
- package/dist/extract-usage-CdZdoj1s.js.map +0 -1
- package/dist/failure-cluster-CqcvCcdR.d.ts.map +0 -1
- package/dist/feedback-trajectory-CacHpxsp.d.ts.map +0 -1
- package/dist/index-BZ3-y4YL.d.ts +0 -391
- package/dist/index-BZ3-y4YL.d.ts.map +0 -1
- package/dist/index-CQTZ-4XN.d.ts.map +0 -1
- package/dist/index-DPPGNJ_R.d.ts.map +0 -1
- package/dist/index-YE4KdKbO2.d.ts +0 -335
- package/dist/index-YE4KdKbO2.d.ts.map +0 -1
- package/dist/paired-arms-iZ08VFMN.js +0 -260
- package/dist/paired-arms-iZ08VFMN.js.map +0 -1
- package/dist/paired-promotion-decision-B6zJ3gYM.d.ts +0 -114
- package/dist/paired-promotion-decision-B6zJ3gYM.d.ts.map +0 -1
- package/dist/prime-protocol-BfSalTfR.js.map +0 -1
- package/dist/promotion-policy-Ckjhzg_4.d.ts +0 -289
- package/dist/promotion-policy-Ckjhzg_4.d.ts.map +0 -1
- package/dist/promotion-policy-CrLrmys8.js.map +0 -1
- package/dist/proposal-findings-2GIUo1et.js.map +0 -1
- package/dist/propose-review-control-dSNPjFUH.js +0 -1458
- package/dist/propose-review-control-dSNPjFUH.js.map +0 -1
- package/dist/release-report-BUYmoKo2.js.map +0 -1
- package/dist/release-report-DA2BCu5a.d.ts.map +0 -1
- package/dist/replay-CohS93nE.js +0 -1859
- package/dist/replay-CohS93nE.js.map +0 -1
- package/dist/replay-DbhZ4Ked.d.ts +0 -834
- package/dist/replay-DbhZ4Ked.d.ts.map +0 -1
- package/dist/reward-hacking-BDToousL.js.map +0 -1
- package/dist/run-evidence-Dhi3C81V.d.ts +0 -225
- package/dist/run-evidence-Dhi3C81V.d.ts.map +0 -1
- package/dist/schema-Cef2cFmb2.d.ts.map +0 -1
- package/dist/semantic-concept-judge-D1z-KepS.js +0 -767
- package/dist/semantic-concept-judge-D1z-KepS.js.map +0 -1
- package/dist/series-convergence-ofsqPWhs.d.ts +0 -35
- package/dist/series-convergence-ofsqPWhs.d.ts.map +0 -1
- package/dist/server-iu0ede49.js.map +0 -1
- package/dist/single-run-lock-DFWHEB09.js.map +0 -1
- package/dist/skill-usage-CO9OLRBx.d.ts.map +0 -1
- package/dist/skillopt-optimization-method-CQdVeM8k.js.map +0 -1
- package/dist/skillopt-optimization-method-USDKhxSA.d.ts.map +0 -1
- package/dist/statistical-heldout-TQ-4CYiN.d.ts.map +0 -1
- package/dist/statistics-ByxzSiOM.js +0 -2212
- package/dist/statistics-ByxzSiOM.js.map +0 -1
- package/dist/statistics-D6Uebe_4.d.ts +0 -968
- package/dist/statistics-D6Uebe_4.d.ts.map +0 -1
- package/dist/supervisor-run-D_sokXcO.js +0 -1690
- package/dist/test-graded-scenario-D1TaI2va.d.ts +0 -141
- package/dist/test-graded-scenario-D1TaI2va.d.ts.map +0 -1
- package/dist/test-graded-scenario-JHcKQNpq.js +0 -318
- package/dist/test-graded-scenario-JHcKQNpq.js.map +0 -1
- package/dist/tool-groups-CZPGGlHf.d.ts.map +0 -1
- package/dist/tool-use-metrics-DEGMKycK.js +0 -370
- package/dist/tool-use-metrics-DEGMKycK.js.map +0 -1
- package/dist/types-D216SgwM.d.ts.map +0 -1
- package/dist/usage-receipt-t7vAzCRQ.js.map +0 -1
- package/dist/verdict-DExhxfgR.d.ts +0 -201
- package/dist/verdict-DExhxfgR.d.ts.map +0 -1
|
@@ -1,1348 +1,1267 @@
|
|
|
1
|
-
import {
|
|
1
|
+
import { i as hashCanonical, r as canonicalString } from "./canonical-D-XsTQ6_.js";
|
|
2
|
+
import { a as combineAbortSignals } from "./run-score-lDzV0X8j.js";
|
|
3
|
+
import { c as deepFreezeCanonicalJson, i as assertValidAnalystUsageReceipt, n as makeFinding, s as validateUsageSettlementTimeout } from "./types-BI4fT3HN.js";
|
|
4
|
+
import { R as findingSubjectGrammarPromptFor, X as spanEpochMillis, nt as snapshotExactExecutionComponentIdentity, rt as snapshotExactExecutionPlan, t as createTraceAnalyst } from "./kind-factory-CPmSd58s.js";
|
|
2
5
|
import { LLM_CONTEXT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_OUTPUT_TOKEN_ATTR_KEYS, TOOL_NAME_ATTR_KEYS } from "./trace-attributes.js";
|
|
6
|
+
import { n as LlmClient } from "./llm-client-d0-2TT1g.js";
|
|
3
7
|
import { t as executionTrackByLane } from "./execution-tracks-CpgFPpS5.js";
|
|
4
|
-
import {
|
|
5
|
-
import { a as deepFreezeCanonicalJson, i as validateUsageSettlementTimeout, s as makeFinding, t as assertValidAnalystUsageReceipt } from "./usage-receipt-t7vAzCRQ.js";
|
|
6
|
-
import { i as hashCanonical, r as canonicalString } from "./canonical-D011XM8r.js";
|
|
7
|
-
import { t as analyzeSupervisorRunIntegrity } from "./integrity-DY6tIbl0.js";
|
|
8
|
-
import { o as combineAbortSignals } from "./proposal-findings-2GIUo1et.js";
|
|
9
|
-
import { z } from "zod";
|
|
8
|
+
import { t as analyzeSupervisorRunIntegrity } from "./integrity-DysDBWDu.js";
|
|
10
9
|
import { randomUUID } from "node:crypto";
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
* Evaluation code receives canonical requests and results without importing a
|
|
17
|
-
* provider SDK.
|
|
18
|
-
*/
|
|
19
|
-
/**
|
|
20
|
-
* Build a ChatClient bound to a specific transport. The returned client
|
|
21
|
-
* is safe to share across analysts in a single registry run.
|
|
22
|
-
*/
|
|
23
|
-
function createChatClient(opts) {
|
|
24
|
-
switch (opts.transport) {
|
|
25
|
-
case "router": return wrapLlmClient(opts.transport, opts.defaultModel, new LlmClient({
|
|
26
|
-
baseUrl: opts.baseUrl ?? "https://router.tangle.tools/v1",
|
|
27
|
-
apiKey: opts.apiKey,
|
|
28
|
-
maximumAttempts: opts.maximumAttempts
|
|
29
|
-
}));
|
|
30
|
-
case "cli-bridge": return wrapLlmClient(opts.transport, opts.defaultModel, new LlmClient({
|
|
31
|
-
baseUrl: opts.baseUrl ?? "http://127.0.0.1:3344/v1",
|
|
32
|
-
apiKey: opts.bearer ?? "",
|
|
33
|
-
maximumAttempts: opts.maximumAttempts
|
|
34
|
-
}));
|
|
35
|
-
case "direct-provider": return wrapLlmClient(opts.transport, opts.defaultModel, new LlmClient({
|
|
36
|
-
baseUrl: opts.baseUrl,
|
|
37
|
-
apiKey: opts.apiKey,
|
|
38
|
-
maximumAttempts: opts.maximumAttempts
|
|
39
|
-
}));
|
|
40
|
-
case "sandbox-sdk": return {
|
|
41
|
-
transport: "sandbox-sdk",
|
|
42
|
-
defaultModel: opts.defaultModel,
|
|
43
|
-
maximumAttempts: opts.maximumAttempts,
|
|
44
|
-
chat: async (req, callOpts) => opts.chat(resolveModel(req, opts.defaultModel), callOpts)
|
|
45
|
-
};
|
|
46
|
-
case "custom": return {
|
|
47
|
-
transport: "custom",
|
|
48
|
-
defaultModel: opts.defaultModel,
|
|
49
|
-
maximumAttempts: opts.maximumAttempts,
|
|
50
|
-
chat: async (req, callOpts) => opts.chat(resolveModel(req, opts.defaultModel), callOpts)
|
|
51
|
-
};
|
|
52
|
-
case "mock": return {
|
|
53
|
-
transport: "mock",
|
|
54
|
-
defaultModel: opts.defaultModel,
|
|
55
|
-
maximumAttempts: 1,
|
|
56
|
-
chat: async (req, callOpts) => opts.handler(resolveModel(req, opts.defaultModel), callOpts)
|
|
57
|
-
};
|
|
58
|
-
}
|
|
10
|
+
import { z } from "zod";
|
|
11
|
+
//#region src/feedback-trajectory-review.ts
|
|
12
|
+
/** Bind an analyst finding's complete canonical JSON content to a stable digest. */
|
|
13
|
+
function analystFindingDigest(finding) {
|
|
14
|
+
return hashCanonical(snapshotAnalystFinding(finding, "analyst finding"));
|
|
59
15
|
}
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
defaultModel,
|
|
64
|
-
maximumAttempts: inner.maximumAttempts,
|
|
65
|
-
chat: (req, callOpts) => {
|
|
66
|
-
const request = {
|
|
67
|
-
model: resolveModel(req, defaultModel).model,
|
|
68
|
-
messages: req.messages,
|
|
69
|
-
jsonMode: req.jsonMode,
|
|
70
|
-
jsonSchema: req.jsonSchema,
|
|
71
|
-
temperature: req.temperature,
|
|
72
|
-
maxTokens: req.maxTokens,
|
|
73
|
-
thinking: req.thinking,
|
|
74
|
-
timeoutMs: req.timeoutMs
|
|
75
|
-
};
|
|
76
|
-
return inner.call(request, {
|
|
77
|
-
signal: callOpts?.signal,
|
|
78
|
-
idempotencyKey: callOpts?.idempotencyKey
|
|
79
|
-
});
|
|
80
|
-
}
|
|
81
|
-
};
|
|
16
|
+
/** Bind the complete analyst result to one immutable review target. */
|
|
17
|
+
function analystRunDigest(run) {
|
|
18
|
+
return hashCanonical(snapshotAnalystRun(run, "analyst run"));
|
|
82
19
|
}
|
|
83
|
-
function
|
|
84
|
-
|
|
85
|
-
if (
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
model: defaultModel
|
|
89
|
-
};
|
|
20
|
+
function snapshotAnalystRun(value, context = "analyst run") {
|
|
21
|
+
const snapshot = snapshotAnalystRunRecord(value, context);
|
|
22
|
+
if (snapshot.execution_plan !== void 0) return sealExactAnalystRunReceipt(snapshot, context);
|
|
23
|
+
if (snapshot.completion !== void 0) throw new TypeError(`${context} completion is valid only for an exact run`);
|
|
24
|
+
return snapshot;
|
|
90
25
|
}
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
* Deterministic behavioral metrics over OTLP spans — pure arithmetic, no LLM.
|
|
95
|
-
*
|
|
96
|
-
* It computes token growth, output decay, tool monoculture, and missing
|
|
97
|
-
* self-verification once in TypeScript with no model judgment.
|
|
98
|
-
*
|
|
99
|
-
* General, not trace-specific: the detectors key off token trajectories and
|
|
100
|
-
* tool usage present in any agentic OTLP trace, not any one benchmark.
|
|
101
|
-
*/
|
|
102
|
-
/** ≥ this input-token growth ratio across a run, with no compression, fires. */
|
|
103
|
-
const INPUT_GROWTH_FACTOR = 3;
|
|
104
|
-
/** Tool-usage signals need at least this many calls to be meaningful. */
|
|
105
|
-
const MIN_TOOL_CALLS = 3;
|
|
106
|
-
/**
|
|
107
|
-
* Serial calls required before a strictly-decreasing output run counts as decay.
|
|
108
|
-
*
|
|
109
|
-
* A run of n independent lengths is strictly decreasing by chance with
|
|
110
|
-
* probability 1/n!, so the old minimum of 3 fired on roughly one sequence in
|
|
111
|
-
* six. The paired `inputIsMonotonic && inputGrew` guard does not offset that:
|
|
112
|
-
* context accumulates by construction in a serial agent loop, so input growth
|
|
113
|
-
* is very nearly free evidence. At 5 calls chance alone accounts for under 1%,
|
|
114
|
-
* which is the bar a signal reported at full confidence has to clear.
|
|
115
|
-
*/
|
|
116
|
-
const OUTPUT_DECAY_MINIMUM_CALLS = 5;
|
|
117
|
-
/**
|
|
118
|
-
* The last output must fall to at most this fraction of the first.
|
|
119
|
-
*
|
|
120
|
-
* Length wanders between turns for reasons that are not degradation, so
|
|
121
|
-
* direction alone is not a finding — an observed 784 → 646 run (18%) was
|
|
122
|
-
* reported as decay and was noise. Requiring the response to lose most of its
|
|
123
|
-
* length keeps the signal on the failure it names: late steps that quietly
|
|
124
|
-
* stop doing the work.
|
|
125
|
-
*/
|
|
126
|
-
const OUTPUT_DECAY_MAXIMUM_RETAINED_FRACTION = .6;
|
|
127
|
-
/** Tool names that read or check state count as self-verification, not mutation.
|
|
128
|
-
* Covers the inspect verbs plus the read/search tools real harnesses use to
|
|
129
|
-
* verify (Claude Code Read/Grep/Glob, codex read_file/ls/cat, git status/diff,
|
|
130
|
-
* test/lint). A pure shell tool (Bash/exec_command) is intentionally NOT matched
|
|
131
|
-
* — its name can't tell a `pytest` from an `rm`. */
|
|
132
|
-
const VERIFY_RE = /verif|eval|inspect|check|assert|validat|review|confirm|read|grep|glob|search|view|\blist\b|\bls\b|\bcat\b|\bfind\b|diff|status|\btest|lint|typecheck/i;
|
|
133
|
-
function num(v) {
|
|
134
|
-
return typeof v === "number" && Number.isFinite(v) ? v : null;
|
|
26
|
+
/** Canonicalize, validate, and deeply freeze one complete or failed exact-run receipt. */
|
|
27
|
+
function snapshotExactAnalystRunReceipt(value, context = "exact analyst run receipt") {
|
|
28
|
+
return sealExactAnalystRunReceipt(snapshotAnalystRunRecord(value, context), context);
|
|
135
29
|
}
|
|
136
|
-
function
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
30
|
+
function snapshotAnalystRunRecord(value, context) {
|
|
31
|
+
let snapshot;
|
|
32
|
+
try {
|
|
33
|
+
snapshot = JSON.parse(canonicalString(value));
|
|
34
|
+
} catch (cause) {
|
|
35
|
+
throw new TypeError(`${context} must have a canonical JSON representation`, { cause });
|
|
140
36
|
}
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
37
|
+
if (!isRecord(snapshot)) throw new TypeError(`${context} must be an object`);
|
|
38
|
+
assertOnlyKeys(snapshot, [
|
|
39
|
+
"run_id",
|
|
40
|
+
"correlation_id",
|
|
41
|
+
"started_at",
|
|
42
|
+
"ended_at",
|
|
43
|
+
"findings",
|
|
44
|
+
"per_analyst",
|
|
45
|
+
"total_cost_usd",
|
|
46
|
+
"total_cost_provenance",
|
|
47
|
+
"execution_plan",
|
|
48
|
+
"completion"
|
|
49
|
+
], context);
|
|
50
|
+
requiredString(snapshot.run_id, `${context} run_id`);
|
|
51
|
+
requiredString(snapshot.correlation_id, `${context} correlation_id`);
|
|
52
|
+
canonicalTimestamp(snapshot.started_at, `${context} started_at`);
|
|
53
|
+
canonicalTimestamp(snapshot.ended_at, `${context} ended_at`);
|
|
54
|
+
snapshot.findings = snapshotAnalystFindings(snapshot.findings, `${context} findings`);
|
|
55
|
+
if (!Array.isArray(snapshot.per_analyst)) throw new TypeError(`${context} per_analyst must be an array`);
|
|
56
|
+
for (const [index, summary] of snapshot.per_analyst.entries()) assertAnalystRunSummary(summary, `${context} per_analyst ${index}`);
|
|
57
|
+
if (typeof snapshot.total_cost_usd !== "number" || !Number.isFinite(snapshot.total_cost_usd) || snapshot.total_cost_usd < 0) throw new TypeError(`${context} total_cost_usd must be a finite non-negative number`);
|
|
58
|
+
if (snapshot.total_cost_provenance !== void 0) assertCostProvenance(snapshot.total_cost_provenance, `${context} total_cost_provenance`);
|
|
59
|
+
return snapshot;
|
|
147
60
|
}
|
|
148
|
-
function
|
|
149
|
-
|
|
61
|
+
function assertAnalystRunSummary(value, context) {
|
|
62
|
+
if (!isRecord(value)) throw new TypeError(`${context} must be an object`);
|
|
63
|
+
assertOnlyKeys(value, [
|
|
64
|
+
"analyst_id",
|
|
65
|
+
"status",
|
|
66
|
+
"reason",
|
|
67
|
+
"findings_count",
|
|
68
|
+
"latency_ms",
|
|
69
|
+
"usage",
|
|
70
|
+
"allocated_budget_usd",
|
|
71
|
+
"error"
|
|
72
|
+
], context);
|
|
73
|
+
requiredString(value.analyst_id, `${context} analyst_id`);
|
|
74
|
+
if (value.status !== "ok" && value.status !== "skipped" && value.status !== "failed") throw new TypeError(`${context} status is invalid`);
|
|
75
|
+
if (value.reason !== void 0) requiredString(value.reason, `${context} reason`);
|
|
76
|
+
if (value.status === "skipped" && value.reason === void 0) throw new TypeError(`${context} skipped summary requires reason`);
|
|
77
|
+
nonnegativeSafeInteger(value.findings_count, `${context} findings_count`);
|
|
78
|
+
finiteNonnegative(value.latency_ms, `${context} latency_ms`);
|
|
79
|
+
assertAnalystUsageReceipt(value.usage, `${context} usage`);
|
|
80
|
+
if (value.allocated_budget_usd !== void 0 && value.allocated_budget_usd !== null) finiteNonnegative(value.allocated_budget_usd, `${context} allocated_budget_usd`);
|
|
81
|
+
if (value.error !== void 0) {
|
|
82
|
+
if (value.status !== "failed" || !isRecord(value.error)) throw new TypeError(`${context} error is valid only for failed summaries`);
|
|
83
|
+
assertOnlyKeys(value.error, ["class", "message"], `${context} error`);
|
|
84
|
+
requiredString(value.error.class, `${context} error class`);
|
|
85
|
+
requiredString(value.error.message, `${context} error message`);
|
|
86
|
+
} else if (value.status === "failed") throw new TypeError(`${context} failed summary requires error`);
|
|
150
87
|
}
|
|
151
|
-
function
|
|
152
|
-
|
|
88
|
+
function assertAnalystUsageReceipt(value, context) {
|
|
89
|
+
if (!isRecord(value)) throw new TypeError(`${context} must be an object`);
|
|
90
|
+
assertOnlyKeys(value, [
|
|
91
|
+
"calls",
|
|
92
|
+
"tokens",
|
|
93
|
+
"cost",
|
|
94
|
+
"knownCostUsd"
|
|
95
|
+
], context);
|
|
96
|
+
for (const field of [
|
|
97
|
+
"calls",
|
|
98
|
+
"tokens",
|
|
99
|
+
"cost"
|
|
100
|
+
]) if (!Object.hasOwn(value, field)) throw new TypeError(`${context} ${field} is required`);
|
|
101
|
+
if (value.tokens !== null) {
|
|
102
|
+
if (!isRecord(value.tokens)) throw new TypeError(`${context} tokens must be an object or null`);
|
|
103
|
+
assertOnlyKeys(value.tokens, [
|
|
104
|
+
"input",
|
|
105
|
+
"output",
|
|
106
|
+
"reasoning",
|
|
107
|
+
"cached",
|
|
108
|
+
"cacheWrite"
|
|
109
|
+
], `${context} tokens`);
|
|
110
|
+
}
|
|
111
|
+
assertCostProvenance(value.cost, `${context} cost`);
|
|
112
|
+
assertValidAnalystUsageReceipt(value, context);
|
|
153
113
|
}
|
|
154
|
-
function
|
|
155
|
-
if (
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
if (
|
|
114
|
+
function assertCostProvenance(value, context) {
|
|
115
|
+
if (!isRecord(value)) throw new TypeError(`${context} must be an object`);
|
|
116
|
+
assertOnlyKeys(value, ["kind", "usd"], context);
|
|
117
|
+
if (value.kind === "uncaptured") {
|
|
118
|
+
if (value.usd !== null) throw new TypeError(`${context} uncaptured usd must be null`);
|
|
119
|
+
return;
|
|
159
120
|
}
|
|
160
|
-
|
|
121
|
+
if (value.kind !== "observed" && value.kind !== "estimated") throw new TypeError(`${context} kind is invalid`);
|
|
122
|
+
finiteNonnegative(value.usd, `${context} usd`);
|
|
161
123
|
}
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
const
|
|
170
|
-
const
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
|
|
181
|
-
|
|
182
|
-
const toolHistogram = {};
|
|
183
|
-
let hasSelfVerification = false;
|
|
184
|
-
for (const s of spans) {
|
|
185
|
-
const tool = toolNameOf(s);
|
|
186
|
-
if (tool) {
|
|
187
|
-
toolHistogram[tool] = (toolHistogram[tool] ?? 0) + 1;
|
|
188
|
-
if (VERIFY_RE.test(tool)) hasSelfVerification = true;
|
|
124
|
+
function sealExactAnalystRunReceipt(run, context) {
|
|
125
|
+
if (run.execution_plan === void 0) throw new TypeError(`${context} exact run requires execution_plan`);
|
|
126
|
+
const plan = snapshotExactExecutionPlan(run.execution_plan, `${context} execution_plan`);
|
|
127
|
+
const completion = snapshotExactRunCompletion(run.completion, `${context} completion`);
|
|
128
|
+
run.execution_plan = plan;
|
|
129
|
+
run.completion = completion;
|
|
130
|
+
const summaries = run.per_analyst;
|
|
131
|
+
const findings = run.findings;
|
|
132
|
+
const planned = plan.analysts.map((analyst) => analyst.id);
|
|
133
|
+
const completed = summaries.map((summary) => summary.analyst_id);
|
|
134
|
+
if (!completed.every((analystId, index) => analystId === planned[index]) || completion.status === "complete" && completed.length !== planned.length) throw new TypeError(completion.status === "complete" ? `${context} complete receipt must contain every execution_plan analyst in exact order` : `${context} failed receipt per_analyst must be an execution_plan prefix`);
|
|
135
|
+
const completedIds = new Set(completed);
|
|
136
|
+
for (const finding of findings) if (!completedIds.has(finding.analyst_id)) throw new TypeError(`${context} finding names an analyst absent from per_analyst`);
|
|
137
|
+
for (const summary of summaries) {
|
|
138
|
+
const actual = findings.filter((finding) => finding.analyst_id === summary.analyst_id).length;
|
|
139
|
+
if (summary.findings_count !== actual) throw new TypeError(`${context} findings_count does not match findings for "${summary.analyst_id}"`);
|
|
140
|
+
const hasAllocation = Object.hasOwn(summary, "allocated_budget_usd");
|
|
141
|
+
if (summary.status === "skipped") {
|
|
142
|
+
if (hasAllocation) throw new TypeError(`${context} skipped summary "${summary.analyst_id}" cannot report an allocated budget`);
|
|
143
|
+
continue;
|
|
189
144
|
}
|
|
145
|
+
const allocation = summary.allocated_budget_usd;
|
|
146
|
+
if (!(hasAllocation && (plan.policy.budget.kind === "none" ? allocation === null : typeof allocation === "number" && plan.policy.budget.allocations_usd[summary.analyst_id] !== null && plan.policy.budget.allocations_usd[summary.analyst_id] !== void 0 && allocation <= plan.policy.budget.allocations_usd[summary.analyst_id]))) throw new TypeError(`${context} summary "${summary.analyst_id}" allocation does not match its execution plan`);
|
|
190
147
|
}
|
|
191
|
-
|
|
192
|
-
const
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
const seenTokenSignals = /* @__PURE__ */ new Set();
|
|
196
|
-
for (const sequence of tokenSequences) for (const signal of tokenSignals(sequence)) {
|
|
197
|
-
if (seenTokenSignals.has(signal.code)) continue;
|
|
198
|
-
seenTokenSignals.add(signal.code);
|
|
199
|
-
signals.push(signal);
|
|
148
|
+
let knownCost = 0;
|
|
149
|
+
for (const summary of summaries) {
|
|
150
|
+
const amount = summary.usage.cost.kind === "uncaptured" ? summary.usage.knownCostUsd ?? 0 : summary.usage.cost.usd ?? 0;
|
|
151
|
+
knownCost = finiteNonnegative(knownCost + amount, `${context} aggregate known cost`);
|
|
200
152
|
}
|
|
201
|
-
if (
|
|
202
|
-
|
|
203
|
-
|
|
204
|
-
|
|
205
|
-
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
|
|
211
|
-
|
|
212
|
-
|
|
153
|
+
if (run.total_cost_usd !== knownCost) throw new TypeError(`${context} total_cost_usd does not match per_analyst usage`);
|
|
154
|
+
if (run.total_cost_provenance === void 0) throw new TypeError(`${context} exact run requires total_cost_provenance`);
|
|
155
|
+
const costs = summaries.map((summary) => summary.usage.cost);
|
|
156
|
+
const expectedProvenance = costs.some((cost) => cost.kind === "uncaptured") ? {
|
|
157
|
+
kind: "uncaptured",
|
|
158
|
+
usd: null
|
|
159
|
+
} : {
|
|
160
|
+
kind: costs.some((cost) => cost.kind === "estimated") ? "estimated" : "observed",
|
|
161
|
+
usd: costs.reduce((sum, cost) => finiteNonnegative(sum + (cost.usd ?? 0), `${context} aggregate captured cost`), 0)
|
|
162
|
+
};
|
|
163
|
+
if (hashCanonical(run.total_cost_provenance) !== hashCanonical(expectedProvenance)) throw new TypeError(`${context} total_cost_provenance does not match per_analyst usage`);
|
|
164
|
+
return deepFreezeCanonicalJson(run);
|
|
165
|
+
}
|
|
166
|
+
function snapshotExactRunCompletion(value, context) {
|
|
167
|
+
if (!isRecord(value)) throw new TypeError(`${context} must be an object`);
|
|
168
|
+
if (value.status === "complete") {
|
|
169
|
+
assertOnlyKeys(value, ["status"], context);
|
|
170
|
+
return value;
|
|
213
171
|
}
|
|
214
|
-
if (
|
|
215
|
-
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
|
|
172
|
+
if (value.status !== "failed") throw new TypeError(`${context} status must be complete or failed`);
|
|
173
|
+
assertOnlyKeys(value, ["status", "error"], context);
|
|
174
|
+
if (!isRecord(value.error)) throw new TypeError(`${context} failed receipt requires error`);
|
|
175
|
+
assertOnlyKeys(value.error, ["class", "message"], `${context} error`);
|
|
176
|
+
requiredString(value.error.class, `${context} error class`);
|
|
177
|
+
requiredString(value.error.message, `${context} error message`);
|
|
178
|
+
return value;
|
|
179
|
+
}
|
|
180
|
+
function finiteNonnegative(value, context) {
|
|
181
|
+
if (typeof value !== "number" || !Number.isFinite(value) || value < 0) throw new TypeError(`${context} must be a non-negative finite number`);
|
|
182
|
+
return value;
|
|
183
|
+
}
|
|
184
|
+
function nonnegativeSafeInteger(value, context) {
|
|
185
|
+
if (!Number.isSafeInteger(value) || value < 0) throw new TypeError(`${context} must be a non-negative safe integer`);
|
|
186
|
+
return value;
|
|
187
|
+
}
|
|
188
|
+
function snapshotAnalystFindings(value, context = "analyst run findings") {
|
|
189
|
+
if (!Array.isArray(value)) throw new TypeError(`${context} must be an array`);
|
|
190
|
+
const findings = value.map((finding, index) => snapshotAnalystFinding(finding, `${context} finding ${index}`));
|
|
191
|
+
assertUniqueFindingIds(findings.map((finding) => finding.finding_id));
|
|
192
|
+
return findings;
|
|
193
|
+
}
|
|
194
|
+
function readAnalystReview(trajectory) {
|
|
195
|
+
const analystAttempts = trajectory.attempts.filter((attempt) => isRecord(attempt.artifact) && attempt.artifact.type === "analyst-run");
|
|
196
|
+
const analysis = isRecord(trajectory.metadata?.analysis) ? trajectory.metadata.analysis : void 0;
|
|
197
|
+
if (analystAttempts.length === 0) {
|
|
198
|
+
if (analysis?.kind === "analyst-run") throw new TypeError(`feedbackTrajectoryToOptimizerRow: analyst trajectory "${trajectory.id}" is missing its archived run`);
|
|
199
|
+
return;
|
|
200
|
+
}
|
|
201
|
+
if (analystAttempts.length !== 1) throw new TypeError(`feedbackTrajectoryToOptimizerRow: analyst trajectory "${trajectory.id}" must contain exactly one archived run`);
|
|
202
|
+
if (analysis?.kind !== "analyst-run") throw new TypeError(`feedbackTrajectoryToOptimizerRow: analyst trajectory "${trajectory.id}" is missing review state`);
|
|
203
|
+
const artifact = analystAttempts[0].artifact;
|
|
204
|
+
if (!isRecord(artifact)) throw new TypeError(`feedbackTrajectoryToOptimizerRow: analyst trajectory "${trajectory.id}" has an invalid archived run`);
|
|
205
|
+
const runId = requiredString(artifact.analystRunId, `analyst trajectory "${trajectory.id}" run id`);
|
|
206
|
+
if (analysis.runId !== runId) throw new TypeError(`feedbackTrajectoryToOptimizerRow: analyst trajectory "${trajectory.id}" run identity does not match its review state`);
|
|
207
|
+
const artifactRunDigest = requiredDigest(artifact.runDigest, `analyst trajectory "${trajectory.id}" archived run digest`);
|
|
208
|
+
const storedRunDigest = requiredDigest(analysis.runDigest, `analyst trajectory "${trajectory.id}" review run digest`);
|
|
209
|
+
if (artifactRunDigest !== storedRunDigest) throw new TypeError(`feedbackTrajectoryToOptimizerRow: analyst trajectory "${trajectory.id}" run digest does not match its review state`);
|
|
210
|
+
const findings = snapshotAnalystFindings(artifact.findings, `analyst trajectory "${trajectory.id}"`);
|
|
211
|
+
const findingIds = findings.map((finding) => finding.finding_id);
|
|
212
|
+
const analystIds = stringArray(artifact.analystIds, `analyst trajectory "${trajectory.id}" analyst ids`);
|
|
213
|
+
const attemptMetadata = analystAttempts[0].metadata;
|
|
214
|
+
if (!isRecord(attemptMetadata)) throw new TypeError(`feedbackTrajectoryToOptimizerRow: analyst trajectory "${trajectory.id}" is missing archived run metadata`);
|
|
215
|
+
const archivedRun = snapshotAnalystRun({
|
|
216
|
+
run_id: runId,
|
|
217
|
+
correlation_id: artifact.correlationId,
|
|
218
|
+
started_at: analysis.startedAt,
|
|
219
|
+
ended_at: analysis.endedAt,
|
|
220
|
+
findings,
|
|
221
|
+
per_analyst: attemptMetadata.perAnalyst,
|
|
222
|
+
total_cost_usd: analysis.knownCostUsd,
|
|
223
|
+
...analysis.costProvenance === void 0 ? {} : { total_cost_provenance: analysis.costProvenance },
|
|
224
|
+
...artifact.executionPlan === void 0 ? {} : {
|
|
225
|
+
execution_plan: artifact.executionPlan,
|
|
226
|
+
completion: artifact.completion
|
|
221
227
|
}
|
|
228
|
+
}, `analyst trajectory "${trajectory.id}" archived run`);
|
|
229
|
+
const knownAnalystIds = new Set(analystIds);
|
|
230
|
+
for (const [index, finding] of findings.entries()) if (!knownAnalystIds.has(finding.analyst_id)) throw new TypeError(`feedbackTrajectoryToOptimizerRow: analyst trajectory "${trajectory.id}" omits generating analyst "${finding.analyst_id}" at finding ${index}`);
|
|
231
|
+
const reviewDecisions = validateAnalystReviewDecisions({
|
|
232
|
+
runId,
|
|
233
|
+
runDigest: storedRunDigest,
|
|
234
|
+
findings,
|
|
235
|
+
analystIds,
|
|
236
|
+
decisions: analysis.reviewDecisions,
|
|
237
|
+
requireComplete: true
|
|
222
238
|
});
|
|
239
|
+
const expectedRunDigest = analystRunDigest(archivedRun);
|
|
240
|
+
if (storedRunDigest !== expectedRunDigest) throw new TypeError(`feedbackTrajectoryToOptimizerRow: analyst trajectory "${trajectory.id}" archived run digest mismatch`);
|
|
223
241
|
return {
|
|
224
|
-
|
|
225
|
-
|
|
226
|
-
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
|
|
230
|
-
totalToolCalls,
|
|
231
|
-
distinctTools,
|
|
232
|
-
toolDiversityRatio,
|
|
233
|
-
hasSelfVerification,
|
|
234
|
-
signals
|
|
242
|
+
runId,
|
|
243
|
+
runDigest: expectedRunDigest,
|
|
244
|
+
findings,
|
|
245
|
+
findingIds,
|
|
246
|
+
analystIds,
|
|
247
|
+
reviewDecisions
|
|
235
248
|
};
|
|
236
249
|
}
|
|
237
|
-
function
|
|
238
|
-
const
|
|
239
|
-
const
|
|
240
|
-
|
|
241
|
-
|
|
242
|
-
|
|
243
|
-
const
|
|
244
|
-
const
|
|
245
|
-
|
|
246
|
-
|
|
247
|
-
|
|
248
|
-
|
|
249
|
-
|
|
250
|
-
|
|
251
|
-
|
|
252
|
-
|
|
253
|
-
|
|
254
|
-
|
|
255
|
-
|
|
256
|
-
|
|
257
|
-
|
|
258
|
-
sequences.push({
|
|
259
|
-
scopeId: runs.length === 1 ? scopeId : `${scopeId}#${index + 1}`,
|
|
260
|
-
spanIds: run.map((sample) => sample.span.span_id),
|
|
261
|
-
inputTokenTrajectory: run.map((sample) => sample.input),
|
|
262
|
-
outputTokenTrajectory: run.map((sample) => sample.output)
|
|
263
|
-
});
|
|
264
|
-
});
|
|
265
|
-
}
|
|
266
|
-
return sequences.sort((a, b) => b.spanIds.length - a.spanIds.length || a.scopeId.localeCompare(b.scopeId) || a.spanIds[0].localeCompare(b.spanIds[0]));
|
|
267
|
-
}
|
|
268
|
-
function createTokenExecutionScopeResolver(spansById) {
|
|
269
|
-
const cache = /* @__PURE__ */ new Map();
|
|
270
|
-
return (span) => {
|
|
271
|
-
const ancestry = resolveAncestorScope(span.parent_span_id, spansById, cache);
|
|
272
|
-
const rootId = ancestry.rootId;
|
|
273
|
-
const scopeId = ancestry.agentId ? `span:${ancestry.agentId}` : ancestry.missingParentId ? `parent:${ancestry.missingParentId}` : rootId ? `root:${rootId}` : span.agent_name ? `agent:${span.agent_name}` : `trace:${span.trace_id}`;
|
|
274
|
-
const scopeSpanId = ancestry.agentId ?? ancestry.missingParentId ?? rootId;
|
|
275
|
-
const laneSpan = ancestry.laneSpanId ? spansById.get(ancestry.laneSpanId) : void 0;
|
|
276
|
-
const direct = scopeSpanId === null || ancestry.laneSpanId === scopeSpanId;
|
|
277
|
-
const timedSpan = direct ? span : laneSpan;
|
|
278
|
-
return {
|
|
279
|
-
key: JSON.stringify([scopeId, direct ? span.span_id : ancestry.laneSpanId]),
|
|
280
|
-
scopeKey: scopeId,
|
|
281
|
-
scopeId,
|
|
282
|
-
start: timedSpan ? spanEpochMillis(timedSpan.start_time) : null,
|
|
283
|
-
end: timedSpan ? spanEpochMillis(timedSpan.end_time) : null
|
|
284
|
-
};
|
|
250
|
+
function completedAnalystReviewQuality(review) {
|
|
251
|
+
const findingDecisions = review.reviewDecisions.filter((decision) => decision.verdict !== "completeness_assessed");
|
|
252
|
+
const completeness = review.reviewDecisions.filter((decision) => decision.verdict === "completeness_assessed");
|
|
253
|
+
if (completeness.length !== 1) throw new TypeError("feedbackTrajectoryToOptimizerRow: analyst run requires exactly one independent completeness_assessed decision");
|
|
254
|
+
const confirmed = findingDecisions.filter((decision) => decision.verdict === "confirmed").length;
|
|
255
|
+
const rejected = findingDecisions.length - confirmed;
|
|
256
|
+
const emitted = review.findingIds.length;
|
|
257
|
+
const missed = completeness[0].missedIssues.length;
|
|
258
|
+
const precision = emitted === 0 ? 1 : confirmed / emitted;
|
|
259
|
+
const recallDenominator = confirmed + missed;
|
|
260
|
+
const recall = recallDenominator === 0 ? 1 : confirmed / recallDenominator;
|
|
261
|
+
return {
|
|
262
|
+
precision,
|
|
263
|
+
recall,
|
|
264
|
+
f1: precision + recall === 0 ? 0 : 2 * precision * recall / (precision + recall),
|
|
265
|
+
counts: {
|
|
266
|
+
emitted,
|
|
267
|
+
confirmed,
|
|
268
|
+
rejected,
|
|
269
|
+
missed
|
|
270
|
+
}
|
|
285
271
|
};
|
|
286
272
|
}
|
|
287
|
-
function
|
|
288
|
-
|
|
289
|
-
|
|
290
|
-
|
|
291
|
-
|
|
292
|
-
|
|
293
|
-
|
|
294
|
-
|
|
295
|
-
const
|
|
296
|
-
|
|
297
|
-
|
|
298
|
-
|
|
299
|
-
|
|
300
|
-
|
|
301
|
-
|
|
302
|
-
|
|
303
|
-
|
|
304
|
-
}
|
|
305
|
-
const
|
|
306
|
-
if (
|
|
307
|
-
|
|
308
|
-
|
|
309
|
-
|
|
310
|
-
|
|
311
|
-
|
|
312
|
-
|
|
313
|
-
|
|
314
|
-
|
|
315
|
-
|
|
316
|
-
|
|
317
|
-
|
|
318
|
-
|
|
319
|
-
|
|
320
|
-
|
|
321
|
-
|
|
322
|
-
|
|
323
|
-
|
|
324
|
-
|
|
325
|
-
|
|
326
|
-
|
|
327
|
-
|
|
328
|
-
|
|
329
|
-
missingParentId: null,
|
|
330
|
-
laneSpanId: current.span_id
|
|
273
|
+
function validateAnalystReviewDecisions(input) {
|
|
274
|
+
if (!Array.isArray(input.decisions)) throw new TypeError("analyst review decisions must be an array");
|
|
275
|
+
const findings = snapshotAnalystFindings(input.findings);
|
|
276
|
+
const expectedRunDigest = requiredDigest(input.runDigest, "analyst review run digest");
|
|
277
|
+
const findingsById = new Map(findings.map((finding) => [finding.finding_id, finding]));
|
|
278
|
+
const generatingAnalystIds = new Set(input.analystIds);
|
|
279
|
+
const seenFindingIds = /* @__PURE__ */ new Set();
|
|
280
|
+
let completenessCount = 0;
|
|
281
|
+
const decisions = input.decisions.map((value, index) => {
|
|
282
|
+
if (!isRecord(value)) throw new TypeError(`analyst review decision ${index} must be an object`);
|
|
283
|
+
const source = requiredString(value.source, `analyst review decision ${index} source`);
|
|
284
|
+
if (!isAnalystReviewSource(source)) throw new TypeError(`analyst review decision ${index} source must be user, judge, environment, metric, or policy`);
|
|
285
|
+
const reviewerId = requiredString(value.reviewerId, `analyst review decision ${index} reviewerId`);
|
|
286
|
+
if (generatingAnalystIds.has(reviewerId)) throw new TypeError(`analyst review decision ${index} reviewerId must differ from the generating analyst`);
|
|
287
|
+
const reviewId = requiredString(value.reviewId, `analyst review decision ${index} reviewId`);
|
|
288
|
+
if (reviewId === input.runId) throw new TypeError(`analyst review decision ${index} reviewId must identify an independent review`);
|
|
289
|
+
const reason = requiredString(value.reason, `analyst review decision ${index} reason`);
|
|
290
|
+
const decidedAt = canonicalTimestamp(value.decidedAt, `analyst review decision ${index} decidedAt`);
|
|
291
|
+
const runDigest = requiredDigest(value.runDigest, `analyst review decision ${index} runDigest`);
|
|
292
|
+
if (runDigest !== expectedRunDigest) throw new TypeError(`analyst review decision ${index} run digest mismatch`);
|
|
293
|
+
if (value.verdict === "completeness_assessed") {
|
|
294
|
+
assertOnlyKeys(value, [
|
|
295
|
+
"runDigest",
|
|
296
|
+
"verdict",
|
|
297
|
+
"missedIssues",
|
|
298
|
+
"source",
|
|
299
|
+
"reviewerId",
|
|
300
|
+
"reviewId",
|
|
301
|
+
"reason",
|
|
302
|
+
"decidedAt"
|
|
303
|
+
], `analyst review decision ${index}`);
|
|
304
|
+
completenessCount += 1;
|
|
305
|
+
if (completenessCount > 1) throw new TypeError("duplicate completeness_assessed analyst review decision");
|
|
306
|
+
return {
|
|
307
|
+
runDigest,
|
|
308
|
+
verdict: "completeness_assessed",
|
|
309
|
+
missedIssues: validateMissedIssues(value.missedIssues, findingsById, `analyst review decision ${index}`),
|
|
310
|
+
source,
|
|
311
|
+
reviewerId,
|
|
312
|
+
reviewId,
|
|
313
|
+
reason,
|
|
314
|
+
decidedAt
|
|
331
315
|
};
|
|
332
|
-
break;
|
|
333
316
|
}
|
|
334
|
-
|
|
335
|
-
|
|
336
|
-
|
|
337
|
-
|
|
338
|
-
|
|
339
|
-
|
|
340
|
-
|
|
341
|
-
|
|
342
|
-
|
|
343
|
-
|
|
344
|
-
|
|
345
|
-
|
|
346
|
-
|
|
317
|
+
if (value.verdict !== "confirmed" && value.verdict !== "rejected") throw new TypeError(`analyst review decision ${index} verdict must be confirmed, rejected, or completeness_assessed`);
|
|
318
|
+
assertOnlyKeys(value, [
|
|
319
|
+
"runDigest",
|
|
320
|
+
"findingId",
|
|
321
|
+
"findingDigest",
|
|
322
|
+
"verdict",
|
|
323
|
+
"source",
|
|
324
|
+
"reviewerId",
|
|
325
|
+
"reviewId",
|
|
326
|
+
"reason",
|
|
327
|
+
"decidedAt"
|
|
328
|
+
], `analyst review decision ${index}`);
|
|
329
|
+
const findingId = requiredString(value.findingId, `analyst review decision ${index} findingId`);
|
|
330
|
+
const finding = findingsById.get(findingId);
|
|
331
|
+
if (!finding) throw new TypeError(`analyst review decision references unknown finding id "${findingId}"`);
|
|
332
|
+
if (seenFindingIds.has(findingId)) throw new TypeError(`duplicate analyst review decision for finding id "${findingId}"`);
|
|
333
|
+
seenFindingIds.add(findingId);
|
|
334
|
+
const findingDigest = requiredString(value.findingDigest, `analyst review decision ${index} findingDigest`);
|
|
335
|
+
const expectedDigest = analystFindingDigest(finding);
|
|
336
|
+
if (findingDigest !== expectedDigest) throw new TypeError(`analyst review decision ${index} digest mismatch for finding id "${findingId}"`);
|
|
337
|
+
return {
|
|
338
|
+
runDigest,
|
|
339
|
+
findingId,
|
|
340
|
+
findingDigest: expectedDigest,
|
|
341
|
+
verdict: value.verdict,
|
|
342
|
+
source,
|
|
343
|
+
reviewerId,
|
|
344
|
+
reviewId,
|
|
345
|
+
reason,
|
|
346
|
+
decidedAt
|
|
347
347
|
};
|
|
348
|
-
|
|
348
|
+
});
|
|
349
|
+
if (input.requireComplete) {
|
|
350
|
+
const missing = findings.map((finding) => finding.finding_id).filter((findingId) => !seenFindingIds.has(findingId));
|
|
351
|
+
if (missing.length > 0) throw new TypeError(`feedbackTrajectoryToOptimizerRow: missing independent decisions for finding ids: ${missing.join(", ")}`);
|
|
352
|
+
if (completenessCount !== 1) throw new TypeError("feedbackTrajectoryToOptimizerRow: analyst run requires exactly one independent completeness_assessed decision");
|
|
349
353
|
}
|
|
350
|
-
return
|
|
351
|
-
}
|
|
352
|
-
function compareTokenSamples(a, b) {
|
|
353
|
-
const aStart = spanEpochMillis(a.span.start_time);
|
|
354
|
-
const bStart = spanEpochMillis(b.span.start_time);
|
|
355
|
-
if (aStart === null && bStart !== null) return 1;
|
|
356
|
-
if (aStart !== null && bStart === null) return -1;
|
|
357
|
-
if (aStart !== null && bStart !== null && aStart !== bStart) return aStart - bStart;
|
|
358
|
-
if (a.step !== null && b.step !== null && a.step !== b.step) return a.step - b.step;
|
|
359
|
-
return a.span.span_id.localeCompare(b.span.span_id);
|
|
354
|
+
return decisions;
|
|
360
355
|
}
|
|
361
|
-
function
|
|
362
|
-
const
|
|
363
|
-
|
|
364
|
-
|
|
365
|
-
|
|
366
|
-
|
|
367
|
-
if (serial.length > 0) runs.push(serial);
|
|
368
|
-
serial = [];
|
|
369
|
-
};
|
|
370
|
-
const flushOverlap = () => {
|
|
371
|
-
if (overlap.length === 1) serial.push(overlap[0]);
|
|
372
|
-
else if (overlap.length > 1) {
|
|
373
|
-
flushSerial();
|
|
374
|
-
for (const sample of overlap) runs.push([sample]);
|
|
375
|
-
}
|
|
376
|
-
overlap = [];
|
|
377
|
-
overlapEnd = Number.NEGATIVE_INFINITY;
|
|
378
|
-
};
|
|
379
|
-
for (const sample of ordered) {
|
|
380
|
-
const start = spanEpochMillis(sample.span.start_time);
|
|
381
|
-
const end = spanEpochMillis(sample.span.end_time);
|
|
382
|
-
if (start === null || end === null || sample.span.duration_ms <= 0 || end < start) {
|
|
383
|
-
flushOverlap();
|
|
384
|
-
flushSerial();
|
|
385
|
-
runs.push([sample]);
|
|
386
|
-
continue;
|
|
387
|
-
}
|
|
388
|
-
if (overlap.length > 0 && start >= overlapEnd) flushOverlap();
|
|
389
|
-
overlap.push(sample);
|
|
390
|
-
overlapEnd = Math.max(overlapEnd, end);
|
|
356
|
+
function assertUniqueFindingIds(findingIds) {
|
|
357
|
+
const seen = /* @__PURE__ */ new Set();
|
|
358
|
+
for (const findingId of findingIds) {
|
|
359
|
+
if (findingId.trim().length === 0) throw new TypeError("analyst finding id must not be empty");
|
|
360
|
+
if (seen.has(findingId)) throw new TypeError(`analyst run contains duplicate finding id "${findingId}"`);
|
|
361
|
+
seen.add(findingId);
|
|
391
362
|
}
|
|
392
|
-
flushOverlap();
|
|
393
|
-
flushSerial();
|
|
394
|
-
return runs;
|
|
395
363
|
}
|
|
396
|
-
function
|
|
397
|
-
|
|
398
|
-
|
|
399
|
-
|
|
400
|
-
|
|
401
|
-
|
|
402
|
-
const last = inputs[inputs.length - 1];
|
|
403
|
-
const isMonotonic = everyAdjacent(inputs, (previous, current) => current >= previous);
|
|
404
|
-
const growthFromZero = first === 0 && last > 0;
|
|
405
|
-
const growth = growthFromZero ? Infinity : first > 0 ? last / first : 0;
|
|
406
|
-
if (isMonotonic && last > first && growth >= INPUT_GROWTH_FACTOR) {
|
|
407
|
-
const growthLabel = growthFromZero ? "0→nonzero (unbounded)" : `${growth.toFixed(1)}x`;
|
|
408
|
-
signals.push({
|
|
409
|
-
code: "monotonic-input-growth",
|
|
410
|
-
severity: "high",
|
|
411
|
-
detail: `LLM input tokens grew ${growthLabel} (${first}→${last}) across ${inputs.length} serial calls without an intervening decrease.`,
|
|
412
|
-
evidence: {
|
|
413
|
-
first,
|
|
414
|
-
last,
|
|
415
|
-
growth_x: growthFromZero ? "unbounded" : Number(growth.toFixed(2)),
|
|
416
|
-
calls: inputs.length,
|
|
417
|
-
scope: sequence.scopeId,
|
|
418
|
-
first_span_id: sequence.spanIds[0],
|
|
419
|
-
last_span_id: sequence.spanIds[sequence.spanIds.length - 1]
|
|
420
|
-
}
|
|
421
|
-
});
|
|
422
|
-
}
|
|
423
|
-
}
|
|
424
|
-
if (inputs.length >= OUTPUT_DECAY_MINIMUM_CALLS && inputs.length === outputs.length && inputs.every((value) => value !== null) && outputs.every((value) => value !== null)) {
|
|
425
|
-
const first = outputs[0];
|
|
426
|
-
const last = outputs[outputs.length - 1];
|
|
427
|
-
const inputIsMonotonic = everyAdjacent(inputs, (previous, current) => current >= previous);
|
|
428
|
-
const outputIsMonotonic = everyAdjacent(outputs, (previous, current) => current <= previous);
|
|
429
|
-
const inputGrew = inputs[inputs.length - 1] > inputs[0];
|
|
430
|
-
const decayIsMaterial = last <= first * OUTPUT_DECAY_MAXIMUM_RETAINED_FRACTION;
|
|
431
|
-
if (inputIsMonotonic && inputGrew && outputIsMonotonic && decayIsMaterial) signals.push({
|
|
432
|
-
code: "output-length-decay",
|
|
433
|
-
severity: "medium",
|
|
434
|
-
detail: `LLM output tokens shrank ${first}→${last} over ${outputs.length} serial calls while input tokens increased monotonically.`,
|
|
435
|
-
evidence: {
|
|
436
|
-
first,
|
|
437
|
-
last,
|
|
438
|
-
calls: outputs.length,
|
|
439
|
-
scope: sequence.scopeId,
|
|
440
|
-
first_span_id: sequence.spanIds[0],
|
|
441
|
-
last_span_id: sequence.spanIds[sequence.spanIds.length - 1]
|
|
442
|
-
}
|
|
443
|
-
});
|
|
364
|
+
function snapshotAnalystFinding(value, context) {
|
|
365
|
+
let snapshot;
|
|
366
|
+
try {
|
|
367
|
+
snapshot = JSON.parse(canonicalString(value));
|
|
368
|
+
} catch (cause) {
|
|
369
|
+
throw new TypeError(`${context} must have a canonical JSON representation`, { cause });
|
|
444
370
|
}
|
|
445
|
-
|
|
446
|
-
|
|
447
|
-
function everyAdjacent(values, predicate) {
|
|
448
|
-
return values.slice(1).every((current, index) => predicate(values[index], current));
|
|
371
|
+
assertAnalystFinding(snapshot, context);
|
|
372
|
+
return snapshot;
|
|
449
373
|
}
|
|
450
|
-
|
|
451
|
-
|
|
452
|
-
|
|
453
|
-
|
|
454
|
-
|
|
455
|
-
|
|
456
|
-
|
|
457
|
-
|
|
458
|
-
|
|
459
|
-
|
|
460
|
-
|
|
461
|
-
|
|
462
|
-
|
|
463
|
-
|
|
464
|
-
|
|
465
|
-
|
|
466
|
-
|
|
467
|
-
|
|
468
|
-
|
|
469
|
-
|
|
470
|
-
|
|
471
|
-
|
|
472
|
-
};
|
|
473
|
-
|
|
474
|
-
|
|
475
|
-
|
|
476
|
-
|
|
477
|
-
|
|
478
|
-
|
|
479
|
-
|
|
480
|
-
|
|
481
|
-
|
|
482
|
-
|
|
483
|
-
|
|
484
|
-
if (page.total !== expectedTotal) throw new Error(`behavioralAnalyst: trace count changed during pagination (${expectedTotal} to ${page.total})`);
|
|
485
|
-
if (page.total > maxTraces) throw new RangeError(`behavioralAnalyst: ${page.total} traces exceed maxTraces=${maxTraces}; filter the store or raise the explicit limit`);
|
|
486
|
-
for (const trace of page.traces) traceIds.add(trace.trace_id);
|
|
487
|
-
if (traceIds.size > maxTraces) throw new RangeError(`behavioralAnalyst: more than maxTraces=${maxTraces} unique traces were returned`);
|
|
488
|
-
if (!page.has_more) break;
|
|
489
|
-
if (page.traces.length === 0) throw new Error("behavioralAnalyst: trace store returned an empty page with has_more=true");
|
|
490
|
-
offset += page.traces.length;
|
|
491
|
-
}
|
|
492
|
-
if (traceIds.size !== expectedTotal) throw new Error(`behavioralAnalyst: pagination returned ${traceIds.size}/${expectedTotal ?? 0} unique traces`);
|
|
493
|
-
return [...traceIds].sort();
|
|
374
|
+
function assertAnalystFinding(value, context) {
|
|
375
|
+
if (!isRecord(value)) throw new TypeError(`${context} must be an object`);
|
|
376
|
+
assertOnlyKeys(value, [
|
|
377
|
+
"schema_version",
|
|
378
|
+
"finding_id",
|
|
379
|
+
"analyst_id",
|
|
380
|
+
"produced_at",
|
|
381
|
+
"severity",
|
|
382
|
+
"area",
|
|
383
|
+
"claim",
|
|
384
|
+
"rationale",
|
|
385
|
+
"evidence_refs",
|
|
386
|
+
"recommended_action",
|
|
387
|
+
"validation_plan",
|
|
388
|
+
"confidence",
|
|
389
|
+
"subject",
|
|
390
|
+
"derived_from_judge",
|
|
391
|
+
"metadata"
|
|
392
|
+
], context);
|
|
393
|
+
if (value.schema_version !== "1.0.0") throw new TypeError(`${context} schema_version must be "1.0.0"`);
|
|
394
|
+
requiredString(value.finding_id, `${context} finding_id`);
|
|
395
|
+
requiredString(value.analyst_id, `${context} analyst_id`);
|
|
396
|
+
canonicalTimestamp(value.produced_at, `${context} produced_at`);
|
|
397
|
+
if (value.severity !== "critical" && value.severity !== "high" && value.severity !== "medium" && value.severity !== "low" && value.severity !== "info") throw new TypeError(`${context} severity is invalid`);
|
|
398
|
+
requiredString(value.area, `${context} area`);
|
|
399
|
+
requiredString(value.claim, `${context} claim`);
|
|
400
|
+
optionalString(value.rationale, `${context} rationale`);
|
|
401
|
+
value.evidence_refs = validateEvidenceRefs(value.evidence_refs, `${context} evidence_refs`);
|
|
402
|
+
optionalString(value.recommended_action, `${context} recommended_action`);
|
|
403
|
+
optionalString(value.validation_plan, `${context} validation_plan`);
|
|
404
|
+
if (typeof value.confidence !== "number" || !Number.isFinite(value.confidence) || value.confidence < 0 || value.confidence > 1) throw new TypeError(`${context} confidence must be a finite number from 0 through 1`);
|
|
405
|
+
optionalString(value.subject, `${context} subject`);
|
|
406
|
+
if (value.derived_from_judge !== void 0 && typeof value.derived_from_judge !== "boolean") throw new TypeError(`${context} derived_from_judge must be a boolean`);
|
|
407
|
+
if (value.metadata !== void 0 && !isRecord(value.metadata)) throw new TypeError(`${context} metadata must be an object`);
|
|
494
408
|
}
|
|
495
|
-
|
|
496
|
-
|
|
497
|
-
|
|
498
|
-
|
|
499
|
-
|
|
500
|
-
|
|
501
|
-
|
|
502
|
-
|
|
503
|
-
|
|
504
|
-
|
|
505
|
-
|
|
506
|
-
|
|
507
|
-
|
|
508
|
-
|
|
509
|
-
|
|
510
|
-
|
|
511
|
-
|
|
512
|
-
|
|
513
|
-
|
|
514
|
-
|
|
515
|
-
|
|
516
|
-
deterministic: true,
|
|
517
|
-
evidence: sig.evidence,
|
|
518
|
-
...traceId ? { trace_id: traceId } : {}
|
|
519
|
-
},
|
|
520
|
-
id_basis: sig.code,
|
|
521
|
-
...opts.producedAt ? { produced_at: opts.producedAt } : {}
|
|
522
|
-
}));
|
|
409
|
+
function validateMissedIssues(value, findingsById, context) {
|
|
410
|
+
if (!Array.isArray(value)) throw new TypeError(`${context} missedIssues must be an array`);
|
|
411
|
+
const seen = /* @__PURE__ */ new Set();
|
|
412
|
+
return value.map((issue, index) => {
|
|
413
|
+
const issueContext = `${context} missedIssues ${index}`;
|
|
414
|
+
if (!isRecord(issue)) throw new TypeError(`${issueContext} must be an object`);
|
|
415
|
+
assertOnlyKeys(issue, [
|
|
416
|
+
"id",
|
|
417
|
+
"reason",
|
|
418
|
+
"evidence"
|
|
419
|
+
], issueContext);
|
|
420
|
+
const id = requiredString(issue.id, `${issueContext} id`);
|
|
421
|
+
if (findingsById.has(id)) throw new TypeError(`${issueContext} id "${id}" is already an emitted finding id`);
|
|
422
|
+
if (seen.has(id)) throw new TypeError(`duplicate missed issue id "${id}"`);
|
|
423
|
+
seen.add(id);
|
|
424
|
+
return {
|
|
425
|
+
id,
|
|
426
|
+
reason: requiredString(issue.reason, `${issueContext} reason`),
|
|
427
|
+
...issue.evidence === void 0 ? {} : { evidence: validateEvidenceRefs(issue.evidence, `${issueContext} evidence`) }
|
|
428
|
+
};
|
|
429
|
+
});
|
|
523
430
|
}
|
|
524
|
-
|
|
525
|
-
|
|
526
|
-
|
|
527
|
-
|
|
528
|
-
|
|
529
|
-
|
|
530
|
-
|
|
531
|
-
|
|
532
|
-
|
|
533
|
-
|
|
534
|
-
|
|
535
|
-
|
|
536
|
-
|
|
537
|
-
|
|
538
|
-
|
|
539
|
-
|
|
540
|
-
|
|
541
|
-
|
|
542
|
-
|
|
543
|
-
|
|
544
|
-
const viewed = await store.viewTrace({ trace_id: traceId });
|
|
545
|
-
if (viewed.trace_id !== traceId) throw new Error(`behavioralAnalyst: requested trace '${traceId}', received '${viewed.trace_id}'`);
|
|
546
|
-
if (!viewed.spans) throw new Error(`behavioralAnalyst: trace '${traceId}' is oversized; complete spans are required`);
|
|
547
|
-
const metrics = computeTraceMetrics(viewed.spans);
|
|
548
|
-
if (metrics.traceId !== null && metrics.traceId !== traceId) throw new Error(`behavioralAnalyst: requested trace '${traceId}', received '${metrics.traceId}'`);
|
|
549
|
-
for (const finding of deriveEfficiencyFindings(metrics)) {
|
|
550
|
-
const current = findingsById.get(finding.finding_id);
|
|
551
|
-
if (!current) {
|
|
552
|
-
findingsById.set(finding.finding_id, {
|
|
553
|
-
finding,
|
|
554
|
-
observedTraceCount: 1,
|
|
555
|
-
evidenceTraceIds: [traceId],
|
|
556
|
-
evidence: [...finding.evidence_refs]
|
|
557
|
-
});
|
|
558
|
-
continue;
|
|
559
|
-
}
|
|
560
|
-
current.observedTraceCount += 1;
|
|
561
|
-
if (current.evidence.length < maxEvidenceRefsPerFinding) {
|
|
562
|
-
current.evidenceTraceIds.push(traceId);
|
|
563
|
-
current.evidence.push(...finding.evidence_refs);
|
|
564
|
-
}
|
|
565
|
-
}
|
|
566
|
-
}
|
|
567
|
-
return [...findingsById.values()].map(({ finding, observedTraceCount, evidenceTraceIds, evidence }) => ({
|
|
568
|
-
...finding,
|
|
569
|
-
claim: AGGREGATE_CLAIM[finding.subject](observedTraceCount, analyzedTraceIds.length),
|
|
570
|
-
rationale: `${observedTraceCount}/${analyzedTraceIds.length} analyzed traces exhibited this pattern.`,
|
|
571
|
-
evidence_refs: evidence,
|
|
572
|
-
metadata: {
|
|
573
|
-
deterministic: true,
|
|
574
|
-
evidence_trace_ids: evidenceTraceIds,
|
|
575
|
-
omitted_evidence_trace_count: observedTraceCount - evidenceTraceIds.length,
|
|
576
|
-
observed_trace_count: observedTraceCount,
|
|
577
|
-
analyzed_trace_count: analyzedTraceIds.length
|
|
578
|
-
}
|
|
579
|
-
}));
|
|
580
|
-
}
|
|
581
|
-
};
|
|
431
|
+
function validateEvidenceRefs(value, context) {
|
|
432
|
+
if (!Array.isArray(value)) throw new TypeError(`${context} must be an array`);
|
|
433
|
+
return value.map((evidence, index) => {
|
|
434
|
+
const evidenceContext = `${context} ${index}`;
|
|
435
|
+
if (!isRecord(evidence)) throw new TypeError(`${evidenceContext} must be an object`);
|
|
436
|
+
assertOnlyKeys(evidence, [
|
|
437
|
+
"kind",
|
|
438
|
+
"uri",
|
|
439
|
+
"excerpt"
|
|
440
|
+
], evidenceContext);
|
|
441
|
+
if (evidence.kind !== "span" && evidence.kind !== "event" && evidence.kind !== "artifact" && evidence.kind !== "finding" && evidence.kind !== "metric") throw new TypeError(`${evidenceContext} kind is invalid`);
|
|
442
|
+
const uri = requiredString(evidence.uri, `${evidenceContext} uri`);
|
|
443
|
+
const excerpt = evidence.excerpt;
|
|
444
|
+
optionalString(excerpt, `${evidenceContext} excerpt`);
|
|
445
|
+
return {
|
|
446
|
+
kind: evidence.kind,
|
|
447
|
+
uri,
|
|
448
|
+
...excerpt === void 0 ? {} : { excerpt }
|
|
449
|
+
};
|
|
450
|
+
});
|
|
582
451
|
}
|
|
583
|
-
function
|
|
584
|
-
|
|
452
|
+
function assertOnlyKeys(value, allowed, name) {
|
|
453
|
+
const allowedKeys = new Set(allowed);
|
|
454
|
+
const unexpected = Object.keys(value).filter((key) => !allowedKeys.has(key));
|
|
455
|
+
if (unexpected.length > 0) throw new TypeError(`${name} contains unknown fields: ${unexpected.sort().join(", ")}`);
|
|
456
|
+
}
|
|
457
|
+
function stringArray(value, name) {
|
|
458
|
+
if (!Array.isArray(value) || value.some((item) => typeof item !== "string")) throw new TypeError(`${name} must be an array of strings`);
|
|
459
|
+
const strings = value.map((item) => requiredString(item, name));
|
|
460
|
+
if (new Set(strings).size !== strings.length) throw new TypeError(`${name} must contain unique values`);
|
|
461
|
+
return strings;
|
|
462
|
+
}
|
|
463
|
+
function requiredString(value, name) {
|
|
464
|
+
if (typeof value !== "string" || value.trim().length === 0) throw new TypeError(`${name} must be a non-empty string`);
|
|
585
465
|
return value;
|
|
586
466
|
}
|
|
587
|
-
|
|
588
|
-
|
|
589
|
-
|
|
590
|
-
|
|
591
|
-
if (value === void 0) return "<absent>";
|
|
592
|
-
const encoded = JSON.stringify(value);
|
|
593
|
-
return encoded === void 0 ? String(value) : encoded;
|
|
467
|
+
function requiredDigest(value, name) {
|
|
468
|
+
const digest = requiredString(value, name);
|
|
469
|
+
if (!/^sha256:[a-f0-9]{64}$/.test(digest)) throw new TypeError(`${name} must be a sha256 digest`);
|
|
470
|
+
return digest;
|
|
594
471
|
}
|
|
595
|
-
function
|
|
596
|
-
|
|
597
|
-
kind: "metric",
|
|
598
|
-
uri: `supervisor-run://${encodeURIComponent(namespace)}/${value.path}`,
|
|
599
|
-
excerpt: shown(value.value)
|
|
600
|
-
};
|
|
472
|
+
function optionalString(value, name) {
|
|
473
|
+
if (value !== void 0 && typeof value !== "string") throw new TypeError(`${name} must be a string`);
|
|
601
474
|
}
|
|
602
|
-
|
|
603
|
-
|
|
604
|
-
const
|
|
605
|
-
|
|
606
|
-
|
|
607
|
-
|
|
608
|
-
|
|
609
|
-
|
|
610
|
-
|
|
611
|
-
|
|
612
|
-
|
|
613
|
-
evidence_refs: issue.evidence.map((value) => evidenceRef(report.runRef, value)),
|
|
614
|
-
recommended_action: issue.recommendedAction,
|
|
615
|
-
validation_plan: "Re-run this deterministic analyst on the retained SupervisorRunSources or SupervisorRunTree after correcting the producer.",
|
|
616
|
-
confidence: 1,
|
|
617
|
-
metadata: {
|
|
618
|
-
integrity_code: issue.code,
|
|
619
|
-
integrity_input: report.input,
|
|
620
|
-
integrity_run_ref: report.runRef,
|
|
621
|
-
integrity_subject: issue.subject,
|
|
622
|
-
...issue.metadata
|
|
623
|
-
}
|
|
624
|
-
}));
|
|
475
|
+
function canonicalTimestamp(value, name) {
|
|
476
|
+
const timestamp = requiredString(value, name);
|
|
477
|
+
const parsed = new Date(timestamp);
|
|
478
|
+
if (Number.isNaN(parsed.valueOf()) || parsed.toISOString() !== timestamp) throw new TypeError(`${name} must be a canonical ISO 8601 UTC timestamp`);
|
|
479
|
+
return timestamp;
|
|
480
|
+
}
|
|
481
|
+
function isAnalystReviewSource(value) {
|
|
482
|
+
return value === "user" || value === "judge" || value === "environment" || value === "metric" || value === "policy";
|
|
483
|
+
}
|
|
484
|
+
function isRecord(value) {
|
|
485
|
+
return typeof value === "object" && value !== null && !Array.isArray(value);
|
|
625
486
|
}
|
|
626
|
-
/** Deterministic Analyst adapter for `SupervisorRunSources | SupervisorRunTree`. */
|
|
627
|
-
var ControlIntegrityAnalyst = class {
|
|
628
|
-
id = ANALYST_ID;
|
|
629
|
-
description = "Deterministic supervisor-run integrity checks with explicit unavailable evidence.";
|
|
630
|
-
inputKind = "custom";
|
|
631
|
-
cost = {
|
|
632
|
-
kind: "deterministic",
|
|
633
|
-
est_usd_per_run: 0
|
|
634
|
-
};
|
|
635
|
-
version = "2.0.0";
|
|
636
|
-
executionConfig = {
|
|
637
|
-
kind: "control-integrity",
|
|
638
|
-
produced_at_source: "tags.producedAt-or-system-clock"
|
|
639
|
-
};
|
|
640
|
-
async analyze(input, ctx) {
|
|
641
|
-
const findings = emitControlIntegrityFindings(input, ctx.tags?.producedAt ?? (/* @__PURE__ */ new Date()).toISOString());
|
|
642
|
-
ctx.log?.(`control-integrity: ${findings.length} finding(s)`, { input: "nodes" in input ? "SupervisorRunTree" : "SupervisorRunSources" });
|
|
643
|
-
return findings;
|
|
644
|
-
}
|
|
645
|
-
};
|
|
646
|
-
const CONTROL_INTEGRITY_ANALYST = new ControlIntegrityAnalyst();
|
|
647
|
-
const FAILURE_MODE_KIND_SPEC = {
|
|
648
|
-
id: "failure-mode",
|
|
649
|
-
description: "Ranks failures by the work each one burned — spans, model calls, and wall-clock between the error and the span where the agent resumed — and treats a failure that reached a human as the most expensive class.",
|
|
650
|
-
area: "failure-mode",
|
|
651
|
-
version: "2.0.0",
|
|
652
|
-
instructions: `You are a failure-cost analyst for an OTLP trace dataset. Your job is to find the failures that **burned the most agent work** and to price each one in spans, model calls, and wall-clock. How often an error occurred is not its cost and is not your output.
|
|
653
|
-
|
|
654
|
-
${findingSubjectGrammarPromptFor("failure-mode")}
|
|
655
|
-
|
|
656
|
-
Use a lowercase cluster label that names the cost shape, not the error string: "human-corrected-file-overwrite", "unrecovered-auth-loop", "long-detour-schema-guess".
|
|
657
|
-
|
|
658
|
-
WORKLIST → RECOVERY BOUNDARY → PRICE → CITE protocol:
|
|
659
|
-
|
|
660
|
-
1. \`getDatasetOverview({})\` first. Read \`errors.trace_count\`, \`errors.span_count\`, and \`error_clusters[]\` — each cluster carries \`signature\`, \`tool_name\`, \`trace_count\`, \`span_count\`, \`prevalence\`, \`exemplar_trace_ids\`, and \`exemplar_span_ids\`. This is the deterministic analyzer's output, already computed for free. It is your WORKLIST: the exemplar ids name the spans you go price. It is never your findings.
|
|
661
|
-
2. Reach error-bearing traces the exemplars miss with \`queryTraces(filters={"has_errors": true}, limit=...)\`. Prefer long traces and traces whose errors sit early: a failure has room to be expensive only when work followed it.
|
|
662
|
-
3. Take one candidate error span E and read its trace in order. \`viewTrace\` returns that trace's spans sorted by \`start_time\`, which is the ordering every distance below is counted in. When \`viewTrace\` returns \`oversized\` instead of spans, locate candidates with \`searchTrace\` and pull the exact ids with \`viewSpans\`; use \`searchSpan\` when one span's payload is truncated.
|
|
663
|
-
4. Find the RECOVERY BOUNDARY R — the first span after E in which the agent resumed productive work. Productive means a different, task-advancing action. A retry of the same call with the same arguments, a re-read of the same file, a re-plan of the same step, and an apology are all still inside the failure. Then price it:
|
|
664
|
-
- spans_burned: spans strictly between E and R
|
|
665
|
-
- model_calls_burned: how many of those spans have kind \`LLM\`
|
|
666
|
-
- wall_clock_ms: R \`end_time\` minus E \`start_time\`
|
|
667
|
-
Report all three. If the agent never resumed, R is the trace's last span and the failure is UNRECOVERED.
|
|
668
|
-
5. Set severity from what ENDED the failure and from the measured distance, never from how many times the signature appears:
|
|
669
|
-
- HUMAN-CORRECTED — a human turn after E supplies the correction. This is the most expensive class at any span count, because the failure escaped the agent entirely and reached the user: severity critical. Detect it as a new user-role message in a later \`LLM\` span's input attributes, or a new root-level \`AGENT\` span in which the user restates or repairs the task. Search with patterns like \`"role"\\s*:\\s*"user"\`, \`(that|this) is (wrong|not what)\`, \`I (said|asked|told you)\`, \`stop\`, \`undo\`, \`revert\`, \`you (deleted|broke|missed|ignored)\`, \`try again\`. Quote the human's words exactly.
|
|
670
|
-
- UNRECOVERED — the trace ends without resumption: severity critical.
|
|
671
|
-
- LONG DETOUR — 10 or more spans burned, or 3 or more \`LLM\` spans burned: severity high.
|
|
672
|
-
- SHORT DETOUR — 2 to 9 spans burned: severity medium.
|
|
673
|
-
- SELF-RECOVERED ON THE NEXT SPAN — cost is approximately zero. Do not emit it. That is resilience, and reporting it dilutes the ranking.
|
|
674
|
-
One error that cost 20 spans outranks 50 errors that each cost one retry. Frequency may appear in a finding only as a multiplier on a measured per-instance cost, never as the reason for its severity.
|
|
675
|
-
6. **Cluster, do not enumerate.** Errors sharing a root cause AND a recovery shape are ONE finding: report the summed cost across instances, state how many instances it covers, and cite the bounding pair of the single most expensive instance. Two errors with the same signature but different recovery shapes — one shrugged off, one human-corrected — are NOT the same finding; the expensive one is the finding and the cheap one is noise.
|
|
676
|
-
|
|
677
|
-
FORBIDDEN OUTPUT. Each of the following duplicates the deterministic pass, and a duplicate finding is worse than no finding because it costs a reviewer the same attention while carrying no new information:
|
|
678
|
-
- a count of errors by tool ("Bash is the dominant failure surface, 54 error spans")
|
|
679
|
-
- a count or inventory of error signatures ("47 distinct error signatures")
|
|
680
|
-
- an error rate ("79 of 5260 tool calls ended in errors")
|
|
681
|
-
- any restatement of \`prevalence\`, \`trace_count\`, or \`span_count\` from \`error_clusters[]\`
|
|
682
|
-
Every finding MUST state its measured cost — spans burned, model calls burned, wall-clock — in the claim, and MUST cite the two spans that bound it: the error span and the recovery span, or the escaping human turn, or the trace's last span when unrecovered. A failure you cannot bound with two spans is one you did not measure: drop it.
|
|
683
|
-
|
|
684
|
-
**Adjudicate boundaries with subqueries.** The single judgement call in this protocol is whether a post-error span is genuine resumption or more flailing. Load E, the spans between, and the candidate R, then send one bounded \`llm_query\` per candidate carrying those exact excerpts and asking which span first advances the task. Subqueries cannot call trace tools, so a trace id tells them nothing — paste the excerpts. Accept a boundary only when the excerpts you loaded support the answer.
|
|
685
|
-
|
|
686
|
-
Confidence 0.9+ when both bounding spans are quoted and the spans between them were counted directly; 0.6-0.8 when an oversized trace forced you to sample the interval and the distance is an estimate; below 0.5 does not belong in this analyst, because an unmeasurable cost is not a finding. Keep the recommended action a short imperative; the improvement analyst expands it.
|
|
687
|
-
|
|
688
|
-
If every error in this dataset was recovered on the next span, return an empty findings array. That is the correct answer for a dataset whose failures were all cheap. Do not backfill it with the counts listed above and do not pad it with speculation.`,
|
|
689
|
-
toolGroup: "all",
|
|
690
|
-
limits: {
|
|
691
|
-
maxLlmCalls: 8,
|
|
692
|
-
maxIterations: 24,
|
|
693
|
-
maxToolCalls: 80
|
|
694
|
-
},
|
|
695
|
-
minimumEvidenceCitations: 2
|
|
696
|
-
};
|
|
697
|
-
const IMPROVEMENT_KIND_SPEC = {
|
|
698
|
-
id: "improvement",
|
|
699
|
-
description: "Converts upstream failure / gap / poisoning findings into concrete locus-named edits (prompt, tool-doc, RAG, scaffolding) with leverage grades.",
|
|
700
|
-
area: "improvement",
|
|
701
|
-
version: "1.2.0",
|
|
702
|
-
instructions: `You are a self-improvement analyst. Your job is to propose **concrete, locus-named edits** the agent's runtime should adopt to fix the failure modes, knowledge gaps, and poisonings present in this dataset.
|
|
703
|
-
|
|
704
|
-
Upstream analysts have already classified the problems. Your job is to convert each problem into a *change to make* and grade its expected leverage. Each finding is one proposed edit.
|
|
705
|
-
|
|
706
|
-
${findingSubjectGrammarPromptFor("improvement")}
|
|
707
|
-
|
|
708
|
-
DISCOVERY → CANDIDATE-FIXES → COMPETE → CITE protocol:
|
|
709
|
-
|
|
710
|
-
1. \`getDatasetOverview({})\` first. Note the agents, tools, and any system-prompt fingerprints (look for the prompt text echoed in early spans).
|
|
711
|
-
2. For each high-severity failure pattern, generate 2-3 candidate fixes. Real candidate axes:
|
|
712
|
-
- **System-prompt edit** — add an instruction, remove a misleading one, restructure precedence
|
|
713
|
-
- **Tool description edit** — rewrite a tool's description so the agent picks it correctly / passes valid args
|
|
714
|
-
- **New tool** — add a tool the agent kept emulating in code
|
|
715
|
-
- **RAG ingestion** — add a document or correct a stale one
|
|
716
|
-
- **Memory invalidation** — clear cached prior-run decisions that no longer apply
|
|
717
|
-
- **Scaffolding** — add a precondition check, a retry policy, a turn budget, a verification step
|
|
718
|
-
- **Output schema** — narrow the agent's output to forbid the failure shape
|
|
719
|
-
- **Skill / MCP / hook / subagent** — change the reusable profile component responsible for the behavior
|
|
720
|
-
- **Workflow / rollout policy** — change orchestration, budget, sampling, or stopping behavior
|
|
721
|
-
- **Code** — change an implementation path when profile edits cannot repair the behavior
|
|
722
|
-
3. **Compare candidate fixes with bounded subqueries.** Load the representative failure excerpts, then send one \`llm_query\` per candidate-fix axis the same evidence. Ask for likely effect, side effects, and implementation scope. Subqueries cannot call trace tools; trace ids alone are insufficient context.
|
|
723
|
-
4. After the comparisons return, **pick the winning candidate per cluster** based on expected effect and risk, then emit ONE finding. Keep the alternatives and rejection reasons in the rationale so the recommendation is auditable.
|
|
724
|
-
5. **Cross-reference upstream findings.** Cite prior failure-mode or knowledge-gap findings as \`finding://<prior-finding-id>\`. This builds the dependency graph that lets the dashboard show "fix #X resolves failure modes A, B, C."
|
|
725
|
-
|
|
726
|
-
For each winning recommendation, emit ONE finding. Use one exact locus from the subject grammar and state the edit in one sentence. Match leverage to the source failure's severity; use medium for quality-of-life changes and info for cleanup with no behavioral effect. Cite the targeted \`finding://<id>\` when available and the most representative span when useful. Quote the problem being fixed. Use confidence 0.85+ for a mechanical fix to a well-evidenced failure, 0.6-0.8 when judgment is required, and <0.5 for speculation. Explain in at most two sentences why this candidate beat its alternatives. The recommended action must be the literal diff, quoted replacement, tool description, or setting change.
|
|
727
|
-
|
|
728
|
-
If no upstream failure findings exist in this run, derive your own from the trace dataset using the failure-mode protocol inline (\`searchTrace\` for STATUS_CODE_ERROR / MaxTurnsExceeded / etc.). Prefer upstream findings when present because the analysts are designed to chain.
|
|
729
|
-
|
|
730
|
-
Do NOT propose a fix you cannot defend with evidence. "Tighten the prompt" is not a finding; "Add 'When the user asks for X, always Y' to the system prompt section "request-classification"" is.`,
|
|
731
|
-
toolGroup: "all",
|
|
732
|
-
limits: {
|
|
733
|
-
maxLlmCalls: 8,
|
|
734
|
-
maxIterations: 30,
|
|
735
|
-
maxToolCalls: 80,
|
|
736
|
-
maxOutputChars: 12e3
|
|
737
|
-
}
|
|
738
|
-
};
|
|
739
|
-
const INTENT_DIVERGENCE_KIND_SPEC = {
|
|
740
|
-
id: "intent-divergence",
|
|
741
|
-
description: "Anchors each corrective human turn to the earliest assistant turn where the divergence from stated intent was already detectable, and prices it in burned turns.",
|
|
742
|
-
area: "intent-divergence",
|
|
743
|
-
version: "1.0.0",
|
|
744
|
-
instructions: `You are an intent-divergence analyst for an OTLP trace dataset. Your job is to find where the agent stopped doing what the user asked, and to establish **how early that was knowable**.
|
|
745
|
-
|
|
746
|
-
This dataset carries ground truth the other analysts do not have. When a human turn corrects the agent — "no, I meant…", "stop doing X", "why are you still asking" — the user has labelled a divergence for you. That label marks the END of the divergence, not its start. Your finding is the start: the earliest assistant turn already off-intent, the signal in the user's stated intent it contradicted, and the number of turns burned before the correction landed.
|
|
747
|
-
|
|
748
|
-
${findingSubjectGrammarPromptFor("intent-divergence")}
|
|
749
|
-
|
|
750
|
-
DISCOVERY → BACKTRACK → QUANTIFY → CITE protocol:
|
|
751
|
-
|
|
752
|
-
1. \`getDatasetOverview({})\` first. Note the agent names and \`sample_trace_ids\`; conversational datasets carry several human turns per trace and those are the ones worth pulling.
|
|
753
|
-
2. **DISCOVERY — find the corrective human turns.** Use \`searchTrace\` for user-authored corrections:
|
|
754
|
-
Every pattern MUST begin with \`(?i)\`. The trace store compiles a search pattern case-sensitively unless it opens with that flag, and corrections are overwhelmingly sentence-initial and capitalised ("Stop", "No,", "I said"), so a lowercase pattern silently returns zero hits.
|
|
755
|
-
- Imperative stops: \`(?i)\\bstop\\b\`, \`(?i)\\b(don'?t|do not)\\b\`, \`(?i)^\\s*no[,.!]\`, \`(?i)\\bnot what I (wanted|asked|meant)\\b\`, \`(?i)\\bthat'?s not what\\b\`, \`(?i)\\brevert\\b\`, \`(?i)\\bstart over\\b\`
|
|
756
|
-
- Repetition and frustration: \`(?i)\\bI (said|asked|told you)\\b\`, \`(?i)\\balready told you\\b\`, \`(?i)\\bwhy are you (still )?\\w+ing\\b\`, \`(?i)\\bjust (do|use|write|make)\\b\`, \`(?i)\\bagain\\b\`
|
|
757
|
-
- Clarification markers (second-order, see step 6): \`(?i)\\bI mean(t)?\\b\`, \`(?i)\\bto be clear\\b\`, \`(?i)\\bactually,? I (want|need)\\b\`, \`(?i)\\bwhat I meant\\b\`, \`(?i)\\blet me rephrase\\b\`
|
|
758
|
-
Attribute every match to a role before using it. The same phrases inside an assistant turn are hedging, not correction.
|
|
759
|
-
3. **BACKTRACK — walk the trace backward from the correction.** Read the ordered turns preceding it with \`viewTrace\` / \`viewSpans\`. Locate the user's ORIGINAL statement of intent (usually the first human turn, sometimes a constraint added mid-run), then locate the FIRST assistant turn that contradicts it. That turn, not the correction, anchors the finding. Name the concrete signal already available at that turn: an explicit constraint the agent violated, a scope it exceeded, a different question it answered, a file or target the user never named, a decision the user had already made.
|
|
760
|
-
4. **QUANTIFY — count the burned turns.** K = (index of the corrective human turn) − (index of the earliest detectable assistant turn). Report both indices and K. If you cannot locate an earliest-detectable turn, you have a correction and not a divergence: drop it.
|
|
761
|
-
5. **CLUSTER.** Repeated corrections about the same misread intent are ONE finding citing all of them, not one finding per corrective turn. Corrections about genuinely different intents in the same trace are separate findings.
|
|
762
|
-
6. **Second-order — the request itself was ambiguous.** When the user CLARIFIES rather than corrects ("I mean the staging config, not prod"), the agent's reading was defensible and the request was underspecified or self-contradictory. Emit those against a \`scaffolding:*\` locus — a clarifying-question policy or a plan checkpoint — because the fix is one sharp question before acting, priced against the K turns the ambiguity cost. Do not charge the agent's instructions for a request that could not be resolved from its own text.
|
|
763
|
-
|
|
764
|
-
**Test competing anchors with subqueries.** After the backward walk, load the excerpts for the original intent turn, your candidate earliest-detectable turn, and the turn before it. Send one bounded \`llm_query\` per candidate anchor asking whether the divergence is already present in that turn's text alone. Subqueries cannot call trace tools, and a turn index alone is insufficient context. Take the EARLIEST candidate the loaded excerpts support.
|
|
765
|
-
|
|
766
|
-
For each cluster, emit ONE finding. Use an exact locus from the subject grammar — the surface whose edit prevents the divergence, not the surface where it surfaced. State the claim in the shape "turn N did X against stated intent Y; user corrected at turn N+K, burning K turns." Rate it critical when the divergence produced a user-visible or irreversible action, high when K is 3 or more or the run ended without the intent satisfied, medium for one or two burned turns, and low for a cosmetic correction. Cite BOTH spans with exact quotes: the earliest-detectable assistant turn and the corrective human turn. Use confidence 0.85+ when the stated intent and the diverging turn are both quotable and the contradiction is explicit, and 0.6-0.8 when the earliest-detectable turn is inferred from surrounding context. The recommended action must be the literal instruction, question, or checkpoint to add.
|
|
767
|
-
|
|
768
|
-
Do NOT report a divergence the agent noticed and reversed inside the same turn — that is self-correction and it cost the user nothing. Do NOT report a human turn that adds NEW scope; a changed mind is not a divergence. If the dataset holds no corrective or clarifying human turns, return an empty findings array instead of grading tone.`,
|
|
769
|
-
toolGroup: "all",
|
|
770
|
-
limits: {
|
|
771
|
-
maxLlmCalls: 6,
|
|
772
|
-
maxIterations: 22,
|
|
773
|
-
maxToolCalls: 64
|
|
774
|
-
},
|
|
775
|
-
minimumEvidenceCitations: 2
|
|
776
|
-
};
|
|
777
|
-
const KNOWLEDGE_GAP_KIND_SPEC = {
|
|
778
|
-
id: "knowledge-gap",
|
|
779
|
-
description: "Identifies missing or stale pieces of knowledge — primarily against the agent-knowledge wiki — and attributes each to the runtime layer (wiki page, claim, raw source, websearch, tool-doc, system-prompt, memory) that should have held it.",
|
|
780
|
-
area: "knowledge-gap",
|
|
781
|
-
version: "1.2.0",
|
|
782
|
-
instructions: `You are a knowledge-gap analyst for an OTLP trace dataset. Your job is to identify the **specific pieces of information the agent lacked, or that were stale**, that caused poor decisions.
|
|
783
|
-
|
|
784
|
-
The agent under analysis maintains a curated knowledge base via \`@tangle-network/agent-knowledge\` — a wiki of \`KnowledgePage\`s with raw source anchors, claims, and relations. The primary expected store of agent-knowable facts IS that wiki. A "knowledge gap" is anything the agent had to discover or guess at run-time that the wiki should have held — or an outdated/contradictory fact the agent picked up from a non-wiki source.
|
|
785
|
-
|
|
786
|
-
${findingSubjectGrammarPromptFor("knowledge-gap")}
|
|
787
|
-
|
|
788
|
-
DISCOVERY → ATTRIBUTE-TO-LAYER → CITE protocol:
|
|
789
|
-
|
|
790
|
-
1. \`getDatasetOverview({})\` first. Note which agents, tools, and models appear.
|
|
791
|
-
2. Pull traces where the agent shows gap signals. The strongest signals are:
|
|
792
|
-
- Self-correction turns ("I assumed X but…", "let me re-check", "actually,")
|
|
793
|
-
- Clarifying-question turns where the agent asked the user something the runtime should have surfaced
|
|
794
|
-
- Repeated retrieval / lookup calls for the same artifact with slightly varied queries
|
|
795
|
-
- Tool errors that name a missing argument or unknown resource
|
|
796
|
-
- Web-search calls returning pages dated before a known cutoff for content that changes (versioned APIs, schemas, policies)
|
|
797
|
-
- Agent quoting a tool's docs / system prompt incorrectly because the actual text was insufficient
|
|
798
|
-
- Fabricated identifiers that don't appear in dataset \`sample_trace_ids\`
|
|
799
|
-
Use \`searchTrace\` with patterns like \`I (don.?t|do not) know\`, \`assumed\`, \`unclear\`, \`could you (clarify|tell me|provide)\`, \`not found\`, \`undefined\`, \`unknown\`, \`null\`, dates older than the analysis window, or the agent's specific clarification phrases.
|
|
800
|
-
3. For each gap, identify the **layer of the runtime that should have prevented it** and use its exact locus from the subject grammar above.
|
|
801
|
-
4. For each defensible gap, emit ONE finding. Use an exact locus from the subject grammar and name the missing or stale knowledge (for example, "wiki has no page on invoice line-item shape; agent re-derived it from raw spans"). Rate it high when it caused failure or a clarifying question, medium for unnecessary turns, and low for minor inefficiency. Cite the span where the question, correction, retrieval miss, or stale result surfaced and quote it exactly. Use confidence 0.85+ when the agent articulated the gap and 0.6-0.8 when inferred. Recommend a concrete wiki edit for an agent-knowledge locus or a prompt/tool-description edit otherwise.
|
|
802
|
-
|
|
803
|
-
**Compare layers over loaded evidence.** After the first scan, load the exact excerpts behind candidates across \`agent-knowledge:*\`, \`websearch:outdated\`, \`tool-doc:*\`, \`system-prompt:*\`, and \`memory:*\`. Use one bounded \`llm_query\` per layer to classify those excerpts. Subqueries cannot call trace tools. Merge their classifications into the final finding set only when the source excerpts support them.
|
|
804
|
-
|
|
805
|
-
Do NOT report a gap that the agent later recovered from cleanly within the same turn. That is resilience, not a gap. Cite the non-recovery version when both exist.`,
|
|
806
|
-
toolGroup: "discoveryAndSearch",
|
|
807
|
-
limits: {
|
|
808
|
-
maxLlmCalls: 5,
|
|
809
|
-
maxIterations: 18,
|
|
810
|
-
maxToolCalls: 48
|
|
811
|
-
}
|
|
812
|
-
};
|
|
813
|
-
const KNOWLEDGE_POISONING_KIND_SPEC = {
|
|
814
|
-
id: "knowledge-poisoning",
|
|
815
|
-
description: "Identifies confident-but-wrong actions caused by stale memory, contradicting RAG, deprecated tool docs, or outdated system-prompt instructions.",
|
|
816
|
-
area: "knowledge-poisoning",
|
|
817
|
-
version: "1.2.0",
|
|
818
|
-
instructions: `You are a knowledge-poisoning analyst for an OTLP trace dataset. Your job is to identify cases where the agent **confidently used wrong information** — not where it lacked information (that's the knowledge-gap analyst).
|
|
819
|
-
|
|
820
|
-
${findingSubjectGrammarPromptFor("knowledge-poisoning")}
|
|
821
|
-
|
|
822
|
-
DISCOVERY → DUAL-VERIFY → CITE protocol:
|
|
823
|
-
|
|
824
|
-
1. \`getDatasetOverview({})\` first. Identify the agents, models, and tools.
|
|
825
|
-
2. Pull traces where the agent's confident action was later contradicted. Strongest signals:
|
|
826
|
-
- Agent stated a fact in one span; a later span surfaced contradictory evidence; the agent then proceeded anyway or fabricated reconciliation.
|
|
827
|
-
- Tool call with stale arguments (an id that no longer exists, an API shape that changed).
|
|
828
|
-
- Agent cited an \`agent-knowledge\` wiki page or claim whose content contradicts the trace's own evidence — the wiki itself drifted.
|
|
829
|
-
- Web-search result the agent cited that returned an outdated page; agent treated it as canonical.
|
|
830
|
-
- System-prompt instruction the agent followed that ground-truth evidence in the trace contradicts (e.g. prompt says "use endpoint A"; tool reply says "endpoint A deprecated, use B").
|
|
831
|
-
- Repeated wrong-shape parsing despite the tool's actual output proving the shape.
|
|
832
|
-
3. Use \`searchTrace\` with regex on phrases like \`actually\`, \`turns out\`, \`previously assumed\`, \`old version\`, \`deprecated\`, \`updated to\`, \`now uses\`, or specific entity names you suspect have changed.
|
|
833
|
-
4. For each candidate poisoning, **DUAL-VERIFY**:
|
|
834
|
-
- Confirm the agent actually acted on the false belief (cite the span where it did)
|
|
835
|
-
- Confirm the belief is actually false in this trace's own evidence (cite the span that contradicts it)
|
|
836
|
-
Only emit a finding when both halves are supported. If you can only support one, drop it because single-evidence poisoning findings are too speculative to be useful.
|
|
837
|
-
|
|
838
|
-
**Independently assess both halves.** Load the action excerpt and contradicting excerpt yourself, then send bounded \`llm_query\` calls the exact evidence for "did the agent act?" and "does the trace contradict the belief?" Subqueries cannot call trace tools. Accept a poisoning only when both assessments and the source excerpts support it.
|
|
839
|
-
|
|
840
|
-
For each confirmed poisoning, emit ONE finding. Use the source of the false belief as the exact subject. State "agent believed X (from source S); trace evidence shows X is false." Rate it critical for a wrong user-visible action, high when caught internally after significant waste, and medium for inefficiency. Cite BOTH the action span and the contradicting span with exact quotes. Use confidence 0.85+ when both halves have exact quotes and 0.6-0.8 when one half is inferred. Recommend the literal source correction: update the wiki claim, invalidate and re-curate the raw source, or replace the stale prompt/tool instruction.
|
|
841
|
-
|
|
842
|
-
Do NOT report a finding if the agent caught and corrected the false belief in the same turn. Reserve poisoning for cases where the false belief shaped downstream action.`,
|
|
843
|
-
toolGroup: "all",
|
|
844
|
-
limits: {
|
|
845
|
-
maxLlmCalls: 8,
|
|
846
|
-
maxIterations: 20,
|
|
847
|
-
maxToolCalls: 64
|
|
848
|
-
},
|
|
849
|
-
minimumEvidenceCitations: 2
|
|
850
|
-
};
|
|
851
487
|
//#endregion
|
|
852
|
-
//#region src/analyst/
|
|
488
|
+
//#region src/trace-analyst/behavioral-metrics.ts
|
|
853
489
|
/**
|
|
854
|
-
*
|
|
855
|
-
*
|
|
856
|
-
*
|
|
857
|
-
*
|
|
858
|
-
*
|
|
859
|
-
*
|
|
860
|
-
*
|
|
490
|
+
* Deterministic behavioral metrics over OTLP spans — pure arithmetic, no LLM.
|
|
491
|
+
*
|
|
492
|
+
* It computes token growth, output decay, tool monoculture, and missing
|
|
493
|
+
* self-verification once in TypeScript with no model judgment.
|
|
494
|
+
*
|
|
495
|
+
* General, not trace-specific: the detectors key off token trajectories and
|
|
496
|
+
* tool usage present in any agentic OTLP trace, not any one benchmark.
|
|
861
497
|
*/
|
|
862
|
-
|
|
863
|
-
|
|
864
|
-
|
|
865
|
-
|
|
866
|
-
|
|
867
|
-
|
|
868
|
-
|
|
869
|
-
|
|
870
|
-
|
|
871
|
-
|
|
872
|
-
|
|
873
|
-
|
|
498
|
+
/** ≥ this input-token growth ratio across a run, with no compression, fires. */
|
|
499
|
+
const INPUT_GROWTH_FACTOR = 3;
|
|
500
|
+
/** Tool-usage signals need at least this many calls to be meaningful. */
|
|
501
|
+
const MIN_TOOL_CALLS = 3;
|
|
502
|
+
/**
|
|
503
|
+
* Serial calls required before a strictly-decreasing output run counts as decay.
|
|
504
|
+
*
|
|
505
|
+
* A run of n independent lengths is strictly decreasing by chance with
|
|
506
|
+
* probability 1/n!, so the old minimum of 3 fired on roughly one sequence in
|
|
507
|
+
* six. The paired `inputIsMonotonic && inputGrew` guard does not offset that:
|
|
508
|
+
* context accumulates by construction in a serial agent loop, so input growth
|
|
509
|
+
* is very nearly free evidence. At 5 calls chance alone accounts for under 1%,
|
|
510
|
+
* which is the bar a signal reported at full confidence has to clear.
|
|
511
|
+
*/
|
|
512
|
+
const OUTPUT_DECAY_MINIMUM_CALLS = 5;
|
|
513
|
+
/**
|
|
514
|
+
* The last output must fall to at most this fraction of the first.
|
|
515
|
+
*
|
|
516
|
+
* Length wanders between turns for reasons that are not degradation, so
|
|
517
|
+
* direction alone is not a finding — an observed 784 → 646 run (18%) was
|
|
518
|
+
* reported as decay and was noise. Requiring the response to lose most of its
|
|
519
|
+
* length keeps the signal on the failure it names: late steps that quietly
|
|
520
|
+
* stop doing the work.
|
|
521
|
+
*/
|
|
522
|
+
const OUTPUT_DECAY_MAXIMUM_RETAINED_FRACTION = .6;
|
|
523
|
+
/** Tool names that read or check state count as self-verification, not mutation.
|
|
524
|
+
* Covers the inspect verbs plus the read/search tools real harnesses use to
|
|
525
|
+
* verify (Claude Code Read/Grep/Glob, codex read_file/ls/cat, git status/diff,
|
|
526
|
+
* test/lint). A pure shell tool (Bash/exec_command) is intentionally NOT matched
|
|
527
|
+
* — its name can't tell a `pytest` from an `rm`. */
|
|
528
|
+
const VERIFY_RE = /verif|eval|inspect|check|assert|validat|review|confirm|read|grep|glob|search|view|\blist\b|\bls\b|\bcat\b|\bfind\b|diff|status|\btest|lint|typecheck/i;
|
|
529
|
+
function num(v) {
|
|
530
|
+
return typeof v === "number" && Number.isFinite(v) ? v : null;
|
|
874
531
|
}
|
|
875
|
-
|
|
876
|
-
|
|
877
|
-
|
|
532
|
+
function numAttr(attrs, keys) {
|
|
533
|
+
for (const key of keys) {
|
|
534
|
+
const value = num(attrs[key]);
|
|
535
|
+
if (value !== null) return value;
|
|
536
|
+
}
|
|
537
|
+
return null;
|
|
878
538
|
}
|
|
879
|
-
function
|
|
880
|
-
const
|
|
881
|
-
if (
|
|
882
|
-
|
|
883
|
-
return snapshot;
|
|
539
|
+
function inputTokensOf(s) {
|
|
540
|
+
const exactContext = num(s.attributes[LLM_CONTEXT_TOKENS]);
|
|
541
|
+
if (exactContext !== null) return exactContext;
|
|
542
|
+
return numAttr(s.attributes, LLM_INPUT_TOKEN_ATTR_KEYS) ?? num(s.attributes["llm.usage.input_tokens"]);
|
|
884
543
|
}
|
|
885
|
-
|
|
886
|
-
|
|
887
|
-
return sealExactAnalystRunReceipt(snapshotAnalystRunRecord(value, context), context);
|
|
544
|
+
function outputTokensOf(s) {
|
|
545
|
+
return numAttr(s.attributes, LLM_OUTPUT_TOKEN_ATTR_KEYS) ?? num(s.attributes["llm.usage.output_tokens"]);
|
|
888
546
|
}
|
|
889
|
-
function
|
|
890
|
-
|
|
891
|
-
|
|
892
|
-
|
|
893
|
-
|
|
894
|
-
|
|
547
|
+
function stepOf(s) {
|
|
548
|
+
return num(s.attributes.step);
|
|
549
|
+
}
|
|
550
|
+
function toolNameOf(s) {
|
|
551
|
+
if (s.tool_name) return s.tool_name;
|
|
552
|
+
for (const key of TOOL_NAME_ATTR_KEYS) {
|
|
553
|
+
const t = s.attributes[key];
|
|
554
|
+
if (typeof t === "string" && t.length > 0) return t;
|
|
895
555
|
}
|
|
896
|
-
|
|
897
|
-
assertOnlyKeys(snapshot, [
|
|
898
|
-
"run_id",
|
|
899
|
-
"correlation_id",
|
|
900
|
-
"started_at",
|
|
901
|
-
"ended_at",
|
|
902
|
-
"findings",
|
|
903
|
-
"per_analyst",
|
|
904
|
-
"total_cost_usd",
|
|
905
|
-
"total_cost_provenance",
|
|
906
|
-
"execution_plan",
|
|
907
|
-
"completion"
|
|
908
|
-
], context);
|
|
909
|
-
requiredString(snapshot.run_id, `${context} run_id`);
|
|
910
|
-
requiredString(snapshot.correlation_id, `${context} correlation_id`);
|
|
911
|
-
canonicalTimestamp(snapshot.started_at, `${context} started_at`);
|
|
912
|
-
canonicalTimestamp(snapshot.ended_at, `${context} ended_at`);
|
|
913
|
-
snapshot.findings = snapshotAnalystFindings(snapshot.findings, `${context} findings`);
|
|
914
|
-
if (!Array.isArray(snapshot.per_analyst)) throw new TypeError(`${context} per_analyst must be an array`);
|
|
915
|
-
for (const [index, summary] of snapshot.per_analyst.entries()) assertAnalystRunSummary(summary, `${context} per_analyst ${index}`);
|
|
916
|
-
if (typeof snapshot.total_cost_usd !== "number" || !Number.isFinite(snapshot.total_cost_usd) || snapshot.total_cost_usd < 0) throw new TypeError(`${context} total_cost_usd must be a finite non-negative number`);
|
|
917
|
-
if (snapshot.total_cost_provenance !== void 0) assertCostProvenance(snapshot.total_cost_provenance, `${context} total_cost_provenance`);
|
|
918
|
-
return snapshot;
|
|
556
|
+
return null;
|
|
919
557
|
}
|
|
920
|
-
|
|
921
|
-
|
|
922
|
-
|
|
923
|
-
|
|
924
|
-
|
|
925
|
-
|
|
926
|
-
|
|
927
|
-
|
|
928
|
-
|
|
929
|
-
|
|
930
|
-
|
|
931
|
-
|
|
932
|
-
|
|
933
|
-
|
|
934
|
-
|
|
935
|
-
|
|
936
|
-
|
|
937
|
-
|
|
938
|
-
|
|
939
|
-
|
|
940
|
-
|
|
941
|
-
|
|
942
|
-
|
|
943
|
-
|
|
944
|
-
|
|
945
|
-
|
|
558
|
+
/**
|
|
559
|
+
* Reduce a span list to behavioral metrics + fired suboptimality signals.
|
|
560
|
+
* Pure + deterministic: same spans → same output, on any machine, no model.
|
|
561
|
+
*/
|
|
562
|
+
function computeTraceMetrics(spans) {
|
|
563
|
+
const traceIds = new Set(spans.map((span) => span.trace_id));
|
|
564
|
+
if (traceIds.size > 1) throw new Error(`computeTraceMetrics: expected spans from one trace, received ${traceIds.size} traces`);
|
|
565
|
+
const traceId = traceIds.values().next().value ?? null;
|
|
566
|
+
const samples = spans.map((span) => ({
|
|
567
|
+
span,
|
|
568
|
+
input: inputTokensOf(span),
|
|
569
|
+
output: outputTokensOf(span),
|
|
570
|
+
step: stepOf(span)
|
|
571
|
+
}));
|
|
572
|
+
const llmSamples = samples.filter((sample) => sample.span.kind === "LLM");
|
|
573
|
+
const tokenSamples = llmSamples.length > 0 ? llmSamples : samples.filter((sample) => sample.input !== null || sample.output !== null);
|
|
574
|
+
const tokenSequences = buildTokenSequences(tokenSamples, spans);
|
|
575
|
+
const primarySequence = tokenSequences[0];
|
|
576
|
+
const inputTokenTrajectory = primarySequence?.inputTokenTrajectory.filter((value) => value !== null) ?? [];
|
|
577
|
+
const outputTokenTrajectory = primarySequence?.outputTokenTrajectory.filter((value) => value !== null) ?? [];
|
|
578
|
+
const toolHistogram = {};
|
|
579
|
+
let hasSelfVerification = false;
|
|
580
|
+
for (const s of spans) {
|
|
581
|
+
const tool = toolNameOf(s);
|
|
582
|
+
if (tool) {
|
|
583
|
+
toolHistogram[tool] = (toolHistogram[tool] ?? 0) + 1;
|
|
584
|
+
if (VERIFY_RE.test(tool)) hasSelfVerification = true;
|
|
585
|
+
}
|
|
586
|
+
}
|
|
587
|
+
const totalToolCalls = Object.values(toolHistogram).reduce((a, b) => a + b, 0);
|
|
588
|
+
const distinctTools = Object.keys(toolHistogram).length;
|
|
589
|
+
const toolDiversityRatio = totalToolCalls === 0 ? 1 : distinctTools / totalToolCalls;
|
|
590
|
+
const signals = [];
|
|
591
|
+
const seenTokenSignals = /* @__PURE__ */ new Set();
|
|
592
|
+
for (const sequence of tokenSequences) for (const signal of tokenSignals(sequence)) {
|
|
593
|
+
if (seenTokenSignals.has(signal.code)) continue;
|
|
594
|
+
seenTokenSignals.add(signal.code);
|
|
595
|
+
signals.push(signal);
|
|
596
|
+
}
|
|
597
|
+
if (totalToolCalls >= MIN_TOOL_CALLS && distinctTools === 1) {
|
|
598
|
+
const only = Object.keys(toolHistogram)[0];
|
|
599
|
+
signals.push({
|
|
600
|
+
code: "single-tool-dependency",
|
|
601
|
+
severity: "medium",
|
|
602
|
+
detail: `All ${totalToolCalls} observed tool calls are \`${only}\`; no alternate tool call was observed.`,
|
|
603
|
+
evidence: {
|
|
604
|
+
tool: only,
|
|
605
|
+
calls: totalToolCalls,
|
|
606
|
+
distinct_tools: 1
|
|
607
|
+
}
|
|
608
|
+
});
|
|
609
|
+
}
|
|
610
|
+
if (totalToolCalls >= MIN_TOOL_CALLS && !hasSelfVerification) signals.push({
|
|
611
|
+
code: "no-self-verification",
|
|
612
|
+
severity: "medium",
|
|
613
|
+
detail: `${totalToolCalls} tool calls were observed without a verification-named tool call.`,
|
|
614
|
+
evidence: {
|
|
615
|
+
tool_calls: totalToolCalls,
|
|
616
|
+
verification_calls: 0
|
|
617
|
+
}
|
|
618
|
+
});
|
|
619
|
+
return {
|
|
620
|
+
traceId,
|
|
621
|
+
llmCallCount: tokenSamples.length,
|
|
622
|
+
tokenSequences,
|
|
623
|
+
inputTokenTrajectory,
|
|
624
|
+
outputTokenTrajectory,
|
|
625
|
+
toolHistogram,
|
|
626
|
+
totalToolCalls,
|
|
627
|
+
distinctTools,
|
|
628
|
+
toolDiversityRatio,
|
|
629
|
+
hasSelfVerification,
|
|
630
|
+
signals
|
|
631
|
+
};
|
|
946
632
|
}
|
|
947
|
-
function
|
|
948
|
-
|
|
949
|
-
|
|
950
|
-
|
|
951
|
-
|
|
952
|
-
|
|
953
|
-
|
|
954
|
-
|
|
955
|
-
for (const
|
|
956
|
-
|
|
957
|
-
|
|
958
|
-
|
|
959
|
-
|
|
960
|
-
|
|
961
|
-
|
|
962
|
-
|
|
963
|
-
|
|
964
|
-
|
|
965
|
-
|
|
966
|
-
|
|
967
|
-
|
|
968
|
-
|
|
633
|
+
function buildTokenSequences(samples, spans) {
|
|
634
|
+
const executionScopeFor = createTokenExecutionScopeResolver(new Map(spans.map((span) => [span.span_id, span])));
|
|
635
|
+
const scopedSamples = samples.map((sample) => ({
|
|
636
|
+
sample,
|
|
637
|
+
...executionScopeFor(sample.span)
|
|
638
|
+
}));
|
|
639
|
+
const trackByLane = executionTrackByLane(scopedSamples);
|
|
640
|
+
const byTrack = /* @__PURE__ */ new Map();
|
|
641
|
+
for (const scoped of scopedSamples) {
|
|
642
|
+
const trackId = trackByLane.get(scoped.key);
|
|
643
|
+
const track = byTrack.get(trackId) ?? {
|
|
644
|
+
scopeId: scoped.scopeId,
|
|
645
|
+
samples: []
|
|
646
|
+
};
|
|
647
|
+
track.samples.push(scoped.sample);
|
|
648
|
+
byTrack.set(trackId, track);
|
|
649
|
+
}
|
|
650
|
+
const sequences = [];
|
|
651
|
+
for (const { scopeId, samples: tracked } of byTrack.values()) {
|
|
652
|
+
const runs = serialTokenRuns([...tracked].sort(compareTokenSamples));
|
|
653
|
+
runs.forEach((run, index) => {
|
|
654
|
+
sequences.push({
|
|
655
|
+
scopeId: runs.length === 1 ? scopeId : `${scopeId}#${index + 1}`,
|
|
656
|
+
spanIds: run.map((sample) => sample.span.span_id),
|
|
657
|
+
inputTokenTrajectory: run.map((sample) => sample.input),
|
|
658
|
+
outputTokenTrajectory: run.map((sample) => sample.output)
|
|
659
|
+
});
|
|
660
|
+
});
|
|
969
661
|
}
|
|
970
|
-
|
|
971
|
-
assertValidAnalystUsageReceipt(value, context);
|
|
662
|
+
return sequences.sort((a, b) => b.spanIds.length - a.spanIds.length || a.scopeId.localeCompare(b.scopeId) || a.spanIds[0].localeCompare(b.spanIds[0]));
|
|
972
663
|
}
|
|
973
|
-
function
|
|
974
|
-
|
|
975
|
-
|
|
976
|
-
|
|
977
|
-
|
|
978
|
-
|
|
664
|
+
function createTokenExecutionScopeResolver(spansById) {
|
|
665
|
+
const cache = /* @__PURE__ */ new Map();
|
|
666
|
+
return (span) => {
|
|
667
|
+
const ancestry = resolveAncestorScope(span.parent_span_id, spansById, cache);
|
|
668
|
+
const rootId = ancestry.rootId;
|
|
669
|
+
const scopeId = ancestry.agentId ? `span:${ancestry.agentId}` : ancestry.missingParentId ? `parent:${ancestry.missingParentId}` : rootId ? `root:${rootId}` : span.agent_name ? `agent:${span.agent_name}` : `trace:${span.trace_id}`;
|
|
670
|
+
const scopeSpanId = ancestry.agentId ?? ancestry.missingParentId ?? rootId;
|
|
671
|
+
const laneSpan = ancestry.laneSpanId ? spansById.get(ancestry.laneSpanId) : void 0;
|
|
672
|
+
const direct = scopeSpanId === null || ancestry.laneSpanId === scopeSpanId;
|
|
673
|
+
const timedSpan = direct ? span : laneSpan;
|
|
674
|
+
return {
|
|
675
|
+
key: JSON.stringify([scopeId, direct ? span.span_id : ancestry.laneSpanId]),
|
|
676
|
+
scopeKey: scopeId,
|
|
677
|
+
scopeId,
|
|
678
|
+
start: timedSpan ? spanEpochMillis(timedSpan.start_time) : null,
|
|
679
|
+
end: timedSpan ? spanEpochMillis(timedSpan.end_time) : null
|
|
680
|
+
};
|
|
681
|
+
};
|
|
682
|
+
}
|
|
683
|
+
function resolveAncestorScope(startId, spansById, cache) {
|
|
684
|
+
const empty = {
|
|
685
|
+
agentId: null,
|
|
686
|
+
rootId: null,
|
|
687
|
+
missingParentId: null,
|
|
688
|
+
laneSpanId: null
|
|
689
|
+
};
|
|
690
|
+
if (!startId) return empty;
|
|
691
|
+
const path = [];
|
|
692
|
+
const pathIndex = /* @__PURE__ */ new Map();
|
|
693
|
+
let currentId = startId;
|
|
694
|
+
let resolved = empty;
|
|
695
|
+
while (currentId) {
|
|
696
|
+
const cached = cache.get(currentId);
|
|
697
|
+
if (cached) {
|
|
698
|
+
resolved = cached;
|
|
699
|
+
break;
|
|
700
|
+
}
|
|
701
|
+
const cycleStart = pathIndex.get(currentId);
|
|
702
|
+
if (cycleStart !== void 0) {
|
|
703
|
+
resolved = {
|
|
704
|
+
agentId: null,
|
|
705
|
+
rootId: [...path.slice(cycleStart)].sort()[0],
|
|
706
|
+
missingParentId: null,
|
|
707
|
+
laneSpanId: [...path.slice(cycleStart)].sort()[0]
|
|
708
|
+
};
|
|
709
|
+
break;
|
|
710
|
+
}
|
|
711
|
+
const current = spansById.get(currentId);
|
|
712
|
+
if (!current) {
|
|
713
|
+
resolved = {
|
|
714
|
+
agentId: null,
|
|
715
|
+
rootId: null,
|
|
716
|
+
missingParentId: currentId,
|
|
717
|
+
laneSpanId: currentId
|
|
718
|
+
};
|
|
719
|
+
break;
|
|
720
|
+
}
|
|
721
|
+
if (current.kind === "AGENT") {
|
|
722
|
+
resolved = {
|
|
723
|
+
agentId: current.span_id,
|
|
724
|
+
rootId: null,
|
|
725
|
+
missingParentId: null,
|
|
726
|
+
laneSpanId: current.span_id
|
|
727
|
+
};
|
|
728
|
+
break;
|
|
729
|
+
}
|
|
730
|
+
pathIndex.set(currentId, path.length);
|
|
731
|
+
path.push(currentId);
|
|
732
|
+
currentId = current.parent_span_id;
|
|
979
733
|
}
|
|
980
|
-
|
|
981
|
-
|
|
734
|
+
for (let index = path.length - 1; index >= 0; index -= 1) {
|
|
735
|
+
if (resolved.agentId === null && resolved.rootId === null) resolved = {
|
|
736
|
+
...resolved,
|
|
737
|
+
rootId: path[index],
|
|
738
|
+
laneSpanId: path[index]
|
|
739
|
+
};
|
|
740
|
+
else if (resolved.laneSpanId === (resolved.agentId ?? resolved.missingParentId ?? resolved.rootId)) resolved = {
|
|
741
|
+
...resolved,
|
|
742
|
+
laneSpanId: path[index]
|
|
743
|
+
};
|
|
744
|
+
cache.set(path[index], resolved);
|
|
745
|
+
}
|
|
746
|
+
return resolved;
|
|
982
747
|
}
|
|
983
|
-
function
|
|
984
|
-
|
|
985
|
-
const
|
|
986
|
-
|
|
987
|
-
|
|
988
|
-
|
|
989
|
-
|
|
990
|
-
|
|
991
|
-
|
|
992
|
-
|
|
993
|
-
|
|
994
|
-
|
|
995
|
-
|
|
996
|
-
|
|
997
|
-
|
|
998
|
-
if (
|
|
999
|
-
|
|
1000
|
-
|
|
1001
|
-
|
|
748
|
+
function compareTokenSamples(a, b) {
|
|
749
|
+
const aStart = spanEpochMillis(a.span.start_time);
|
|
750
|
+
const bStart = spanEpochMillis(b.span.start_time);
|
|
751
|
+
if (aStart === null && bStart !== null) return 1;
|
|
752
|
+
if (aStart !== null && bStart === null) return -1;
|
|
753
|
+
if (aStart !== null && bStart !== null && aStart !== bStart) return aStart - bStart;
|
|
754
|
+
if (a.step !== null && b.step !== null && a.step !== b.step) return a.step - b.step;
|
|
755
|
+
return a.span.span_id.localeCompare(b.span.span_id);
|
|
756
|
+
}
|
|
757
|
+
function serialTokenRuns(ordered) {
|
|
758
|
+
const runs = [];
|
|
759
|
+
let serial = [];
|
|
760
|
+
let overlap = [];
|
|
761
|
+
let overlapEnd = Number.NEGATIVE_INFINITY;
|
|
762
|
+
const flushSerial = () => {
|
|
763
|
+
if (serial.length > 0) runs.push(serial);
|
|
764
|
+
serial = [];
|
|
765
|
+
};
|
|
766
|
+
const flushOverlap = () => {
|
|
767
|
+
if (overlap.length === 1) serial.push(overlap[0]);
|
|
768
|
+
else if (overlap.length > 1) {
|
|
769
|
+
flushSerial();
|
|
770
|
+
for (const sample of overlap) runs.push([sample]);
|
|
771
|
+
}
|
|
772
|
+
overlap = [];
|
|
773
|
+
overlapEnd = Number.NEGATIVE_INFINITY;
|
|
774
|
+
};
|
|
775
|
+
for (const sample of ordered) {
|
|
776
|
+
const start = spanEpochMillis(sample.span.start_time);
|
|
777
|
+
const end = spanEpochMillis(sample.span.end_time);
|
|
778
|
+
if (start === null || end === null || sample.span.duration_ms <= 0 || end < start) {
|
|
779
|
+
flushOverlap();
|
|
780
|
+
flushSerial();
|
|
781
|
+
runs.push([sample]);
|
|
1002
782
|
continue;
|
|
1003
783
|
}
|
|
1004
|
-
|
|
1005
|
-
|
|
784
|
+
if (overlap.length > 0 && start >= overlapEnd) flushOverlap();
|
|
785
|
+
overlap.push(sample);
|
|
786
|
+
overlapEnd = Math.max(overlapEnd, end);
|
|
1006
787
|
}
|
|
1007
|
-
|
|
1008
|
-
|
|
1009
|
-
|
|
1010
|
-
|
|
788
|
+
flushOverlap();
|
|
789
|
+
flushSerial();
|
|
790
|
+
return runs;
|
|
791
|
+
}
|
|
792
|
+
function tokenSignals(sequence) {
|
|
793
|
+
const signals = [];
|
|
794
|
+
const inputs = sequence.inputTokenTrajectory;
|
|
795
|
+
const outputs = sequence.outputTokenTrajectory;
|
|
796
|
+
if (inputs.length >= 3 && inputs.every((value) => value !== null)) {
|
|
797
|
+
const first = inputs[0];
|
|
798
|
+
const last = inputs[inputs.length - 1];
|
|
799
|
+
const isMonotonic = everyAdjacent(inputs, (previous, current) => current >= previous);
|
|
800
|
+
const growthFromZero = first === 0 && last > 0;
|
|
801
|
+
const growth = growthFromZero ? Infinity : first > 0 ? last / first : 0;
|
|
802
|
+
if (isMonotonic && last > first && growth >= INPUT_GROWTH_FACTOR) {
|
|
803
|
+
const growthLabel = growthFromZero ? "0→nonzero (unbounded)" : `${growth.toFixed(1)}x`;
|
|
804
|
+
signals.push({
|
|
805
|
+
code: "monotonic-input-growth",
|
|
806
|
+
severity: "high",
|
|
807
|
+
detail: `LLM input tokens grew ${growthLabel} (${first}→${last}) across ${inputs.length} serial calls without an intervening decrease.`,
|
|
808
|
+
evidence: {
|
|
809
|
+
first,
|
|
810
|
+
last,
|
|
811
|
+
growth_x: growthFromZero ? "unbounded" : Number(growth.toFixed(2)),
|
|
812
|
+
calls: inputs.length,
|
|
813
|
+
scope: sequence.scopeId,
|
|
814
|
+
first_span_id: sequence.spanIds[0],
|
|
815
|
+
last_span_id: sequence.spanIds[sequence.spanIds.length - 1]
|
|
816
|
+
}
|
|
817
|
+
});
|
|
818
|
+
}
|
|
819
|
+
}
|
|
820
|
+
if (inputs.length >= OUTPUT_DECAY_MINIMUM_CALLS && inputs.length === outputs.length && inputs.every((value) => value !== null) && outputs.every((value) => value !== null)) {
|
|
821
|
+
const first = outputs[0];
|
|
822
|
+
const last = outputs[outputs.length - 1];
|
|
823
|
+
const inputIsMonotonic = everyAdjacent(inputs, (previous, current) => current >= previous);
|
|
824
|
+
const outputIsMonotonic = everyAdjacent(outputs, (previous, current) => current <= previous);
|
|
825
|
+
const inputGrew = inputs[inputs.length - 1] > inputs[0];
|
|
826
|
+
const decayIsMaterial = last <= first * OUTPUT_DECAY_MAXIMUM_RETAINED_FRACTION;
|
|
827
|
+
if (inputIsMonotonic && inputGrew && outputIsMonotonic && decayIsMaterial) signals.push({
|
|
828
|
+
code: "output-length-decay",
|
|
829
|
+
severity: "medium",
|
|
830
|
+
detail: `LLM output tokens shrank ${first}→${last} over ${outputs.length} serial calls while input tokens increased monotonically.`,
|
|
831
|
+
evidence: {
|
|
832
|
+
first,
|
|
833
|
+
last,
|
|
834
|
+
calls: outputs.length,
|
|
835
|
+
scope: sequence.scopeId,
|
|
836
|
+
first_span_id: sequence.spanIds[0],
|
|
837
|
+
last_span_id: sequence.spanIds[sequence.spanIds.length - 1]
|
|
838
|
+
}
|
|
839
|
+
});
|
|
840
|
+
}
|
|
841
|
+
return signals;
|
|
842
|
+
}
|
|
843
|
+
function everyAdjacent(values, predicate) {
|
|
844
|
+
return values.slice(1).every((current, index) => predicate(values[index], current));
|
|
845
|
+
}
|
|
846
|
+
//#endregion
|
|
847
|
+
//#region src/analyst/behavioral-analyst.ts
|
|
848
|
+
/**
|
|
849
|
+
* Deterministic behavioral analysis over arithmetic in trace spans.
|
|
850
|
+
* This pass is cheap and repeatable; semantic analysis remains the job of
|
|
851
|
+
* model-backed analysts. Relative quality requires a labeled comparison.
|
|
852
|
+
*/
|
|
853
|
+
const RECOMMENDED_ACTION = {
|
|
854
|
+
"monotonic-input-growth": "Inspect context assembly; if prior history is repeatedly included, summarize completed work before the next model call.",
|
|
855
|
+
"output-length-decay": "Check late-step completeness; if shorter responses omit required work, add explicit completion criteria to the agent instructions.",
|
|
856
|
+
"single-tool-dependency": "Test whether an inspect or verification tool improves outcomes after the repeated call fails or returns no progress.",
|
|
857
|
+
"no-self-verification": "After state-changing actions, require an observable check before the agent proceeds."
|
|
858
|
+
};
|
|
859
|
+
const ANALYST_ID$1 = "efficiency-behavioral";
|
|
860
|
+
const DEFAULT_MAX_TRACES = 1e3;
|
|
861
|
+
const DEFAULT_MAX_EVIDENCE_REFS = 20;
|
|
862
|
+
const TRACE_PAGE_SIZE = 200;
|
|
863
|
+
const AGGREGATE_CLAIM = {
|
|
864
|
+
"monotonic-input-growth": (observed, analyzed) => `${observed}/${analyzed} analyzed traces showed input tokens grow from zero to nonzero or to at least 3x their initial value across at least 3 serial model calls without a decrease.`,
|
|
865
|
+
"output-length-decay": (observed, analyzed) => `${observed}/${analyzed} analyzed traces showed output tokens decrease while input tokens increased monotonically across at least 3 serial model calls.`,
|
|
866
|
+
"single-tool-dependency": (observed, analyzed) => `${observed}/${analyzed} analyzed traces used only one named tool across at least 3 tool calls.`,
|
|
867
|
+
"no-self-verification": (observed, analyzed) => `${observed}/${analyzed} analyzed traces had at least 3 tool calls without a verification-named tool call.`
|
|
868
|
+
};
|
|
869
|
+
async function listTraceIds(store, maxTraces, signal) {
|
|
870
|
+
const traceIds = /* @__PURE__ */ new Set();
|
|
871
|
+
let offset = 0;
|
|
872
|
+
let expectedTotal;
|
|
873
|
+
while (true) {
|
|
874
|
+
signal?.throwIfAborted();
|
|
875
|
+
const page = await store.queryTraces({
|
|
876
|
+
limit: TRACE_PAGE_SIZE,
|
|
877
|
+
offset
|
|
878
|
+
});
|
|
879
|
+
if (expectedTotal === void 0) expectedTotal = page.total;
|
|
880
|
+
if (page.total !== expectedTotal) throw new Error(`behavioralAnalyst: trace count changed during pagination (${expectedTotal} to ${page.total})`);
|
|
881
|
+
if (page.total > maxTraces) throw new RangeError(`behavioralAnalyst: ${page.total} traces exceed maxTraces=${maxTraces}; filter the store or raise the explicit limit`);
|
|
882
|
+
for (const trace of page.traces) traceIds.add(trace.trace_id);
|
|
883
|
+
if (traceIds.size > maxTraces) throw new RangeError(`behavioralAnalyst: more than maxTraces=${maxTraces} unique traces were returned`);
|
|
884
|
+
if (!page.has_more) break;
|
|
885
|
+
if (page.traces.length === 0) throw new Error("behavioralAnalyst: trace store returned an empty page with has_more=true");
|
|
886
|
+
offset += page.traces.length;
|
|
1011
887
|
}
|
|
1012
|
-
if (
|
|
1013
|
-
|
|
1014
|
-
const costs = summaries.map((summary) => summary.usage.cost);
|
|
1015
|
-
const expectedProvenance = costs.some((cost) => cost.kind === "uncaptured") ? {
|
|
1016
|
-
kind: "uncaptured",
|
|
1017
|
-
usd: null
|
|
1018
|
-
} : {
|
|
1019
|
-
kind: costs.some((cost) => cost.kind === "estimated") ? "estimated" : "observed",
|
|
1020
|
-
usd: costs.reduce((sum, cost) => finiteNonnegative(sum + (cost.usd ?? 0), `${context} aggregate captured cost`), 0)
|
|
1021
|
-
};
|
|
1022
|
-
if (hashCanonical(run.total_cost_provenance) !== hashCanonical(expectedProvenance)) throw new TypeError(`${context} total_cost_provenance does not match per_analyst usage`);
|
|
1023
|
-
return deepFreezeCanonicalJson(run);
|
|
888
|
+
if (traceIds.size !== expectedTotal) throw new Error(`behavioralAnalyst: pagination returned ${traceIds.size}/${expectedTotal ?? 0} unique traces`);
|
|
889
|
+
return [...traceIds].sort();
|
|
1024
890
|
}
|
|
1025
|
-
|
|
1026
|
-
|
|
1027
|
-
|
|
1028
|
-
|
|
1029
|
-
|
|
1030
|
-
|
|
1031
|
-
|
|
1032
|
-
|
|
1033
|
-
|
|
1034
|
-
|
|
1035
|
-
|
|
1036
|
-
|
|
1037
|
-
|
|
891
|
+
/**
|
|
892
|
+
* Map computed signals → structured AnalystFindings. Pure: no LLM, no clock
|
|
893
|
+
* dependence beyond `produced_at` (overridable for deterministic tests).
|
|
894
|
+
*/
|
|
895
|
+
function deriveEfficiencyFindings(metrics, opts = {}) {
|
|
896
|
+
const analystId = opts.analystId ?? ANALYST_ID$1;
|
|
897
|
+
const traceId = metrics.traceId;
|
|
898
|
+
return metrics.signals.map((sig) => makeFinding({
|
|
899
|
+
analyst_id: analystId,
|
|
900
|
+
area: "efficiency",
|
|
901
|
+
subject: sig.code,
|
|
902
|
+
claim: sig.detail,
|
|
903
|
+
severity: sig.severity,
|
|
904
|
+
confidence: 1,
|
|
905
|
+
evidence_refs: [{
|
|
906
|
+
kind: "metric",
|
|
907
|
+
uri: traceId ? `metric://trace/${encodeURIComponent(traceId)}/efficiency/${sig.code}` : `metric://efficiency/${sig.code}`,
|
|
908
|
+
excerpt: JSON.stringify(sig.evidence)
|
|
909
|
+
}],
|
|
910
|
+
recommended_action: RECOMMENDED_ACTION[sig.code],
|
|
911
|
+
metadata: {
|
|
912
|
+
deterministic: true,
|
|
913
|
+
evidence: sig.evidence,
|
|
914
|
+
...traceId ? { trace_id: traceId } : {}
|
|
915
|
+
},
|
|
916
|
+
id_basis: sig.code,
|
|
917
|
+
...opts.producedAt ? { produced_at: opts.producedAt } : {}
|
|
918
|
+
}));
|
|
1038
919
|
}
|
|
1039
|
-
|
|
1040
|
-
|
|
1041
|
-
|
|
920
|
+
/** The deterministic behavioral/efficiency analyst (no LLM, any-model). */
|
|
921
|
+
function behavioralAnalyst(options = {}) {
|
|
922
|
+
const maxTraces = positiveInteger(options.maxTraces ?? DEFAULT_MAX_TRACES, "maxTraces");
|
|
923
|
+
const maxEvidenceRefsPerFinding = positiveInteger(options.maxEvidenceRefsPerFinding ?? DEFAULT_MAX_EVIDENCE_REFS, "maxEvidenceRefsPerFinding");
|
|
924
|
+
return {
|
|
925
|
+
id: ANALYST_ID$1,
|
|
926
|
+
description: "Deterministic behavioral/efficiency findings over OTLP spans — token-growth, output-decay, tool-monoculture, missing self-verification. Zero LLM; model-agnostic by construction.",
|
|
927
|
+
inputKind: "trace-store",
|
|
928
|
+
cost: { kind: "deterministic" },
|
|
929
|
+
version: "2.0.0",
|
|
930
|
+
executionConfig: {
|
|
931
|
+
kind: "behavioral-efficiency",
|
|
932
|
+
max_traces: maxTraces,
|
|
933
|
+
max_evidence_refs_per_finding: maxEvidenceRefsPerFinding
|
|
934
|
+
},
|
|
935
|
+
async analyze(store, context) {
|
|
936
|
+
const analyzedTraceIds = await listTraceIds(store, maxTraces, context.signal);
|
|
937
|
+
const findingsById = /* @__PURE__ */ new Map();
|
|
938
|
+
for (const traceId of analyzedTraceIds) {
|
|
939
|
+
context.signal?.throwIfAborted();
|
|
940
|
+
const viewed = await store.viewTrace({ trace_id: traceId });
|
|
941
|
+
if (viewed.trace_id !== traceId) throw new Error(`behavioralAnalyst: requested trace '${traceId}', received '${viewed.trace_id}'`);
|
|
942
|
+
if (!viewed.spans) throw new Error(`behavioralAnalyst: trace '${traceId}' is oversized; complete spans are required`);
|
|
943
|
+
const metrics = computeTraceMetrics(viewed.spans);
|
|
944
|
+
if (metrics.traceId !== null && metrics.traceId !== traceId) throw new Error(`behavioralAnalyst: requested trace '${traceId}', received '${metrics.traceId}'`);
|
|
945
|
+
for (const finding of deriveEfficiencyFindings(metrics)) {
|
|
946
|
+
const current = findingsById.get(finding.finding_id);
|
|
947
|
+
if (!current) {
|
|
948
|
+
findingsById.set(finding.finding_id, {
|
|
949
|
+
finding,
|
|
950
|
+
observedTraceCount: 1,
|
|
951
|
+
evidenceTraceIds: [traceId],
|
|
952
|
+
evidence: [...finding.evidence_refs]
|
|
953
|
+
});
|
|
954
|
+
continue;
|
|
955
|
+
}
|
|
956
|
+
current.observedTraceCount += 1;
|
|
957
|
+
if (current.evidence.length < maxEvidenceRefsPerFinding) {
|
|
958
|
+
current.evidenceTraceIds.push(traceId);
|
|
959
|
+
current.evidence.push(...finding.evidence_refs);
|
|
960
|
+
}
|
|
961
|
+
}
|
|
962
|
+
}
|
|
963
|
+
return [...findingsById.values()].map(({ finding, observedTraceCount, evidenceTraceIds, evidence }) => ({
|
|
964
|
+
...finding,
|
|
965
|
+
claim: AGGREGATE_CLAIM[finding.subject](observedTraceCount, analyzedTraceIds.length),
|
|
966
|
+
rationale: `${observedTraceCount}/${analyzedTraceIds.length} analyzed traces exhibited this pattern.`,
|
|
967
|
+
evidence_refs: evidence,
|
|
968
|
+
metadata: {
|
|
969
|
+
deterministic: true,
|
|
970
|
+
evidence_trace_ids: evidenceTraceIds,
|
|
971
|
+
omitted_evidence_trace_count: observedTraceCount - evidenceTraceIds.length,
|
|
972
|
+
observed_trace_count: observedTraceCount,
|
|
973
|
+
analyzed_trace_count: analyzedTraceIds.length
|
|
974
|
+
}
|
|
975
|
+
}));
|
|
976
|
+
}
|
|
977
|
+
};
|
|
1042
978
|
}
|
|
1043
|
-
function
|
|
1044
|
-
if (!Number.isSafeInteger(value) || value <
|
|
979
|
+
function positiveInteger(value, name) {
|
|
980
|
+
if (!Number.isSafeInteger(value) || value < 1) throw new RangeError(`behavioralAnalyst: ${name} must be a positive safe integer`);
|
|
1045
981
|
return value;
|
|
1046
982
|
}
|
|
1047
|
-
|
|
1048
|
-
|
|
1049
|
-
|
|
1050
|
-
|
|
1051
|
-
return
|
|
1052
|
-
|
|
1053
|
-
|
|
1054
|
-
const analystAttempts = trajectory.attempts.filter((attempt) => isRecord(attempt.artifact) && attempt.artifact.type === "analyst-run");
|
|
1055
|
-
const analysis = isRecord(trajectory.metadata?.analysis) ? trajectory.metadata.analysis : void 0;
|
|
1056
|
-
if (analystAttempts.length === 0) {
|
|
1057
|
-
if (analysis?.kind === "analyst-run") throw new TypeError(`feedbackTrajectoryToOptimizerRow: analyst trajectory "${trajectory.id}" is missing its archived run`);
|
|
1058
|
-
return;
|
|
1059
|
-
}
|
|
1060
|
-
if (analystAttempts.length !== 1) throw new TypeError(`feedbackTrajectoryToOptimizerRow: analyst trajectory "${trajectory.id}" must contain exactly one archived run`);
|
|
1061
|
-
if (analysis?.kind !== "analyst-run") throw new TypeError(`feedbackTrajectoryToOptimizerRow: analyst trajectory "${trajectory.id}" is missing review state`);
|
|
1062
|
-
const artifact = analystAttempts[0].artifact;
|
|
1063
|
-
if (!isRecord(artifact)) throw new TypeError(`feedbackTrajectoryToOptimizerRow: analyst trajectory "${trajectory.id}" has an invalid archived run`);
|
|
1064
|
-
const runId = requiredString(artifact.analystRunId, `analyst trajectory "${trajectory.id}" run id`);
|
|
1065
|
-
if (analysis.runId !== runId) throw new TypeError(`feedbackTrajectoryToOptimizerRow: analyst trajectory "${trajectory.id}" run identity does not match its review state`);
|
|
1066
|
-
const artifactRunDigest = requiredDigest(artifact.runDigest, `analyst trajectory "${trajectory.id}" archived run digest`);
|
|
1067
|
-
const storedRunDigest = requiredDigest(analysis.runDigest, `analyst trajectory "${trajectory.id}" review run digest`);
|
|
1068
|
-
if (artifactRunDigest !== storedRunDigest) throw new TypeError(`feedbackTrajectoryToOptimizerRow: analyst trajectory "${trajectory.id}" run digest does not match its review state`);
|
|
1069
|
-
const findings = snapshotAnalystFindings(artifact.findings, `analyst trajectory "${trajectory.id}"`);
|
|
1070
|
-
const findingIds = findings.map((finding) => finding.finding_id);
|
|
1071
|
-
const analystIds = stringArray(artifact.analystIds, `analyst trajectory "${trajectory.id}" analyst ids`);
|
|
1072
|
-
const attemptMetadata = analystAttempts[0].metadata;
|
|
1073
|
-
if (!isRecord(attemptMetadata)) throw new TypeError(`feedbackTrajectoryToOptimizerRow: analyst trajectory "${trajectory.id}" is missing archived run metadata`);
|
|
1074
|
-
const archivedRun = snapshotAnalystRun({
|
|
1075
|
-
run_id: runId,
|
|
1076
|
-
correlation_id: artifact.correlationId,
|
|
1077
|
-
started_at: analysis.startedAt,
|
|
1078
|
-
ended_at: analysis.endedAt,
|
|
1079
|
-
findings,
|
|
1080
|
-
per_analyst: attemptMetadata.perAnalyst,
|
|
1081
|
-
total_cost_usd: analysis.knownCostUsd,
|
|
1082
|
-
...analysis.costProvenance === void 0 ? {} : { total_cost_provenance: analysis.costProvenance },
|
|
1083
|
-
...artifact.executionPlan === void 0 ? {} : {
|
|
1084
|
-
execution_plan: artifact.executionPlan,
|
|
1085
|
-
completion: artifact.completion
|
|
1086
|
-
}
|
|
1087
|
-
}, `analyst trajectory "${trajectory.id}" archived run`);
|
|
1088
|
-
const knownAnalystIds = new Set(analystIds);
|
|
1089
|
-
for (const [index, finding] of findings.entries()) if (!knownAnalystIds.has(finding.analyst_id)) throw new TypeError(`feedbackTrajectoryToOptimizerRow: analyst trajectory "${trajectory.id}" omits generating analyst "${finding.analyst_id}" at finding ${index}`);
|
|
1090
|
-
const reviewDecisions = validateAnalystReviewDecisions({
|
|
1091
|
-
runId,
|
|
1092
|
-
runDigest: storedRunDigest,
|
|
1093
|
-
findings,
|
|
1094
|
-
analystIds,
|
|
1095
|
-
decisions: analysis.reviewDecisions,
|
|
1096
|
-
requireComplete: true
|
|
1097
|
-
});
|
|
1098
|
-
const expectedRunDigest = analystRunDigest(archivedRun);
|
|
1099
|
-
if (storedRunDigest !== expectedRunDigest) throw new TypeError(`feedbackTrajectoryToOptimizerRow: analyst trajectory "${trajectory.id}" archived run digest mismatch`);
|
|
1100
|
-
return {
|
|
1101
|
-
runId,
|
|
1102
|
-
runDigest: expectedRunDigest,
|
|
1103
|
-
findings,
|
|
1104
|
-
findingIds,
|
|
1105
|
-
analystIds,
|
|
1106
|
-
reviewDecisions
|
|
1107
|
-
};
|
|
983
|
+
//#endregion
|
|
984
|
+
//#region src/analyst/kinds/control-integrity.ts
|
|
985
|
+
const ANALYST_ID = "control-integrity";
|
|
986
|
+
function shown(value) {
|
|
987
|
+
if (value === void 0) return "<absent>";
|
|
988
|
+
const encoded = JSON.stringify(value);
|
|
989
|
+
return encoded === void 0 ? String(value) : encoded;
|
|
1108
990
|
}
|
|
1109
|
-
function
|
|
1110
|
-
const findingDecisions = review.reviewDecisions.filter((decision) => decision.verdict !== "completeness_assessed");
|
|
1111
|
-
const completeness = review.reviewDecisions.filter((decision) => decision.verdict === "completeness_assessed");
|
|
1112
|
-
if (completeness.length !== 1) throw new TypeError("feedbackTrajectoryToOptimizerRow: analyst run requires exactly one independent completeness_assessed decision");
|
|
1113
|
-
const confirmed = findingDecisions.filter((decision) => decision.verdict === "confirmed").length;
|
|
1114
|
-
const rejected = findingDecisions.length - confirmed;
|
|
1115
|
-
const emitted = review.findingIds.length;
|
|
1116
|
-
const missed = completeness[0].missedIssues.length;
|
|
1117
|
-
const precision = emitted === 0 ? 1 : confirmed / emitted;
|
|
1118
|
-
const recallDenominator = confirmed + missed;
|
|
1119
|
-
const recall = recallDenominator === 0 ? 1 : confirmed / recallDenominator;
|
|
991
|
+
function evidenceRef(namespace, value) {
|
|
1120
992
|
return {
|
|
1121
|
-
|
|
1122
|
-
|
|
1123
|
-
|
|
1124
|
-
counts: {
|
|
1125
|
-
emitted,
|
|
1126
|
-
confirmed,
|
|
1127
|
-
rejected,
|
|
1128
|
-
missed
|
|
1129
|
-
}
|
|
993
|
+
kind: "metric",
|
|
994
|
+
uri: `supervisor-run://${encodeURIComponent(namespace)}/${value.path}`,
|
|
995
|
+
excerpt: shown(value.value)
|
|
1130
996
|
};
|
|
1131
|
-
}
|
|
1132
|
-
|
|
1133
|
-
|
|
1134
|
-
const
|
|
1135
|
-
|
|
1136
|
-
|
|
1137
|
-
|
|
1138
|
-
|
|
1139
|
-
|
|
1140
|
-
|
|
1141
|
-
|
|
1142
|
-
|
|
1143
|
-
|
|
1144
|
-
|
|
1145
|
-
|
|
1146
|
-
|
|
1147
|
-
|
|
1148
|
-
|
|
1149
|
-
|
|
1150
|
-
|
|
1151
|
-
|
|
1152
|
-
|
|
1153
|
-
assertOnlyKeys(value, [
|
|
1154
|
-
"runDigest",
|
|
1155
|
-
"verdict",
|
|
1156
|
-
"missedIssues",
|
|
1157
|
-
"source",
|
|
1158
|
-
"reviewerId",
|
|
1159
|
-
"reviewId",
|
|
1160
|
-
"reason",
|
|
1161
|
-
"decidedAt"
|
|
1162
|
-
], `analyst review decision ${index}`);
|
|
1163
|
-
completenessCount += 1;
|
|
1164
|
-
if (completenessCount > 1) throw new TypeError("duplicate completeness_assessed analyst review decision");
|
|
1165
|
-
return {
|
|
1166
|
-
runDigest,
|
|
1167
|
-
verdict: "completeness_assessed",
|
|
1168
|
-
missedIssues: validateMissedIssues(value.missedIssues, findingsById, `analyst review decision ${index}`),
|
|
1169
|
-
source,
|
|
1170
|
-
reviewerId,
|
|
1171
|
-
reviewId,
|
|
1172
|
-
reason,
|
|
1173
|
-
decidedAt
|
|
1174
|
-
};
|
|
997
|
+
}
|
|
998
|
+
/** Translate typed supervisor-run integrity issues into the shared analyst envelope. */
|
|
999
|
+
function emitControlIntegrityFindings(input, producedAt) {
|
|
1000
|
+
const report = analyzeSupervisorRunIntegrity(input, { capturedAt: producedAt });
|
|
1001
|
+
return report.issues.map((issue) => makeFinding({
|
|
1002
|
+
analyst_id: ANALYST_ID,
|
|
1003
|
+
produced_at: producedAt,
|
|
1004
|
+
area: issue.area,
|
|
1005
|
+
severity: issue.severity,
|
|
1006
|
+
subject: `${report.runRef}/${issue.subject}`,
|
|
1007
|
+
claim: issue.claim,
|
|
1008
|
+
rationale: issue.detail,
|
|
1009
|
+
evidence_refs: issue.evidence.map((value) => evidenceRef(report.runRef, value)),
|
|
1010
|
+
recommended_action: issue.recommendedAction,
|
|
1011
|
+
validation_plan: "Re-run this deterministic analyst on the retained SupervisorRunSources or SupervisorRunTree after correcting the producer.",
|
|
1012
|
+
confidence: 1,
|
|
1013
|
+
metadata: {
|
|
1014
|
+
integrity_code: issue.code,
|
|
1015
|
+
integrity_input: report.input,
|
|
1016
|
+
integrity_run_ref: report.runRef,
|
|
1017
|
+
integrity_subject: issue.subject,
|
|
1018
|
+
...issue.metadata
|
|
1175
1019
|
}
|
|
1176
|
-
|
|
1177
|
-
assertOnlyKeys(value, [
|
|
1178
|
-
"runDigest",
|
|
1179
|
-
"findingId",
|
|
1180
|
-
"findingDigest",
|
|
1181
|
-
"verdict",
|
|
1182
|
-
"source",
|
|
1183
|
-
"reviewerId",
|
|
1184
|
-
"reviewId",
|
|
1185
|
-
"reason",
|
|
1186
|
-
"decidedAt"
|
|
1187
|
-
], `analyst review decision ${index}`);
|
|
1188
|
-
const findingId = requiredString(value.findingId, `analyst review decision ${index} findingId`);
|
|
1189
|
-
const finding = findingsById.get(findingId);
|
|
1190
|
-
if (!finding) throw new TypeError(`analyst review decision references unknown finding id "${findingId}"`);
|
|
1191
|
-
if (seenFindingIds.has(findingId)) throw new TypeError(`duplicate analyst review decision for finding id "${findingId}"`);
|
|
1192
|
-
seenFindingIds.add(findingId);
|
|
1193
|
-
const findingDigest = requiredString(value.findingDigest, `analyst review decision ${index} findingDigest`);
|
|
1194
|
-
const expectedDigest = analystFindingDigest(finding);
|
|
1195
|
-
if (findingDigest !== expectedDigest) throw new TypeError(`analyst review decision ${index} digest mismatch for finding id "${findingId}"`);
|
|
1196
|
-
return {
|
|
1197
|
-
runDigest,
|
|
1198
|
-
findingId,
|
|
1199
|
-
findingDigest: expectedDigest,
|
|
1200
|
-
verdict: value.verdict,
|
|
1201
|
-
source,
|
|
1202
|
-
reviewerId,
|
|
1203
|
-
reviewId,
|
|
1204
|
-
reason,
|
|
1205
|
-
decidedAt
|
|
1206
|
-
};
|
|
1207
|
-
});
|
|
1208
|
-
if (input.requireComplete) {
|
|
1209
|
-
const missing = findings.map((finding) => finding.finding_id).filter((findingId) => !seenFindingIds.has(findingId));
|
|
1210
|
-
if (missing.length > 0) throw new TypeError(`feedbackTrajectoryToOptimizerRow: missing independent decisions for finding ids: ${missing.join(", ")}`);
|
|
1211
|
-
if (completenessCount !== 1) throw new TypeError("feedbackTrajectoryToOptimizerRow: analyst run requires exactly one independent completeness_assessed decision");
|
|
1212
|
-
}
|
|
1213
|
-
return decisions;
|
|
1020
|
+
}));
|
|
1214
1021
|
}
|
|
1215
|
-
|
|
1216
|
-
|
|
1217
|
-
|
|
1218
|
-
|
|
1219
|
-
|
|
1220
|
-
|
|
1022
|
+
/** Deterministic Analyst adapter for `SupervisorRunSources | SupervisorRunTree`. */
|
|
1023
|
+
var ControlIntegrityAnalyst = class {
|
|
1024
|
+
id = ANALYST_ID;
|
|
1025
|
+
description = "Deterministic supervisor-run integrity checks with explicit unavailable evidence.";
|
|
1026
|
+
inputKind = "custom";
|
|
1027
|
+
cost = {
|
|
1028
|
+
kind: "deterministic",
|
|
1029
|
+
est_usd_per_run: 0
|
|
1030
|
+
};
|
|
1031
|
+
version = "2.0.0";
|
|
1032
|
+
executionConfig = {
|
|
1033
|
+
kind: "control-integrity",
|
|
1034
|
+
produced_at_source: "tags.producedAt-or-system-clock"
|
|
1035
|
+
};
|
|
1036
|
+
async analyze(input, ctx) {
|
|
1037
|
+
const findings = emitControlIntegrityFindings(input, ctx.tags?.producedAt ?? (/* @__PURE__ */ new Date()).toISOString());
|
|
1038
|
+
ctx.log?.(`control-integrity: ${findings.length} finding(s)`, { input: "nodes" in input ? "SupervisorRunTree" : "SupervisorRunSources" });
|
|
1039
|
+
return findings;
|
|
1221
1040
|
}
|
|
1222
|
-
}
|
|
1223
|
-
|
|
1224
|
-
|
|
1225
|
-
|
|
1226
|
-
|
|
1227
|
-
|
|
1228
|
-
|
|
1041
|
+
};
|
|
1042
|
+
const CONTROL_INTEGRITY_ANALYST = new ControlIntegrityAnalyst();
|
|
1043
|
+
const FAILURE_MODE_KIND_SPEC = {
|
|
1044
|
+
id: "failure-mode",
|
|
1045
|
+
description: "Ranks failures by the work each one burned — spans, model calls, and wall-clock between the error and the span where the agent resumed — and treats a failure that reached a human as the most expensive class.",
|
|
1046
|
+
area: "failure-mode",
|
|
1047
|
+
version: "2.0.0",
|
|
1048
|
+
instructions: `You are a failure-cost analyst for an OTLP trace dataset. Your job is to find the failures that **burned the most agent work** and to price each one in spans, model calls, and wall-clock. How often an error occurred is not its cost and is not your output.
|
|
1049
|
+
|
|
1050
|
+
${findingSubjectGrammarPromptFor("failure-mode")}
|
|
1051
|
+
|
|
1052
|
+
Use a lowercase cluster label that names the cost shape, not the error string: "human-corrected-file-overwrite", "unrecovered-auth-loop", "long-detour-schema-guess".
|
|
1053
|
+
|
|
1054
|
+
WORKLIST → RECOVERY BOUNDARY → PRICE → CITE protocol:
|
|
1055
|
+
|
|
1056
|
+
1. \`getDatasetOverview({})\` first. Read \`errors.trace_count\`, \`errors.span_count\`, and \`error_clusters[]\` — each cluster carries \`signature\`, \`tool_name\`, \`trace_count\`, \`span_count\`, \`prevalence\`, \`exemplar_trace_ids\`, and \`exemplar_span_ids\`. This is the deterministic analyzer's output, already computed for free. It is your WORKLIST: the exemplar ids name the spans you go price. It is never your findings.
|
|
1057
|
+
2. Reach error-bearing traces the exemplars miss with \`queryTraces(filters={"has_errors": true}, limit=...)\`. Prefer long traces and traces whose errors sit early: a failure has room to be expensive only when work followed it.
|
|
1058
|
+
3. Take one candidate error span E and read its trace in order. \`viewTrace\` returns that trace's spans sorted by \`start_time\`, which is the ordering every distance below is counted in. When \`viewTrace\` returns \`oversized\` instead of spans, locate candidates with \`searchTrace\` and pull the exact ids with \`viewSpans\`; use \`searchSpan\` when one span's payload is truncated.
|
|
1059
|
+
4. Find the RECOVERY BOUNDARY R — the first span after E in which the agent resumed productive work. Productive means a different, task-advancing action. A retry of the same call with the same arguments, a re-read of the same file, a re-plan of the same step, and an apology are all still inside the failure. Then price it:
|
|
1060
|
+
- spans_burned: spans strictly between E and R
|
|
1061
|
+
- model_calls_burned: how many of those spans have kind \`LLM\`
|
|
1062
|
+
- wall_clock_ms: R \`end_time\` minus E \`start_time\`
|
|
1063
|
+
Report all three. If the agent never resumed, R is the trace's last span and the failure is UNRECOVERED.
|
|
1064
|
+
5. Set severity from what ENDED the failure and from the measured distance, never from how many times the signature appears:
|
|
1065
|
+
- HUMAN-CORRECTED — a human turn after E supplies the correction. This is the most expensive class at any span count, because the failure escaped the agent entirely and reached the user: severity critical. Detect it as a new user-role message in a later \`LLM\` span's input attributes, or a new root-level \`AGENT\` span in which the user restates or repairs the task. Search with patterns like \`"role"\\s*:\\s*"user"\`, \`(that|this) is (wrong|not what)\`, \`I (said|asked|told you)\`, \`stop\`, \`undo\`, \`revert\`, \`you (deleted|broke|missed|ignored)\`, \`try again\`. Quote the human's words exactly.
|
|
1066
|
+
- UNRECOVERED — the trace ends without resumption: severity critical.
|
|
1067
|
+
- LONG DETOUR — 10 or more spans burned, or 3 or more \`LLM\` spans burned: severity high.
|
|
1068
|
+
- SHORT DETOUR — 2 to 9 spans burned: severity medium.
|
|
1069
|
+
- SELF-RECOVERED ON THE NEXT SPAN — cost is approximately zero. Do not emit it. That is resilience, and reporting it dilutes the ranking.
|
|
1070
|
+
One error that cost 20 spans outranks 50 errors that each cost one retry. Frequency may appear in a finding only as a multiplier on a measured per-instance cost, never as the reason for its severity.
|
|
1071
|
+
6. **Cluster, do not enumerate.** Errors sharing a root cause AND a recovery shape are ONE finding: report the summed cost across instances, state how many instances it covers, and cite the bounding pair of the single most expensive instance. Two errors with the same signature but different recovery shapes — one shrugged off, one human-corrected — are NOT the same finding; the expensive one is the finding and the cheap one is noise.
|
|
1072
|
+
|
|
1073
|
+
FORBIDDEN OUTPUT. Each of the following duplicates the deterministic pass, and a duplicate finding is worse than no finding because it costs a reviewer the same attention while carrying no new information:
|
|
1074
|
+
- a count of errors by tool ("Bash is the dominant failure surface, 54 error spans")
|
|
1075
|
+
- a count or inventory of error signatures ("47 distinct error signatures")
|
|
1076
|
+
- an error rate ("79 of 5260 tool calls ended in errors")
|
|
1077
|
+
- any restatement of \`prevalence\`, \`trace_count\`, or \`span_count\` from \`error_clusters[]\`
|
|
1078
|
+
Every finding MUST state its measured cost — spans burned, model calls burned, wall-clock — in the claim, and MUST cite the two spans that bound it: the error span and the recovery span, or the escaping human turn, or the trace's last span when unrecovered. A failure you cannot bound with two spans is one you did not measure: drop it.
|
|
1079
|
+
|
|
1080
|
+
**Adjudicate boundaries with subqueries.** The single judgement call in this protocol is whether a post-error span is genuine resumption or more flailing. Load E, the spans between, and the candidate R, then send one bounded \`llm_query\` per candidate carrying those exact excerpts and asking which span first advances the task. Subqueries cannot call trace tools, so a trace id tells them nothing — paste the excerpts. Accept a boundary only when the excerpts you loaded support the answer.
|
|
1081
|
+
|
|
1082
|
+
Confidence 0.9+ when both bounding spans are quoted and the spans between them were counted directly; 0.6-0.8 when an oversized trace forced you to sample the interval and the distance is an estimate; below 0.5 does not belong in this analyst, because an unmeasurable cost is not a finding. Keep the recommended action a short imperative; the improvement analyst expands it.
|
|
1083
|
+
|
|
1084
|
+
If every error in this dataset was recovered on the next span, return an empty findings array. That is the correct answer for a dataset whose failures were all cheap. Do not backfill it with the counts listed above and do not pad it with speculation.`,
|
|
1085
|
+
toolGroup: "all",
|
|
1086
|
+
limits: {
|
|
1087
|
+
maxLlmCalls: 8,
|
|
1088
|
+
maxIterations: 24,
|
|
1089
|
+
maxToolCalls: 80
|
|
1090
|
+
},
|
|
1091
|
+
minimumEvidenceCitations: 2
|
|
1092
|
+
};
|
|
1093
|
+
const IMPROVEMENT_KIND_SPEC = {
|
|
1094
|
+
id: "improvement",
|
|
1095
|
+
description: "Converts upstream failure / gap / poisoning findings into concrete locus-named edits (prompt, tool-doc, RAG, scaffolding) with leverage grades.",
|
|
1096
|
+
area: "improvement",
|
|
1097
|
+
version: "1.2.0",
|
|
1098
|
+
instructions: `You are a self-improvement analyst. Your job is to propose **concrete, locus-named edits** the agent's runtime should adopt to fix the failure modes, knowledge gaps, and poisonings present in this dataset.
|
|
1099
|
+
|
|
1100
|
+
Upstream analysts have already classified the problems. Your job is to convert each problem into a *change to make* and grade its expected leverage. Each finding is one proposed edit.
|
|
1101
|
+
|
|
1102
|
+
${findingSubjectGrammarPromptFor("improvement")}
|
|
1103
|
+
|
|
1104
|
+
DISCOVERY → CANDIDATE-FIXES → COMPETE → CITE protocol:
|
|
1105
|
+
|
|
1106
|
+
1. \`getDatasetOverview({})\` first. Note the agents, tools, and any system-prompt fingerprints (look for the prompt text echoed in early spans).
|
|
1107
|
+
2. For each high-severity failure pattern, generate 2-3 candidate fixes. Real candidate axes:
|
|
1108
|
+
- **System-prompt edit** — add an instruction, remove a misleading one, restructure precedence
|
|
1109
|
+
- **Tool description edit** — rewrite a tool's description so the agent picks it correctly / passes valid args
|
|
1110
|
+
- **New tool** — add a tool the agent kept emulating in code
|
|
1111
|
+
- **RAG ingestion** — add a document or correct a stale one
|
|
1112
|
+
- **Memory invalidation** — clear cached prior-run decisions that no longer apply
|
|
1113
|
+
- **Scaffolding** — add a precondition check, a retry policy, a turn budget, a verification step
|
|
1114
|
+
- **Output schema** — narrow the agent's output to forbid the failure shape
|
|
1115
|
+
- **Skill / MCP / hook / subagent** — change the reusable profile component responsible for the behavior
|
|
1116
|
+
- **Workflow / rollout policy** — change orchestration, budget, sampling, or stopping behavior
|
|
1117
|
+
- **Code** — change an implementation path when profile edits cannot repair the behavior
|
|
1118
|
+
3. **Compare candidate fixes with bounded subqueries.** Load the representative failure excerpts, then send one \`llm_query\` per candidate-fix axis the same evidence. Ask for likely effect, side effects, and implementation scope. Subqueries cannot call trace tools; trace ids alone are insufficient context.
|
|
1119
|
+
4. After the comparisons return, **pick the winning candidate per cluster** based on expected effect and risk, then emit ONE finding. Keep the alternatives and rejection reasons in the rationale so the recommendation is auditable.
|
|
1120
|
+
5. **Cross-reference upstream findings.** Cite prior failure-mode or knowledge-gap findings as \`finding://<prior-finding-id>\`. This builds the dependency graph that lets the dashboard show "fix #X resolves failure modes A, B, C."
|
|
1121
|
+
|
|
1122
|
+
For each winning recommendation, emit ONE finding. Use one exact locus from the subject grammar and state the edit in one sentence. Match leverage to the source failure's severity; use medium for quality-of-life changes and info for cleanup with no behavioral effect. Cite the targeted \`finding://<id>\` when available and the most representative span when useful. Quote the problem being fixed. Use confidence 0.85+ for a mechanical fix to a well-evidenced failure, 0.6-0.8 when judgment is required, and <0.5 for speculation. Explain in at most two sentences why this candidate beat its alternatives. The recommended action must be the literal diff, quoted replacement, tool description, or setting change.
|
|
1123
|
+
|
|
1124
|
+
If no upstream failure findings exist in this run, derive your own from the trace dataset using the failure-mode protocol inline (\`searchTrace\` for STATUS_CODE_ERROR / MaxTurnsExceeded / etc.). Prefer upstream findings when present because the analysts are designed to chain.
|
|
1125
|
+
|
|
1126
|
+
Do NOT propose a fix you cannot defend with evidence. "Tighten the prompt" is not a finding; "Add 'When the user asks for X, always Y' to the system prompt section "request-classification"" is.`,
|
|
1127
|
+
toolGroup: "all",
|
|
1128
|
+
limits: {
|
|
1129
|
+
maxLlmCalls: 8,
|
|
1130
|
+
maxIterations: 30,
|
|
1131
|
+
maxToolCalls: 80,
|
|
1132
|
+
maxOutputChars: 12e3
|
|
1133
|
+
}
|
|
1134
|
+
};
|
|
1135
|
+
const INTENT_DIVERGENCE_KIND_SPEC = {
|
|
1136
|
+
id: "intent-divergence",
|
|
1137
|
+
description: "Anchors each corrective human turn to the earliest assistant turn where the divergence from stated intent was already detectable, and prices it in burned turns.",
|
|
1138
|
+
area: "intent-divergence",
|
|
1139
|
+
version: "1.0.0",
|
|
1140
|
+
instructions: `You are an intent-divergence analyst for an OTLP trace dataset. Your job is to find where the agent stopped doing what the user asked, and to establish **how early that was knowable**.
|
|
1141
|
+
|
|
1142
|
+
This dataset carries ground truth the other analysts do not have. When a human turn corrects the agent — "no, I meant…", "stop doing X", "why are you still asking" — the user has labelled a divergence for you. That label marks the END of the divergence, not its start. Your finding is the start: the earliest assistant turn already off-intent, the signal in the user's stated intent it contradicted, and the number of turns burned before the correction landed.
|
|
1143
|
+
|
|
1144
|
+
${findingSubjectGrammarPromptFor("intent-divergence")}
|
|
1145
|
+
|
|
1146
|
+
DISCOVERY → BACKTRACK → QUANTIFY → CITE protocol:
|
|
1147
|
+
|
|
1148
|
+
1. \`getDatasetOverview({})\` first. Note the agent names and \`sample_trace_ids\`; conversational datasets carry several human turns per trace and those are the ones worth pulling.
|
|
1149
|
+
2. **DISCOVERY — find the corrective human turns.** Use \`searchTrace\` for user-authored corrections:
|
|
1150
|
+
Every pattern MUST begin with \`(?i)\`. The trace store compiles a search pattern case-sensitively unless it opens with that flag, and corrections are overwhelmingly sentence-initial and capitalised ("Stop", "No,", "I said"), so a lowercase pattern silently returns zero hits.
|
|
1151
|
+
- Imperative stops: \`(?i)\\bstop\\b\`, \`(?i)\\b(don'?t|do not)\\b\`, \`(?i)^\\s*no[,.!]\`, \`(?i)\\bnot what I (wanted|asked|meant)\\b\`, \`(?i)\\bthat'?s not what\\b\`, \`(?i)\\brevert\\b\`, \`(?i)\\bstart over\\b\`
|
|
1152
|
+
- Repetition and frustration: \`(?i)\\bI (said|asked|told you)\\b\`, \`(?i)\\balready told you\\b\`, \`(?i)\\bwhy are you (still )?\\w+ing\\b\`, \`(?i)\\bjust (do|use|write|make)\\b\`, \`(?i)\\bagain\\b\`
|
|
1153
|
+
- Clarification markers (second-order, see step 6): \`(?i)\\bI mean(t)?\\b\`, \`(?i)\\bto be clear\\b\`, \`(?i)\\bactually,? I (want|need)\\b\`, \`(?i)\\bwhat I meant\\b\`, \`(?i)\\blet me rephrase\\b\`
|
|
1154
|
+
Attribute every match to a role before using it. The same phrases inside an assistant turn are hedging, not correction.
|
|
1155
|
+
3. **BACKTRACK — walk the trace backward from the correction.** Read the ordered turns preceding it with \`viewTrace\` / \`viewSpans\`. Locate the user's ORIGINAL statement of intent (usually the first human turn, sometimes a constraint added mid-run), then locate the FIRST assistant turn that contradicts it. That turn, not the correction, anchors the finding. Name the concrete signal already available at that turn: an explicit constraint the agent violated, a scope it exceeded, a different question it answered, a file or target the user never named, a decision the user had already made.
|
|
1156
|
+
4. **QUANTIFY — count the burned turns.** K = (index of the corrective human turn) − (index of the earliest detectable assistant turn). Report both indices and K. If you cannot locate an earliest-detectable turn, you have a correction and not a divergence: drop it.
|
|
1157
|
+
5. **CLUSTER.** Repeated corrections about the same misread intent are ONE finding citing all of them, not one finding per corrective turn. Corrections about genuinely different intents in the same trace are separate findings.
|
|
1158
|
+
6. **Second-order — the request itself was ambiguous.** When the user CLARIFIES rather than corrects ("I mean the staging config, not prod"), the agent's reading was defensible and the request was underspecified or self-contradictory. Emit those against a \`scaffolding:*\` locus — a clarifying-question policy or a plan checkpoint — because the fix is one sharp question before acting, priced against the K turns the ambiguity cost. Do not charge the agent's instructions for a request that could not be resolved from its own text.
|
|
1159
|
+
|
|
1160
|
+
**Test competing anchors with subqueries.** After the backward walk, load the excerpts for the original intent turn, your candidate earliest-detectable turn, and the turn before it. Send one bounded \`llm_query\` per candidate anchor asking whether the divergence is already present in that turn's text alone. Subqueries cannot call trace tools, and a turn index alone is insufficient context. Take the EARLIEST candidate the loaded excerpts support.
|
|
1161
|
+
|
|
1162
|
+
For each cluster, emit ONE finding. Use an exact locus from the subject grammar — the surface whose edit prevents the divergence, not the surface where it surfaced. State the claim in the shape "turn N did X against stated intent Y; user corrected at turn N+K, burning K turns." Rate it critical when the divergence produced a user-visible or irreversible action, high when K is 3 or more or the run ended without the intent satisfied, medium for one or two burned turns, and low for a cosmetic correction. Cite BOTH spans with exact quotes: the earliest-detectable assistant turn and the corrective human turn. Use confidence 0.85+ when the stated intent and the diverging turn are both quotable and the contradiction is explicit, and 0.6-0.8 when the earliest-detectable turn is inferred from surrounding context. The recommended action must be the literal instruction, question, or checkpoint to add.
|
|
1163
|
+
|
|
1164
|
+
Do NOT report a divergence the agent noticed and reversed inside the same turn — that is self-correction and it cost the user nothing. Do NOT report a human turn that adds NEW scope; a changed mind is not a divergence. If the dataset holds no corrective or clarifying human turns, return an empty findings array instead of grading tone.`,
|
|
1165
|
+
toolGroup: "all",
|
|
1166
|
+
limits: {
|
|
1167
|
+
maxLlmCalls: 6,
|
|
1168
|
+
maxIterations: 22,
|
|
1169
|
+
maxToolCalls: 64
|
|
1170
|
+
},
|
|
1171
|
+
minimumEvidenceCitations: 2
|
|
1172
|
+
};
|
|
1173
|
+
const KNOWLEDGE_GAP_KIND_SPEC = {
|
|
1174
|
+
id: "knowledge-gap",
|
|
1175
|
+
description: "Identifies missing or stale pieces of knowledge — primarily against the agent-knowledge wiki — and attributes each to the runtime layer (wiki page, claim, raw source, websearch, tool-doc, system-prompt, memory) that should have held it.",
|
|
1176
|
+
area: "knowledge-gap",
|
|
1177
|
+
version: "1.2.0",
|
|
1178
|
+
instructions: `You are a knowledge-gap analyst for an OTLP trace dataset. Your job is to identify the **specific pieces of information the agent lacked, or that were stale**, that caused poor decisions.
|
|
1179
|
+
|
|
1180
|
+
The agent under analysis maintains a curated knowledge base via \`@tangle-network/agent-knowledge\` — a wiki of \`KnowledgePage\`s with raw source anchors, claims, and relations. The primary expected store of agent-knowable facts IS that wiki. A "knowledge gap" is anything the agent had to discover or guess at run-time that the wiki should have held — or an outdated/contradictory fact the agent picked up from a non-wiki source.
|
|
1181
|
+
|
|
1182
|
+
${findingSubjectGrammarPromptFor("knowledge-gap")}
|
|
1183
|
+
|
|
1184
|
+
DISCOVERY → ATTRIBUTE-TO-LAYER → CITE protocol:
|
|
1185
|
+
|
|
1186
|
+
1. \`getDatasetOverview({})\` first. Note which agents, tools, and models appear.
|
|
1187
|
+
2. Pull traces where the agent shows gap signals. The strongest signals are:
|
|
1188
|
+
- Self-correction turns ("I assumed X but…", "let me re-check", "actually,")
|
|
1189
|
+
- Clarifying-question turns where the agent asked the user something the runtime should have surfaced
|
|
1190
|
+
- Repeated retrieval / lookup calls for the same artifact with slightly varied queries
|
|
1191
|
+
- Tool errors that name a missing argument or unknown resource
|
|
1192
|
+
- Web-search calls returning pages dated before a known cutoff for content that changes (versioned APIs, schemas, policies)
|
|
1193
|
+
- Agent quoting a tool's docs / system prompt incorrectly because the actual text was insufficient
|
|
1194
|
+
- Fabricated identifiers that don't appear in dataset \`sample_trace_ids\`
|
|
1195
|
+
Use \`searchTrace\` with patterns like \`I (don.?t|do not) know\`, \`assumed\`, \`unclear\`, \`could you (clarify|tell me|provide)\`, \`not found\`, \`undefined\`, \`unknown\`, \`null\`, dates older than the analysis window, or the agent's specific clarification phrases.
|
|
1196
|
+
3. For each gap, identify the **layer of the runtime that should have prevented it** and use its exact locus from the subject grammar above.
|
|
1197
|
+
4. For each defensible gap, emit ONE finding. Use an exact locus from the subject grammar and name the missing or stale knowledge (for example, "wiki has no page on invoice line-item shape; agent re-derived it from raw spans"). Rate it high when it caused failure or a clarifying question, medium for unnecessary turns, and low for minor inefficiency. Cite the span where the question, correction, retrieval miss, or stale result surfaced and quote it exactly. Use confidence 0.85+ when the agent articulated the gap and 0.6-0.8 when inferred. Recommend a concrete wiki edit for an agent-knowledge locus or a prompt/tool-description edit otherwise.
|
|
1198
|
+
|
|
1199
|
+
**Compare layers over loaded evidence.** After the first scan, load the exact excerpts behind candidates across \`agent-knowledge:*\`, \`websearch:outdated\`, \`tool-doc:*\`, \`system-prompt:*\`, and \`memory:*\`. Use one bounded \`llm_query\` per layer to classify those excerpts. Subqueries cannot call trace tools. Merge their classifications into the final finding set only when the source excerpts support them.
|
|
1200
|
+
|
|
1201
|
+
Do NOT report a gap that the agent later recovered from cleanly within the same turn. That is resilience, not a gap. Cite the non-recovery version when both exist.`,
|
|
1202
|
+
toolGroup: "discoveryAndSearch",
|
|
1203
|
+
limits: {
|
|
1204
|
+
maxLlmCalls: 5,
|
|
1205
|
+
maxIterations: 18,
|
|
1206
|
+
maxToolCalls: 48
|
|
1229
1207
|
}
|
|
1230
|
-
|
|
1231
|
-
|
|
1232
|
-
|
|
1233
|
-
|
|
1234
|
-
|
|
1235
|
-
|
|
1236
|
-
|
|
1237
|
-
|
|
1238
|
-
|
|
1239
|
-
|
|
1240
|
-
|
|
1241
|
-
|
|
1242
|
-
|
|
1243
|
-
|
|
1244
|
-
|
|
1245
|
-
|
|
1246
|
-
|
|
1247
|
-
|
|
1248
|
-
|
|
1249
|
-
|
|
1250
|
-
|
|
1251
|
-
|
|
1252
|
-
|
|
1253
|
-
|
|
1254
|
-
|
|
1255
|
-
|
|
1256
|
-
|
|
1257
|
-
|
|
1258
|
-
|
|
1259
|
-
|
|
1260
|
-
|
|
1261
|
-
|
|
1262
|
-
|
|
1263
|
-
|
|
1264
|
-
|
|
1265
|
-
|
|
1266
|
-
|
|
1267
|
-
|
|
1268
|
-
|
|
1269
|
-
|
|
1270
|
-
|
|
1271
|
-
|
|
1272
|
-
|
|
1273
|
-
|
|
1274
|
-
|
|
1275
|
-
|
|
1276
|
-
|
|
1277
|
-
|
|
1278
|
-
|
|
1279
|
-
|
|
1280
|
-
|
|
1281
|
-
|
|
1282
|
-
|
|
1283
|
-
|
|
1284
|
-
|
|
1285
|
-
|
|
1286
|
-
|
|
1287
|
-
};
|
|
1288
|
-
});
|
|
1289
|
-
}
|
|
1290
|
-
function validateEvidenceRefs(value, context) {
|
|
1291
|
-
if (!Array.isArray(value)) throw new TypeError(`${context} must be an array`);
|
|
1292
|
-
return value.map((evidence, index) => {
|
|
1293
|
-
const evidenceContext = `${context} ${index}`;
|
|
1294
|
-
if (!isRecord(evidence)) throw new TypeError(`${evidenceContext} must be an object`);
|
|
1295
|
-
assertOnlyKeys(evidence, [
|
|
1296
|
-
"kind",
|
|
1297
|
-
"uri",
|
|
1298
|
-
"excerpt"
|
|
1299
|
-
], evidenceContext);
|
|
1300
|
-
if (evidence.kind !== "span" && evidence.kind !== "event" && evidence.kind !== "artifact" && evidence.kind !== "finding" && evidence.kind !== "metric") throw new TypeError(`${evidenceContext} kind is invalid`);
|
|
1301
|
-
const uri = requiredString(evidence.uri, `${evidenceContext} uri`);
|
|
1302
|
-
const excerpt = evidence.excerpt;
|
|
1303
|
-
optionalString(excerpt, `${evidenceContext} excerpt`);
|
|
1304
|
-
return {
|
|
1305
|
-
kind: evidence.kind,
|
|
1306
|
-
uri,
|
|
1307
|
-
...excerpt === void 0 ? {} : { excerpt }
|
|
1308
|
-
};
|
|
1309
|
-
});
|
|
1310
|
-
}
|
|
1311
|
-
function assertOnlyKeys(value, allowed, name) {
|
|
1312
|
-
const allowedKeys = new Set(allowed);
|
|
1313
|
-
const unexpected = Object.keys(value).filter((key) => !allowedKeys.has(key));
|
|
1314
|
-
if (unexpected.length > 0) throw new TypeError(`${name} contains unknown fields: ${unexpected.sort().join(", ")}`);
|
|
1315
|
-
}
|
|
1316
|
-
function stringArray(value, name) {
|
|
1317
|
-
if (!Array.isArray(value) || value.some((item) => typeof item !== "string")) throw new TypeError(`${name} must be an array of strings`);
|
|
1318
|
-
const strings = value.map((item) => requiredString(item, name));
|
|
1319
|
-
if (new Set(strings).size !== strings.length) throw new TypeError(`${name} must contain unique values`);
|
|
1320
|
-
return strings;
|
|
1321
|
-
}
|
|
1322
|
-
function requiredString(value, name) {
|
|
1323
|
-
if (typeof value !== "string" || value.trim().length === 0) throw new TypeError(`${name} must be a non-empty string`);
|
|
1324
|
-
return value;
|
|
1325
|
-
}
|
|
1326
|
-
function requiredDigest(value, name) {
|
|
1327
|
-
const digest = requiredString(value, name);
|
|
1328
|
-
if (!/^sha256:[a-f0-9]{64}$/.test(digest)) throw new TypeError(`${name} must be a sha256 digest`);
|
|
1329
|
-
return digest;
|
|
1330
|
-
}
|
|
1331
|
-
function optionalString(value, name) {
|
|
1332
|
-
if (value !== void 0 && typeof value !== "string") throw new TypeError(`${name} must be a string`);
|
|
1333
|
-
}
|
|
1334
|
-
function canonicalTimestamp(value, name) {
|
|
1335
|
-
const timestamp = requiredString(value, name);
|
|
1336
|
-
const parsed = new Date(timestamp);
|
|
1337
|
-
if (Number.isNaN(parsed.valueOf()) || parsed.toISOString() !== timestamp) throw new TypeError(`${name} must be a canonical ISO 8601 UTC timestamp`);
|
|
1338
|
-
return timestamp;
|
|
1339
|
-
}
|
|
1340
|
-
function isAnalystReviewSource(value) {
|
|
1341
|
-
return value === "user" || value === "judge" || value === "environment" || value === "metric" || value === "policy";
|
|
1342
|
-
}
|
|
1343
|
-
function isRecord(value) {
|
|
1344
|
-
return typeof value === "object" && value !== null && !Array.isArray(value);
|
|
1345
|
-
}
|
|
1208
|
+
};
|
|
1209
|
+
const KNOWLEDGE_POISONING_KIND_SPEC = {
|
|
1210
|
+
id: "knowledge-poisoning",
|
|
1211
|
+
description: "Identifies confident-but-wrong actions caused by stale memory, contradicting RAG, deprecated tool docs, or outdated system-prompt instructions.",
|
|
1212
|
+
area: "knowledge-poisoning",
|
|
1213
|
+
version: "1.2.0",
|
|
1214
|
+
instructions: `You are a knowledge-poisoning analyst for an OTLP trace dataset. Your job is to identify cases where the agent **confidently used wrong information** — not where it lacked information (that's the knowledge-gap analyst).
|
|
1215
|
+
|
|
1216
|
+
${findingSubjectGrammarPromptFor("knowledge-poisoning")}
|
|
1217
|
+
|
|
1218
|
+
DISCOVERY → DUAL-VERIFY → CITE protocol:
|
|
1219
|
+
|
|
1220
|
+
1. \`getDatasetOverview({})\` first. Identify the agents, models, and tools.
|
|
1221
|
+
2. Pull traces where the agent's confident action was later contradicted. Strongest signals:
|
|
1222
|
+
- Agent stated a fact in one span; a later span surfaced contradictory evidence; the agent then proceeded anyway or fabricated reconciliation.
|
|
1223
|
+
- Tool call with stale arguments (an id that no longer exists, an API shape that changed).
|
|
1224
|
+
- Agent cited an \`agent-knowledge\` wiki page or claim whose content contradicts the trace's own evidence — the wiki itself drifted.
|
|
1225
|
+
- Web-search result the agent cited that returned an outdated page; agent treated it as canonical.
|
|
1226
|
+
- System-prompt instruction the agent followed that ground-truth evidence in the trace contradicts (e.g. prompt says "use endpoint A"; tool reply says "endpoint A deprecated, use B").
|
|
1227
|
+
- Repeated wrong-shape parsing despite the tool's actual output proving the shape.
|
|
1228
|
+
3. Use \`searchTrace\` with regex on phrases like \`actually\`, \`turns out\`, \`previously assumed\`, \`old version\`, \`deprecated\`, \`updated to\`, \`now uses\`, or specific entity names you suspect have changed.
|
|
1229
|
+
4. For each candidate poisoning, **DUAL-VERIFY**:
|
|
1230
|
+
- Confirm the agent actually acted on the false belief (cite the span where it did)
|
|
1231
|
+
- Confirm the belief is actually false in this trace's own evidence (cite the span that contradicts it)
|
|
1232
|
+
Only emit a finding when both halves are supported. If you can only support one, drop it because single-evidence poisoning findings are too speculative to be useful.
|
|
1233
|
+
|
|
1234
|
+
**Independently assess both halves.** Load the action excerpt and contradicting excerpt yourself, then send bounded \`llm_query\` calls the exact evidence for "did the agent act?" and "does the trace contradict the belief?" Subqueries cannot call trace tools. Accept a poisoning only when both assessments and the source excerpts support it.
|
|
1235
|
+
|
|
1236
|
+
For each confirmed poisoning, emit ONE finding. Use the source of the false belief as the exact subject. State "agent believed X (from source S); trace evidence shows X is false." Rate it critical for a wrong user-visible action, high when caught internally after significant waste, and medium for inefficiency. Cite BOTH the action span and the contradicting span with exact quotes. Use confidence 0.85+ when both halves have exact quotes and 0.6-0.8 when one half is inferred. Recommend the literal source correction: update the wiki claim, invalidate and re-curate the raw source, or replace the stale prompt/tool instruction.
|
|
1237
|
+
|
|
1238
|
+
Do NOT report a finding if the agent caught and corrected the false belief in the same turn. Reserve poisoning for cases where the false belief shaped downstream action.`,
|
|
1239
|
+
toolGroup: "all",
|
|
1240
|
+
limits: {
|
|
1241
|
+
maxLlmCalls: 8,
|
|
1242
|
+
maxIterations: 20,
|
|
1243
|
+
maxToolCalls: 64
|
|
1244
|
+
},
|
|
1245
|
+
minimumEvidenceCitations: 2
|
|
1246
|
+
};
|
|
1247
|
+
//#endregion
|
|
1248
|
+
//#region src/analyst/kinds/index.ts
|
|
1249
|
+
/**
|
|
1250
|
+
* The default kind suite. Order is the run order operators should use:
|
|
1251
|
+
* failure-mode and intent-divergence first (neither reads upstream
|
|
1252
|
+
* findings), gap + poisoning next (both explain the problems those two
|
|
1253
|
+
* found), improvement last (chains all four). Intent-divergence sits
|
|
1254
|
+
* ahead of improvement deliberately — a proposed edit should be able to
|
|
1255
|
+
* act on a priced divergence, and it is the only kind whose findings
|
|
1256
|
+
* carry a burned-turn cost to rank against.
|
|
1257
|
+
*/
|
|
1258
|
+
const DEFAULT_TRACE_ANALYST_KINDS = [
|
|
1259
|
+
FAILURE_MODE_KIND_SPEC,
|
|
1260
|
+
INTENT_DIVERGENCE_KIND_SPEC,
|
|
1261
|
+
KNOWLEDGE_GAP_KIND_SPEC,
|
|
1262
|
+
KNOWLEDGE_POISONING_KIND_SPEC,
|
|
1263
|
+
IMPROVEMENT_KIND_SPEC
|
|
1264
|
+
];
|
|
1346
1265
|
//#endregion
|
|
1347
1266
|
//#region src/analyst/registry.ts
|
|
1348
1267
|
/**
|
|
@@ -2447,6 +2366,87 @@ function buildDefaultAnalystRegistry(options = {}) {
|
|
|
2447
2366
|
return registry;
|
|
2448
2367
|
}
|
|
2449
2368
|
//#endregion
|
|
2450
|
-
|
|
2369
|
+
//#region src/analyst/chat-client.ts
|
|
2370
|
+
/**
|
|
2371
|
+
* Provider-neutral chat contract for every model call made by agent-eval.
|
|
2372
|
+
*
|
|
2373
|
+
* Callers choose the transport at the package boundary with `createChatClient`.
|
|
2374
|
+
* Evaluation code receives canonical requests and results without importing a
|
|
2375
|
+
* provider SDK.
|
|
2376
|
+
*/
|
|
2377
|
+
/**
|
|
2378
|
+
* Build a ChatClient bound to a specific transport. The returned client
|
|
2379
|
+
* is safe to share across analysts in a single registry run.
|
|
2380
|
+
*/
|
|
2381
|
+
function createChatClient(opts) {
|
|
2382
|
+
switch (opts.transport) {
|
|
2383
|
+
case "router": return wrapLlmClient(opts.transport, opts.defaultModel, new LlmClient({
|
|
2384
|
+
baseUrl: opts.baseUrl ?? "https://router.tangle.tools/v1",
|
|
2385
|
+
apiKey: opts.apiKey,
|
|
2386
|
+
maximumAttempts: opts.maximumAttempts
|
|
2387
|
+
}));
|
|
2388
|
+
case "cli-bridge": return wrapLlmClient(opts.transport, opts.defaultModel, new LlmClient({
|
|
2389
|
+
baseUrl: opts.baseUrl ?? "http://127.0.0.1:3344/v1",
|
|
2390
|
+
apiKey: opts.bearer ?? "",
|
|
2391
|
+
maximumAttempts: opts.maximumAttempts
|
|
2392
|
+
}));
|
|
2393
|
+
case "direct-provider": return wrapLlmClient(opts.transport, opts.defaultModel, new LlmClient({
|
|
2394
|
+
baseUrl: opts.baseUrl,
|
|
2395
|
+
apiKey: opts.apiKey,
|
|
2396
|
+
maximumAttempts: opts.maximumAttempts
|
|
2397
|
+
}));
|
|
2398
|
+
case "sandbox-sdk": return {
|
|
2399
|
+
transport: "sandbox-sdk",
|
|
2400
|
+
defaultModel: opts.defaultModel,
|
|
2401
|
+
maximumAttempts: opts.maximumAttempts,
|
|
2402
|
+
chat: async (req, callOpts) => opts.chat(resolveModel(req, opts.defaultModel), callOpts)
|
|
2403
|
+
};
|
|
2404
|
+
case "custom": return {
|
|
2405
|
+
transport: "custom",
|
|
2406
|
+
defaultModel: opts.defaultModel,
|
|
2407
|
+
maximumAttempts: opts.maximumAttempts,
|
|
2408
|
+
chat: async (req, callOpts) => opts.chat(resolveModel(req, opts.defaultModel), callOpts)
|
|
2409
|
+
};
|
|
2410
|
+
case "mock": return {
|
|
2411
|
+
transport: "mock",
|
|
2412
|
+
defaultModel: opts.defaultModel,
|
|
2413
|
+
maximumAttempts: 1,
|
|
2414
|
+
chat: async (req, callOpts) => opts.handler(resolveModel(req, opts.defaultModel), callOpts)
|
|
2415
|
+
};
|
|
2416
|
+
}
|
|
2417
|
+
}
|
|
2418
|
+
function wrapLlmClient(transport, defaultModel, inner) {
|
|
2419
|
+
return {
|
|
2420
|
+
transport,
|
|
2421
|
+
defaultModel,
|
|
2422
|
+
maximumAttempts: inner.maximumAttempts,
|
|
2423
|
+
chat: (req, callOpts) => {
|
|
2424
|
+
const request = {
|
|
2425
|
+
model: resolveModel(req, defaultModel).model,
|
|
2426
|
+
messages: req.messages,
|
|
2427
|
+
jsonMode: req.jsonMode,
|
|
2428
|
+
jsonSchema: req.jsonSchema,
|
|
2429
|
+
temperature: req.temperature,
|
|
2430
|
+
maxTokens: req.maxTokens,
|
|
2431
|
+
thinking: req.thinking,
|
|
2432
|
+
timeoutMs: req.timeoutMs
|
|
2433
|
+
};
|
|
2434
|
+
return inner.call(request, {
|
|
2435
|
+
signal: callOpts?.signal,
|
|
2436
|
+
idempotencyKey: callOpts?.idempotencyKey
|
|
2437
|
+
});
|
|
2438
|
+
}
|
|
2439
|
+
};
|
|
2440
|
+
}
|
|
2441
|
+
function resolveModel(req, defaultModel) {
|
|
2442
|
+
if (req.model) return req;
|
|
2443
|
+
if (!defaultModel) throw new Error("ChatClient.chat: no model on request and no defaultModel on the client. Either pass req.model or bind defaultModel at createChatClient().");
|
|
2444
|
+
return {
|
|
2445
|
+
...req,
|
|
2446
|
+
model: defaultModel
|
|
2447
|
+
};
|
|
2448
|
+
}
|
|
2449
|
+
//#endregion
|
|
2450
|
+
export { validateAnalystReviewDecisions as S, analystRunDigest as _, assertExactRegistryRunOpts as a, readAnalystReview as b, KNOWLEDGE_GAP_KIND_SPEC as c, CONTROL_INTEGRITY_ANALYST as d, ControlIntegrityAnalyst as f, analystFindingDigest as g, deriveEfficiencyFindings as h, ExactAnalystRunExecutionError as i, IMPROVEMENT_KIND_SPEC as l, behavioralAnalyst as m, buildDefaultAnalystRegistry as n, DEFAULT_TRACE_ANALYST_KINDS as o, emitControlIntegrityFindings as p, AnalystRegistry as r, KNOWLEDGE_POISONING_KIND_SPEC as s, createChatClient as t, FAILURE_MODE_KIND_SPEC as u, assertUniqueFindingIds as v, snapshotAnalystRun as x, completedAnalystReviewQuality as y };
|
|
2451
2451
|
|
|
2452
|
-
//# sourceMappingURL=
|
|
2452
|
+
//# sourceMappingURL=chat-client-Bp4Ebuuc.js.map
|