@tangle-network/agent-eval 0.163.2 → 0.171.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +127 -0
- package/README.md +2 -0
- package/dist/adapters/http.d.ts +108 -0
- package/dist/adapters/http.d.ts.map +1 -0
- package/dist/adapters/http.js +208 -0
- package/dist/adapters/http.js.map +1 -0
- package/dist/analyst/index.d.ts +40 -70
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +18 -311
- package/dist/analyst/index.js.map +1 -1
- package/dist/{backend-integrity-DxuQCu_A.d.ts → backend-integrity-e79K3UPD.d.ts} +3 -3
- package/dist/{backend-integrity-DxuQCu_A.d.ts.map → backend-integrity-e79K3UPD.d.ts.map} +1 -1
- package/dist/{benchmark-BhT16ep9.js → benchmark-C4wk_Sjr.js} +10 -3
- package/dist/benchmark-C4wk_Sjr.js.map +1 -0
- package/dist/{benchmark-command-CF-4GEWZ.js → benchmark-command-D8k3Gf0J.js} +236 -251
- package/dist/benchmark-command-D8k3Gf0J.js.map +1 -0
- package/dist/{benchmark-CGPp-kDC.d.ts → benchmark-h-h4bfqj.d.ts} +3 -3
- package/dist/{benchmark-CGPp-kDC.d.ts.map → benchmark-h-h4bfqj.d.ts.map} +1 -1
- package/dist/benchmarks/index.d.ts +5 -5
- package/dist/benchmarks/index.js +3 -3
- package/dist/builder-eval/index.d.ts +3 -3
- package/dist/builder-eval/index.js +1 -1
- package/dist/campaign/index.d.ts +8 -8
- package/dist/campaign/index.js +7 -7
- package/dist/{campaign-DQZmc2Dq.js → campaign-B72njjHj.js} +15 -14
- package/dist/campaign-B72njjHj.js.map +1 -0
- package/dist/{canonical-IL-Bu-14.js → canonical-DPyQ_rpt.js} +22 -2
- package/dist/{canonical-IL-Bu-14.js.map → canonical-DPyQ_rpt.js.map} +1 -1
- package/dist/{chat-client-DlMlAeYI.js → chat-client-DEtybj5i.js} +5 -5
- package/dist/{chat-client-DlMlAeYI.js.map → chat-client-DEtybj5i.js.map} +1 -1
- package/dist/cli.js +2 -2
- package/dist/{client-CX7KqIdB.js → client-BvwNkIRN.js} +2 -2
- package/dist/{client-CX7KqIdB.js.map → client-BvwNkIRN.js.map} +1 -1
- package/dist/{client-L9VVPkim.d.ts → client-CDtcZ3p9.d.ts} +4 -4
- package/dist/{client-L9VVPkim.d.ts.map → client-CDtcZ3p9.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +12 -703
- package/dist/contract/index.js +11 -11
- package/dist/{counterfactual-BaFUWK3H.d.ts → counterfactual-Bee5_BIn.d.ts} +4 -4
- package/dist/{counterfactual-BaFUWK3H.d.ts.map → counterfactual-Bee5_BIn.d.ts.map} +1 -1
- package/dist/{default-registry-G9CKMNkc.d.ts → default-registry-B0s2zU-s.d.ts} +6 -6
- package/dist/{default-registry-G9CKMNkc.d.ts.map → default-registry-B0s2zU-s.d.ts.map} +1 -1
- package/dist/{define-agent-eval-Dx1JnPEa.d.ts → define-agent-eval-Cjy2yhqP.d.ts} +7 -7
- package/dist/{define-agent-eval-Dx1JnPEa.d.ts.map → define-agent-eval-Cjy2yhqP.d.ts.map} +1 -1
- package/dist/{define-agent-eval-D08pWIJb.js → define-agent-eval-Dy8QgxAI.js} +20 -8
- package/dist/{define-agent-eval-D08pWIJb.js.map → define-agent-eval-Dy8QgxAI.js.map} +1 -1
- package/dist/{dspy-rlm-engine-DhA9qKIm.js → dspy-rlm-engine-CS3qcCEk.js} +3 -10
- package/dist/dspy-rlm-engine-CS3qcCEk.js.map +1 -0
- package/dist/{emitter-D_jYSGRd.d.ts → emitter-Bvnu0VzL.d.ts} +3 -3
- package/dist/{emitter-D_jYSGRd.d.ts.map → emitter-Bvnu0VzL.d.ts.map} +1 -1
- package/dist/{engine-Cu5qD5Fc.d.ts → engine-CAmTUk52.d.ts} +7 -7
- package/dist/{engine-Cu5qD5Fc.d.ts.map → engine-CAmTUk52.d.ts.map} +1 -1
- package/dist/{eval-campaign-BfohKmzx.js → eval-campaign-JDTeE6Pl.js} +4 -4
- package/dist/{eval-campaign-BfohKmzx.js.map → eval-campaign-JDTeE6Pl.js.map} +1 -1
- package/dist/{exact-types-qnexxJ1Z.d.ts → exact-types-BEecmnWm.d.ts} +2 -2
- package/dist/{exact-types-qnexxJ1Z.d.ts.map → exact-types-BEecmnWm.d.ts.map} +1 -1
- package/dist/experiment/index.d.ts +5 -5
- package/dist/experiment/index.js +4 -4
- package/dist/{experiment-tracker-DCO6Cz4s.d.ts → experiment-tracker-Dm8yQMqb.d.ts} +2 -2
- package/dist/{experiment-tracker-DCO6Cz4s.d.ts.map → experiment-tracker-Dm8yQMqb.d.ts.map} +1 -1
- package/dist/{external-optimizer-process-BFmh36vW.js → external-optimizer-process-CQxylYeG.js} +4 -11
- package/dist/external-optimizer-process-CQxylYeG.js.map +1 -0
- package/dist/{external-optimizer-subprocess-CqLMW3nh.js → external-optimizer-subprocess-Cex8Da2i.js} +25 -11
- package/dist/external-optimizer-subprocess-Cex8Da2i.js.map +1 -0
- package/dist/{failure-cluster-CXL8NbEw.d.ts → failure-cluster-6YSvsKlp.d.ts} +3 -3
- package/dist/{failure-cluster-CXL8NbEw.d.ts.map → failure-cluster-6YSvsKlp.d.ts.map} +1 -1
- package/dist/{feedback-trajectory-B3ZHaHV_.d.ts → feedback-trajectory-DIqpCyF0.d.ts} +6 -6
- package/dist/{feedback-trajectory-B3ZHaHV_.d.ts.map → feedback-trajectory-DIqpCyF0.d.ts.map} +1 -1
- package/dist/fuzz.d.ts +1 -1
- package/dist/{skillopt-optimization-method-x7TTF23P.d.ts → heldout-gate-Dh2b62w8.d.ts} +111 -111
- package/dist/heldout-gate-Dh2b62w8.d.ts.map +1 -0
- package/dist/hosted/index.d.ts +2 -2
- package/dist/hosted/index.js +1 -1
- package/dist/{index-D-V8gCs_.d.ts → index-8VIogTyS.d.ts} +30 -22
- package/dist/index-8VIogTyS.d.ts.map +1 -0
- package/dist/{index-D-IiQIBB.d.ts → index-DMoxLG8P.d.ts} +3 -3
- package/dist/{index-D-IiQIBB.d.ts.map → index-DMoxLG8P.d.ts.map} +1 -1
- package/dist/{index-D_P7Ye43.d.ts → index-DNgf5gyG.d.ts} +2 -2
- package/dist/{index-D_P7Ye43.d.ts.map → index-DNgf5gyG.d.ts.map} +1 -1
- package/dist/{index-CGtH1piv.d.ts → index-DT73JraI.d.ts} +7 -38
- package/dist/index-DT73JraI.d.ts.map +1 -0
- package/dist/index-fNXZMCzX.d.ts +704 -0
- package/dist/index-fNXZMCzX.d.ts.map +1 -0
- package/dist/index.d.ts +197 -38
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +736 -28
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-DRe8LB6d.d.ts → insight-report-08F022xN.d.ts} +4 -4
- package/dist/{insight-report-DRe8LB6d.d.ts.map → insight-report-08F022xN.d.ts.map} +1 -1
- package/dist/{integrity-DUNX9Fao.d.ts → integrity-B_EDELom.d.ts} +2 -2
- package/dist/{integrity-DUNX9Fao.d.ts.map → integrity-B_EDELom.d.ts.map} +1 -1
- package/dist/internal-BMFSR8Ns.js.map +1 -1
- package/dist/{kind-factory-DY8FdoXf.js → kind-factory-DMeEoMQZ.js} +3 -10
- package/dist/kind-factory-DMeEoMQZ.js.map +1 -0
- package/dist/ledger-core/index.js +2 -2
- package/dist/{ledger-core-BOzlRygb.js → ledger-core-PIfjCbKn.js} +2 -2
- package/dist/{ledger-core-BOzlRygb.js.map → ledger-core-PIfjCbKn.js.map} +1 -1
- package/dist/{llm-judge-Du7WQPh7.js → llm-judge-BtJ2Sfk_.js} +101 -20
- package/dist/llm-judge-BtJ2Sfk_.js.map +1 -0
- package/dist/matrix/index.d.ts +2 -2
- package/dist/{matrix-eXKRMHnL.d.ts → matrix-DiHmUobV.d.ts} +3 -3
- package/dist/{matrix-eXKRMHnL.d.ts.map → matrix-DiHmUobV.d.ts.map} +1 -1
- package/dist/meta-eval/index.d.ts +3 -3
- package/dist/meta-eval/index.js +1 -1
- package/dist/{mint-DfODW1KW.js → mint-DjfDUMHr.js} +2 -2
- package/dist/{mint-DfODW1KW.js.map → mint-DjfDUMHr.js.map} +1 -1
- package/dist/multishot/golden/index.d.ts +1 -1
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +4 -4
- package/dist/pipelines/index.d.ts +5 -5
- package/dist/pipelines/index.js +3 -3
- package/dist/{pareto-BqNW3LJR.d.ts → power-preflight-Ptse_Kq7.d.ts} +43 -43
- package/dist/power-preflight-Ptse_Kq7.d.ts.map +1 -0
- package/dist/{pre-registration-KN9jkh58.js → pre-registration-D94b7Of5.js} +2 -2
- package/dist/{pre-registration-KN9jkh58.js.map → pre-registration-D94b7Of5.js.map} +1 -1
- package/dist/{pre-registration-CzFCcwYk.d.ts → pre-registration-DHz6P_6f.d.ts} +2 -2
- package/dist/{pre-registration-CzFCcwYk.d.ts.map → pre-registration-DHz6P_6f.d.ts.map} +1 -1
- package/dist/{produced-state-Be0BK3RN.js → produced-state-Cm6DU_Ao.js} +4 -4
- package/dist/{produced-state-Be0BK3RN.js.map → produced-state-Cm6DU_Ao.js.map} +1 -1
- package/dist/profile-cell.js +1 -1
- package/dist/{promotion-policy-DtnOIZvk.d.ts → promotion-policy-WSXtBgBb.d.ts} +3 -3
- package/dist/{promotion-policy-DtnOIZvk.d.ts.map → promotion-policy-WSXtBgBb.d.ts.map} +1 -1
- package/dist/{run-score-lDzV0X8j.js → proposal-findings-bko3GGy-.js} +2 -31
- package/dist/proposal-findings-bko3GGy-.js.map +1 -0
- package/dist/{transient-failure-DKF5Mofa.d.ts → provenance-CafMdZKM.d.ts} +895 -891
- package/dist/provenance-CafMdZKM.d.ts.map +1 -0
- package/dist/{query-_5g6re3_.js → query-BPGMVlbM.js} +3 -3
- package/dist/{query-_5g6re3_.js.map → query-BPGMVlbM.js.map} +1 -1
- package/dist/{query-CwnHlu5p.d.ts → query-Na5gEIGd.d.ts} +3 -3
- package/dist/{query-CwnHlu5p.d.ts.map → query-Na5gEIGd.d.ts.map} +1 -1
- package/dist/{registry-8You7OK1.d.ts → registry-xEb_xfns.d.ts} +3 -3
- package/dist/{registry-8You7OK1.d.ts.map → registry-xEb_xfns.d.ts.map} +1 -1
- package/dist/{release-confidence-nGDJiiwc.js → release-confidence-CzUHc4z4.js} +3 -3
- package/dist/{release-confidence-nGDJiiwc.js.map → release-confidence-CzUHc4z4.js.map} +1 -1
- package/dist/{release-confidence-Dqt0NFep.d.ts → release-confidence-D6lQw_o7.d.ts} +4 -4
- package/dist/{release-confidence-Dqt0NFep.d.ts.map → release-confidence-D6lQw_o7.d.ts.map} +1 -1
- package/dist/reporting.d.ts +3 -3
- package/dist/reporting.js +2 -2
- package/dist/{researcher-Cz565b7D.d.ts → researcher-CMUTQXD7.d.ts} +6 -6
- package/dist/{researcher-Cz565b7D.d.ts.map → researcher-CMUTQXD7.d.ts.map} +1 -1
- package/dist/{reward-hacking-MBf7qpSB.d.ts → reward-hacking-CgPRUesA.d.ts} +2 -2
- package/dist/{reward-hacking-MBf7qpSB.d.ts.map → reward-hacking-CgPRUesA.d.ts.map} +1 -1
- package/dist/{reward-hacking-O5zKDANP.js → reward-hacking-SkxYgT0x.js} +2 -2
- package/dist/{reward-hacking-O5zKDANP.js.map → reward-hacking-SkxYgT0x.js.map} +1 -1
- package/dist/rl.d.ts +8 -8
- package/dist/rl.d.ts.map +1 -1
- package/dist/rl.js +6 -5
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +1 -1
- package/dist/rollout/index.js +2 -2
- package/dist/{rollout-Dm2tSdiQ.js → rollout-Crypdx8s.js} +2 -2
- package/dist/{rollout-Dm2tSdiQ.js.map → rollout-Crypdx8s.js.map} +1 -1
- package/dist/{rubric-predictive-validity-CxycqzX5.d.ts → rubric-predictive-validity-DluJLCKQ.d.ts} +2 -2
- package/dist/{rubric-predictive-validity-CxycqzX5.d.ts.map → rubric-predictive-validity-DluJLCKQ.d.ts.map} +1 -1
- package/dist/{run-record-BC0ebuRP.js → run-record-DLORoL7t.js} +2 -2
- package/dist/{run-record-BC0ebuRP.js.map → run-record-DLORoL7t.js.map} +1 -1
- package/dist/{run-record-VVy4T9OW.d.ts → run-record-DQjRcYwA.d.ts} +3 -3
- package/dist/{run-record-VVy4T9OW.d.ts.map → run-record-DQjRcYwA.d.ts.map} +1 -1
- package/dist/{schema-k6ZBftVv.js → schema-CdIX2aHu.js} +5 -1
- package/dist/{schema-k6ZBftVv.js.map → schema-CdIX2aHu.js.map} +1 -1
- package/dist/{schema-Bjgdsn73.d.ts → schema-DID1Cqct.d.ts} +7 -3
- package/dist/{schema-Bjgdsn73.d.ts.map → schema-DID1Cqct.d.ts.map} +1 -1
- package/dist/{semantic-concept-judge-BsDMOwJr.js → semantic-concept-judge-I36eejJx.js} +2 -2
- package/dist/{semantic-concept-judge-BsDMOwJr.js.map → semantic-concept-judge-I36eejJx.js.map} +1 -1
- package/dist/{sequential-BLMbdrD7.js → sequential-B51qAYE4.js} +2 -2
- package/dist/{sequential-BLMbdrD7.js.map → sequential-B51qAYE4.js.map} +1 -1
- package/dist/{server-BjYiJHoJ.js → server-CCEnywOR.js} +26 -18
- package/dist/server-CCEnywOR.js.map +1 -0
- package/dist/{skillopt-optimization-method-UArRo-nr.js → skillopt-optimization-method-DzlF2RM7.js} +9 -9
- package/dist/{skillopt-optimization-method-UArRo-nr.js.map → skillopt-optimization-method-DzlF2RM7.js.map} +1 -1
- package/dist/{statistical-heldout-Cy3EhjlC.d.ts → statistical-heldout-UhiexnjU.d.ts} +3 -3
- package/dist/{statistical-heldout-Cy3EhjlC.d.ts.map → statistical-heldout-UhiexnjU.d.ts.map} +1 -1
- package/dist/{store-B06JdC56.d.ts → store-Cq9oOrI1.d.ts} +2 -2
- package/dist/{store-B06JdC56.d.ts.map → store-Cq9oOrI1.d.ts.map} +1 -1
- package/dist/{store-otlp-C_Rq5I4D.js → store-otlp-CHjBvWQY.js} +2 -2
- package/dist/{store-otlp-C_Rq5I4D.js.map → store-otlp-CHjBvWQY.js.map} +1 -1
- package/dist/{store-tool-spans-BVga3c37.js → store-tool-spans-B9o6tU8f.js} +3 -3
- package/dist/{store-tool-spans-BVga3c37.js.map → store-tool-spans-B9o6tU8f.js.map} +1 -1
- package/dist/{store-tool-spans-DPUG7UUY.d.ts → store-tool-spans-D_qMl2__.d.ts} +102 -36
- package/dist/store-tool-spans-D_qMl2__.d.ts.map +1 -0
- package/dist/storyboard/index.d.ts +1 -1
- package/dist/{summary-report-BXeQ5Ues.js → summary-report-Bgh8CpNK.js} +2 -2
- package/dist/{summary-report-BXeQ5Ues.js.map → summary-report-Bgh8CpNK.js.map} +1 -1
- package/dist/{summary-report-CC07PhEL.d.ts → summary-report-DRstQNBX.d.ts} +3 -3
- package/dist/{summary-report-CC07PhEL.d.ts.map → summary-report-DRstQNBX.d.ts.map} +1 -1
- package/dist/{task-failure-attributes-DTl-7-Kw.js → task-failure-attributes-CBGtLS_H.js} +3 -3
- package/dist/{task-failure-attributes-DTl-7-Kw.js.map → task-failure-attributes-CBGtLS_H.js.map} +1 -1
- package/dist/{tool-groups-Ci8i9ErB.d.ts → tool-groups-RGYfVWpc.d.ts} +3 -3
- package/dist/tool-groups-RGYfVWpc.d.ts.map +1 -0
- package/dist/{tool-waste-CKc7bYIg.d.ts → tool-waste-BrmLKxMw.d.ts} +4 -4
- package/dist/{tool-waste-CKc7bYIg.d.ts.map → tool-waste-BrmLKxMw.d.ts.map} +1 -1
- package/dist/{tool-waste-8BQiUc8K.js → tool-waste-CwGHzBzX.js} +2 -2
- package/dist/{tool-waste-8BQiUc8K.js.map → tool-waste-CwGHzBzX.js.map} +1 -1
- package/dist/trace-repair/index.d.ts +3 -3
- package/dist/trace-repair/index.d.ts.map +1 -1
- package/dist/trace-repair/index.js +4 -3
- package/dist/trace-repair/index.js.map +1 -1
- package/dist/traces.d.ts +59 -61
- package/dist/traces.d.ts.map +1 -1
- package/dist/traces.js +7 -7
- package/dist/{trajectory-Bi157Gun.d.ts → trajectory-r1bQqvBQ.d.ts} +3 -3
- package/dist/{trajectory-Bi157Gun.d.ts.map → trajectory-r1bQqvBQ.d.ts.map} +1 -1
- package/dist/trajectory-replay/index.d.ts +3 -3
- package/dist/trajectory-replay/index.js +1 -1
- package/dist/{types-BPb2Kf_C.d.ts → types-CMyW4GnH.d.ts} +3 -3
- package/dist/{types-BPb2Kf_C.d.ts.map → types-CMyW4GnH.d.ts.map} +1 -1
- package/dist/{types-BI4fT3HN.js → types-CiWITkGo.js} +11 -2
- package/dist/types-CiWITkGo.js.map +1 -0
- package/dist/{types-D9ssmxKL.d.ts → types-DMoNFDWi.d.ts} +6 -3
- package/dist/{types-D9ssmxKL.d.ts.map → types-DMoNFDWi.d.ts.map} +1 -1
- package/dist/{types-D4s7Z6nq.d.ts → types-JHMOqZI4.d.ts} +13 -3
- package/dist/types-JHMOqZI4.d.ts.map +1 -0
- package/dist/{verdict-B0xltqu6.js → verdict-BQ3pCFf8.js} +2 -2
- package/dist/{verdict-B0xltqu6.js.map → verdict-BQ3pCFf8.js.map} +1 -1
- package/dist/{verdict-cache-CdVVTVmn.js → verdict-cache-B3eCVQtY.js} +2 -2
- package/dist/{verdict-cache-CdVVTVmn.js.map → verdict-cache-B3eCVQtY.js.map} +1 -1
- package/dist/wire/index.d.ts +26 -11
- package/dist/wire/index.d.ts.map +1 -1
- package/dist/wire/index.js +2 -2
- package/docs/campaign-proposers.md +4 -0
- package/docs/code-agent-intake.md +64 -0
- package/docs/concepts.md +1 -1
- package/docs/design/statistics-decisions.md +1 -1
- package/docs/distributed-driver.md +3 -6
- package/docs/public-api.md +122 -106
- package/docs/wire-protocol.md +5 -3
- package/package.json +27 -19
- package/dist/benchmark-BhT16ep9.js.map +0 -1
- package/dist/benchmark-command-CF-4GEWZ.js.map +0 -1
- package/dist/campaign-DQZmc2Dq.js.map +0 -1
- package/dist/capture-fetch-CqwsJkkG.d.ts +0 -68
- package/dist/capture-fetch-CqwsJkkG.d.ts.map +0 -1
- package/dist/contract/index.d.ts.map +0 -1
- package/dist/dspy-rlm-engine-DhA9qKIm.js.map +0 -1
- package/dist/external-optimizer-process-BFmh36vW.js.map +0 -1
- package/dist/external-optimizer-subprocess-CqLMW3nh.js.map +0 -1
- package/dist/index-CGtH1piv.d.ts.map +0 -1
- package/dist/index-D-V8gCs_.d.ts.map +0 -1
- package/dist/index-vrJugRal.d.ts +0 -1
- package/dist/kind-factory-DY8FdoXf.js.map +0 -1
- package/dist/llm-judge-Du7WQPh7.js.map +0 -1
- package/dist/pareto-BqNW3LJR.d.ts.map +0 -1
- package/dist/run-score-lDzV0X8j.js.map +0 -1
- package/dist/server-BjYiJHoJ.js.map +0 -1
- package/dist/skillopt-optimization-method-x7TTF23P.d.ts.map +0 -1
- package/dist/store-tool-spans-DPUG7UUY.d.ts.map +0 -1
- package/dist/tool-groups-Ci8i9ErB.d.ts.map +0 -1
- package/dist/transient-failure-DKF5Mofa.d.ts.map +0 -1
- package/dist/types-BI4fT3HN.js.map +0 -1
- package/dist/types-D4s7Z6nq.d.ts.map +0 -1
|
@@ -1,17 +1,17 @@
|
|
|
1
1
|
import { s as ValidationError, t as AgentEvalError } from "./errors-Dngq5h35.js";
|
|
2
|
-
import { a as
|
|
2
|
+
import { a as hashCanonical, i as compareCodeUnits, o as jsonDocument, r as canonicalString } from "./canonical-DPyQ_rpt.js";
|
|
3
3
|
import { r as pairedBootstrap } from "./paired-tests-C8iCsioC.js";
|
|
4
4
|
import { a as resolveModelPricing } from "./metrics-Qv-cpptD.js";
|
|
5
5
|
import { a as CostLedgerPersistenceError, i as CostLedger, n as CostCallConflictError, o as CostReceiptCaptureError, r as CostCeilingReachedError, s as CostReservationExceededError, t as CostAccountingIncompleteError } from "./cost-ledger-B1qx30B4.js";
|
|
6
|
-
import { c as writeLedgerFileAtomically, s as withLedgerFileLock } from "./ledger-core-
|
|
7
|
-
import {
|
|
8
|
-
import {
|
|
9
|
-
import { C as RAW_FINDING_SCHEMA_PROMPT, T as evidenceRefsFromRawFinding, m as TRACE_ANALYSIS_LIMITS, r as runTraceAnalyst, w as RawAnalystFindingSchema } from "./kind-factory-
|
|
6
|
+
import { c as writeLedgerFileAtomically, s as withLedgerFileLock } from "./ledger-core-PIfjCbKn.js";
|
|
7
|
+
import { S as fsCampaignStorage, i as startExternalOptimizerModelProxy, r as runWithCleanup, x as createRunCostLedger, y as resolveExternalOptimizerProcessLimits } from "./external-optimizer-subprocess-Cex8Da2i.js";
|
|
8
|
+
import { r as makeFinding, s as usageReceiptFromCostLedger } from "./types-CiWITkGo.js";
|
|
9
|
+
import { C as RAW_FINDING_SCHEMA_PROMPT, T as evidenceRefsFromRawFinding, m as TRACE_ANALYSIS_LIMITS, r as runTraceAnalyst, w as RawAnalystFindingSchema } from "./kind-factory-DMeEoMQZ.js";
|
|
10
10
|
import { a as callLlmJson, r as LlmResponseError, t as LlmCallError } from "./llm-client-BFMRpmqb.js";
|
|
11
|
-
import { t as createDspyRlmTraceEngine } from "./dspy-rlm-engine-
|
|
12
|
-
import { i as otlpTextToTraceAnalysisStore, r as createOtlpBufferTraceStore, t as DEFAULT_MAX_TRACE_FILE_BYTES } from "./store-otlp-
|
|
13
|
-
import { i as summarizeAnalystBenchmarkRunner, n as runAnalystBenchmark, r as traceStoreEvidenceResolver } from "./benchmark-
|
|
14
|
-
import { n as acquireSingleRunLock } from "./external-optimizer-process-
|
|
11
|
+
import { t as createDspyRlmTraceEngine } from "./dspy-rlm-engine-CS3qcCEk.js";
|
|
12
|
+
import { i as otlpTextToTraceAnalysisStore, r as createOtlpBufferTraceStore, t as DEFAULT_MAX_TRACE_FILE_BYTES } from "./store-otlp-CHjBvWQY.js";
|
|
13
|
+
import { i as summarizeAnalystBenchmarkRunner, n as runAnalystBenchmark, r as traceStoreEvidenceResolver } from "./benchmark-C4wk_Sjr.js";
|
|
14
|
+
import { n as acquireSingleRunLock } from "./external-optimizer-process-CQxylYeG.js";
|
|
15
15
|
import { c as primeProtocolSha256, d as runPrimeExchange, f as decodeReplyRows, n as buildPrimePrompt, p as assertEqualDeclarativeTerms, t as analystUsageReceiptFromPrimeUsage, u as projectPrimeTrajectory } from "./prime-protocol-6tZTVsWm.js";
|
|
16
16
|
import { createHash, randomUUID } from "node:crypto";
|
|
17
17
|
import { constants, existsSync, lstatSync, readFileSync } from "node:fs";
|
|
@@ -625,6 +625,226 @@ function escapeCell$2(value) {
|
|
|
625
625
|
return value.replaceAll("|", "\\|").replaceAll("\n", " ");
|
|
626
626
|
}
|
|
627
627
|
//#endregion
|
|
628
|
+
//#region src/analyst/benchmark-comparison.ts
|
|
629
|
+
/**
|
|
630
|
+
* Every metric a benchmark comparison reports, mapped to the direction that
|
|
631
|
+
* is an improvement. This table is the only declaration of the vocabulary:
|
|
632
|
+
* the type, the reporting order, the artifact schema's accepted values, and
|
|
633
|
+
* each metric's direction all derive from it, so a metric cannot exist in one
|
|
634
|
+
* of those four places and be missing from another.
|
|
635
|
+
*/
|
|
636
|
+
const ANALYST_COMPARISON_METRIC_DIRECTION = {
|
|
637
|
+
completion: "higher",
|
|
638
|
+
issueRecall: "higher",
|
|
639
|
+
findingPrecision: "higher",
|
|
640
|
+
f1: "higher",
|
|
641
|
+
criticalStepAccuracy: "higher",
|
|
642
|
+
citationCoverage: "higher",
|
|
643
|
+
citationExcerptCoverage: "higher",
|
|
644
|
+
citationLabelAgreement: "higher",
|
|
645
|
+
citationResolution: "higher",
|
|
646
|
+
trustedNegativeAccuracy: "higher",
|
|
647
|
+
latencyMs: "lower",
|
|
648
|
+
calls: "lower",
|
|
649
|
+
inputTokens: "lower",
|
|
650
|
+
outputTokens: "lower",
|
|
651
|
+
reasoningTokens: "lower",
|
|
652
|
+
cachedTokens: "lower",
|
|
653
|
+
cacheWriteTokens: "lower",
|
|
654
|
+
costUsd: "lower"
|
|
655
|
+
};
|
|
656
|
+
/** The vocabulary as a non-empty tuple, which is what `z.enum` accepts. Key
|
|
657
|
+
* order is the declaration order above, and it is the reporting order. */
|
|
658
|
+
const ANALYST_COMPARISON_METRICS = Object.keys(ANALYST_COMPARISON_METRIC_DIRECTION);
|
|
659
|
+
/** `'lower'` when a smaller value is the improvement. */
|
|
660
|
+
function analystComparisonMetricDirection(metric) {
|
|
661
|
+
return ANALYST_COMPARISON_METRIC_DIRECTION[metric];
|
|
662
|
+
}
|
|
663
|
+
function compareAnalystRunners(result, options) {
|
|
664
|
+
const confidence = options.confidence ?? .95;
|
|
665
|
+
const resamples = options.resamples ?? 2e3;
|
|
666
|
+
assertComparisonControls(confidence, resamples);
|
|
667
|
+
const runnerIds = new Set(result.summaries.map((summary) => summary.runnerId));
|
|
668
|
+
if (!runnerIds.has(options.baselineRunnerId)) throw new TypeError(`unknown baseline analyst runner '${options.baselineRunnerId}'`);
|
|
669
|
+
if (!runnerIds.has(options.candidateRunnerId)) throw new TypeError(`unknown candidate analyst runner '${options.candidateRunnerId}'`);
|
|
670
|
+
if (options.baselineRunnerId === options.candidateRunnerId) throw new TypeError("baseline and candidate analyst runners must be different");
|
|
671
|
+
const baseline = observationsByCase(result.observations, options.baselineRunnerId);
|
|
672
|
+
const candidate = observationsByCase(result.observations, options.candidateRunnerId);
|
|
673
|
+
const populationRepresentativenessProven = result.provenance.metadata?.populationRepresentativenessProven === true;
|
|
674
|
+
const metrics = ANALYST_COMPARISON_METRICS.map((metric) => compareMetric({
|
|
675
|
+
metric,
|
|
676
|
+
baseline,
|
|
677
|
+
candidate,
|
|
678
|
+
confidence,
|
|
679
|
+
resamples,
|
|
680
|
+
seed: options.seed,
|
|
681
|
+
populationRepresentativenessProven
|
|
682
|
+
}));
|
|
683
|
+
return {
|
|
684
|
+
baselineRunnerId: options.baselineRunnerId,
|
|
685
|
+
candidateRunnerId: options.candidateRunnerId,
|
|
686
|
+
metrics
|
|
687
|
+
};
|
|
688
|
+
}
|
|
689
|
+
function compareMetric(options) {
|
|
690
|
+
const pairedCases = [];
|
|
691
|
+
let eligibleObservations = 0;
|
|
692
|
+
let pairedObservations = 0;
|
|
693
|
+
let baselineMissingObservations = 0;
|
|
694
|
+
let candidateMissingObservations = 0;
|
|
695
|
+
let asymmetricMissingObservations = 0;
|
|
696
|
+
const caseIds = /* @__PURE__ */ new Set([...options.baseline.keys(), ...options.candidate.keys()]);
|
|
697
|
+
for (const caseId of caseIds) {
|
|
698
|
+
const baselineByRepetition = new Map((options.baseline.get(caseId) ?? []).map((observation) => [observation.repetition, observation]));
|
|
699
|
+
const candidateByRepetition = new Map((options.candidate.get(caseId) ?? []).map((observation) => [observation.repetition, observation]));
|
|
700
|
+
const caseBefore = [];
|
|
701
|
+
const caseAfter = [];
|
|
702
|
+
let clusterId;
|
|
703
|
+
const repetitions = /* @__PURE__ */ new Set([...baselineByRepetition.keys(), ...candidateByRepetition.keys()]);
|
|
704
|
+
for (const repetition of repetitions) {
|
|
705
|
+
const baselineObservation = baselineByRepetition.get(repetition);
|
|
706
|
+
const candidateObservation = candidateByRepetition.get(repetition);
|
|
707
|
+
const identity = baselineObservation ?? candidateObservation;
|
|
708
|
+
if (!identity || !metricApplies(identity, options.metric)) continue;
|
|
709
|
+
if (baselineObservation && candidateObservation) assertSameCaseIdentity(baselineObservation, candidateObservation);
|
|
710
|
+
eligibleObservations += 1;
|
|
711
|
+
clusterId = identity.clusterId;
|
|
712
|
+
const baselineValue = baselineObservation ? metricValue(baselineObservation, options.metric) : null;
|
|
713
|
+
const candidateValue = candidateObservation ? metricValue(candidateObservation, options.metric) : null;
|
|
714
|
+
const baselineMissing = baselineValue === null;
|
|
715
|
+
const candidateMissing = candidateValue === null;
|
|
716
|
+
if (baselineMissing) baselineMissingObservations += 1;
|
|
717
|
+
if (candidateMissing) candidateMissingObservations += 1;
|
|
718
|
+
if (baselineMissing !== candidateMissing) asymmetricMissingObservations += 1;
|
|
719
|
+
if (baselineMissing || candidateMissing) continue;
|
|
720
|
+
caseBefore.push(baselineValue);
|
|
721
|
+
caseAfter.push(candidateValue);
|
|
722
|
+
pairedObservations += 1;
|
|
723
|
+
}
|
|
724
|
+
if (caseBefore.length === 0 || !clusterId) continue;
|
|
725
|
+
pairedCases.push({
|
|
726
|
+
clusterId,
|
|
727
|
+
baseline: mean$1(caseBefore),
|
|
728
|
+
candidate: mean$1(caseAfter)
|
|
729
|
+
});
|
|
730
|
+
}
|
|
731
|
+
const byCluster = /* @__PURE__ */ new Map();
|
|
732
|
+
for (const pairedCase of pairedCases) {
|
|
733
|
+
const rows = byCluster.get(pairedCase.clusterId) ?? [];
|
|
734
|
+
rows.push(pairedCase);
|
|
735
|
+
byCluster.set(pairedCase.clusterId, rows);
|
|
736
|
+
}
|
|
737
|
+
const before = [...byCluster.values()].map((rows) => mean$1(rows.map((row) => row.baseline)));
|
|
738
|
+
const after = [...byCluster.values()].map((rows) => mean$1(rows.map((row) => row.candidate)));
|
|
739
|
+
const interval = before.length === 0 ? null : pairedBootstrap(before, after, {
|
|
740
|
+
confidence: options.confidence,
|
|
741
|
+
resamples: options.resamples,
|
|
742
|
+
statistic: "mean",
|
|
743
|
+
seed: options.seed
|
|
744
|
+
});
|
|
745
|
+
const survivorOnly = pairedObservations < eligibleObservations;
|
|
746
|
+
const limitations = [];
|
|
747
|
+
if (!interval?.gateEligible) limitations.push("fewer-than-20-independent-clusters");
|
|
748
|
+
if (!options.populationRepresentativenessProven) limitations.push("population-representativeness-not-proven");
|
|
749
|
+
if (survivorOnly) limitations.push("missing-observations");
|
|
750
|
+
const comparison = {
|
|
751
|
+
metric: options.metric,
|
|
752
|
+
direction: analystComparisonMetricDirection(options.metric),
|
|
753
|
+
pairedCases: pairedCases.length,
|
|
754
|
+
pairedClusters: before.length,
|
|
755
|
+
eligibleObservations,
|
|
756
|
+
pairedObservations,
|
|
757
|
+
baselineMissingObservations,
|
|
758
|
+
candidateMissingObservations,
|
|
759
|
+
asymmetricMissingObservations,
|
|
760
|
+
survivorOnly,
|
|
761
|
+
baselineMean: before.length === 0 ? null : mean$1(before),
|
|
762
|
+
candidateMean: after.length === 0 ? null : mean$1(after),
|
|
763
|
+
meanDelta: interval?.mean ?? null,
|
|
764
|
+
intervalLow: interval?.low ?? null,
|
|
765
|
+
intervalHigh: interval?.high ?? null,
|
|
766
|
+
confidence: options.confidence,
|
|
767
|
+
resamples: options.resamples,
|
|
768
|
+
minimumSampleMet: interval?.gateEligible ?? false,
|
|
769
|
+
populationInferenceEligible: limitations.length === 0,
|
|
770
|
+
inferenceLimitations: limitations
|
|
771
|
+
};
|
|
772
|
+
assertValidComparison(comparison);
|
|
773
|
+
return comparison;
|
|
774
|
+
}
|
|
775
|
+
function observationsByCase(observations, runnerId) {
|
|
776
|
+
const byCase = /* @__PURE__ */ new Map();
|
|
777
|
+
for (const observation of observations) {
|
|
778
|
+
if (observation.runnerId !== runnerId) continue;
|
|
779
|
+
const rows = byCase.get(observation.caseId) ?? [];
|
|
780
|
+
rows.push(observation);
|
|
781
|
+
byCase.set(observation.caseId, rows);
|
|
782
|
+
}
|
|
783
|
+
return byCase;
|
|
784
|
+
}
|
|
785
|
+
function assertSameCaseIdentity(baseline, candidate) {
|
|
786
|
+
if (baseline.clusterId !== candidate.clusterId || baseline.labelState !== candidate.labelState) throw new Error(`analyst comparison case identity differs for '${baseline.caseId}' repetition ${baseline.repetition}`);
|
|
787
|
+
}
|
|
788
|
+
function metricApplies(observation, metric) {
|
|
789
|
+
if (metric === "trustedNegativeAccuracy") return observation.labelState === "trusted-negative";
|
|
790
|
+
if (metric === "issueRecall" || metric === "findingPrecision" || metric === "f1") return observation.labelState === "positive";
|
|
791
|
+
if (metric === "criticalStepAccuracy") return observation.labelState === "positive" && observation.score.criticalStepAccuracy !== null;
|
|
792
|
+
return true;
|
|
793
|
+
}
|
|
794
|
+
function metricValue(observation, metric) {
|
|
795
|
+
if (metric === "completion") return observation.error ? 0 : 1;
|
|
796
|
+
if (metric === "latencyMs") return observation.latencyMs;
|
|
797
|
+
if (metric === "trustedNegativeAccuracy") {
|
|
798
|
+
if (observation.error) return 0;
|
|
799
|
+
return observation.score.predictionOnLabelEmptyCase ? 0 : 1;
|
|
800
|
+
}
|
|
801
|
+
if (observation.error && (metric === "issueRecall" || metric === "findingPrecision" || metric === "f1" || metric === "criticalStepAccuracy")) return 0;
|
|
802
|
+
if (observation.error && (metric === "citationCoverage" || metric === "citationExcerptCoverage" || metric === "citationLabelAgreement" || metric === "citationResolution")) return null;
|
|
803
|
+
if (metric === "issueRecall") return observation.score.issueRecall;
|
|
804
|
+
if (metric === "findingPrecision") return observation.score.findingPrecision;
|
|
805
|
+
if (metric === "f1") return observation.score.f1;
|
|
806
|
+
if (metric === "criticalStepAccuracy") return observation.score.criticalStepAccuracy;
|
|
807
|
+
if (metric === "citationCoverage") return observation.score.citationCoverage;
|
|
808
|
+
if (metric === "citationExcerptCoverage") return observation.score.citationExcerptCoverage;
|
|
809
|
+
if (metric === "citationLabelAgreement") return observation.score.citationLabelAgreement;
|
|
810
|
+
if (metric === "citationResolution") return observation.evidenceResolution?.validity ?? null;
|
|
811
|
+
if (metric === "calls") return observation.usage?.calls ?? null;
|
|
812
|
+
if (metric === "inputTokens") return observation.usage?.tokens?.input ?? null;
|
|
813
|
+
if (metric === "outputTokens") return observation.usage?.tokens?.output ?? null;
|
|
814
|
+
if (metric === "reasoningTokens") return observation.usage?.tokens?.reasoning ?? null;
|
|
815
|
+
if (metric === "cachedTokens") return observation.usage?.tokens?.cached ?? null;
|
|
816
|
+
if (metric === "cacheWriteTokens") return observation.usage?.tokens?.cacheWrite ?? null;
|
|
817
|
+
if (observation.usage?.cost.kind === "uncaptured") return null;
|
|
818
|
+
return observation.usage?.cost.usd ?? null;
|
|
819
|
+
}
|
|
820
|
+
function mean$1(values) {
|
|
821
|
+
return values.reduce((sum, value) => sum + value, 0) / values.length;
|
|
822
|
+
}
|
|
823
|
+
function assertComparisonControls(confidence, resamples) {
|
|
824
|
+
if (!Number.isSafeInteger(resamples) || resamples <= 0 || resamples > 1e6) throw new Error(`compareAnalystRunners: resamples must be a positive safe integer no greater than 1000000, got ${String(resamples)}`);
|
|
825
|
+
if (!Number.isFinite(confidence) || confidence <= 0 || confidence >= 1) throw new Error(`compareAnalystRunners: confidence must be a finite number in (0,1), got ${String(confidence)}`);
|
|
826
|
+
}
|
|
827
|
+
function assertValidComparison(comparison) {
|
|
828
|
+
if ([
|
|
829
|
+
"pairedCases",
|
|
830
|
+
"pairedClusters",
|
|
831
|
+
"eligibleObservations",
|
|
832
|
+
"pairedObservations",
|
|
833
|
+
"baselineMissingObservations",
|
|
834
|
+
"candidateMissingObservations",
|
|
835
|
+
"asymmetricMissingObservations",
|
|
836
|
+
"confidence",
|
|
837
|
+
"resamples"
|
|
838
|
+
].some((field) => !Number.isFinite(comparison[field])) || [
|
|
839
|
+
"baselineMean",
|
|
840
|
+
"candidateMean",
|
|
841
|
+
"meanDelta",
|
|
842
|
+
"intervalLow",
|
|
843
|
+
"intervalHigh"
|
|
844
|
+
].some((field) => comparison[field] !== null && !Number.isFinite(comparison[field]))) throw new Error(`compareAnalystRunners: ${comparison.metric} produced non-finite comparison output`);
|
|
845
|
+
if (comparison.intervalLow !== null && comparison.intervalHigh !== null && comparison.intervalLow > comparison.intervalHigh) throw new Error(`compareAnalystRunners: ${comparison.metric} produced an invalid confidence interval`);
|
|
846
|
+
}
|
|
847
|
+
//#endregion
|
|
628
848
|
//#region src/analyst/benchmark-command-validation.ts
|
|
629
849
|
const nonEmptyString = z.string().refine((value) => value.trim().length > 0, { message: "must be a non-empty string" });
|
|
630
850
|
const safeInteger$1 = z.number().refine(Number.isSafeInteger, { message: "must be a safe integer" });
|
|
@@ -863,26 +1083,7 @@ const resultSchema = z.strictObject({
|
|
|
863
1083
|
summaries: z.array(summarySchema)
|
|
864
1084
|
});
|
|
865
1085
|
const comparisonMetricSchema = z.strictObject({
|
|
866
|
-
metric: z.enum(
|
|
867
|
-
"completion",
|
|
868
|
-
"issueRecall",
|
|
869
|
-
"findingPrecision",
|
|
870
|
-
"f1",
|
|
871
|
-
"criticalStepAccuracy",
|
|
872
|
-
"citationCoverage",
|
|
873
|
-
"citationExcerptCoverage",
|
|
874
|
-
"citationLabelAgreement",
|
|
875
|
-
"citationResolution",
|
|
876
|
-
"trustedNegativeAccuracy",
|
|
877
|
-
"latencyMs",
|
|
878
|
-
"calls",
|
|
879
|
-
"inputTokens",
|
|
880
|
-
"outputTokens",
|
|
881
|
-
"reasoningTokens",
|
|
882
|
-
"cachedTokens",
|
|
883
|
-
"cacheWriteTokens",
|
|
884
|
-
"costUsd"
|
|
885
|
-
]),
|
|
1086
|
+
metric: z.enum(ANALYST_COMPARISON_METRICS),
|
|
886
1087
|
direction: z.enum(["higher", "lower"]),
|
|
887
1088
|
pairedCases: nonNegativeInteger,
|
|
888
1089
|
pairedClusters: nonNegativeInteger,
|
|
@@ -1211,7 +1412,7 @@ const ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES = Object.freeze([
|
|
|
1211
1412
|
"package.json",
|
|
1212
1413
|
"pnpm-lock.yaml"
|
|
1213
1414
|
]);
|
|
1214
|
-
const ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 = "
|
|
1415
|
+
const ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 = "43f2e94849ab4d777ac8c0c358c6c4f0687129c322c0565fb5cc8583922a433d";
|
|
1215
1416
|
const ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256 = "1e03f2daed356d60316aabefb407ec1e437ac94d408d61eea4ae096e9c6fbb5b";
|
|
1216
1417
|
const ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256 = "4dba263b6256a30d56c7fdb2d992d3a953c0035d731f359b704db806f68f75ac";
|
|
1217
1418
|
const ANALYST_BENCHMARK_IMPLEMENTATION_FILES = Object.freeze([
|
|
@@ -1324,7 +1525,7 @@ const ANALYST_BENCHMARK_IMPLEMENTATION_FILES = Object.freeze([
|
|
|
1324
1525
|
"src/trace/raw-provider-sink.ts",
|
|
1325
1526
|
"src/verdict-cache.ts"
|
|
1326
1527
|
]);
|
|
1327
|
-
const ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 = "
|
|
1528
|
+
const ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 = "917ad7bb523495505a8da1e48f9acefe25970c946f401fe4d38dbac4e1c43aad";
|
|
1328
1529
|
function analystBenchmarkImplementationDigest() {
|
|
1329
1530
|
return ANALYST_BENCHMARK_IMPLEMENTATION_SHA256;
|
|
1330
1531
|
}
|
|
@@ -2019,7 +2220,7 @@ async function artifactCandidates(caseDirectory) {
|
|
|
2019
2220
|
if (seen.has(entry.path)) return false;
|
|
2020
2221
|
seen.add(entry.path);
|
|
2021
2222
|
return true;
|
|
2022
|
-
}).sort((left, right) => artifactRoleOrder(left.role) - artifactRoleOrder(right.role) || left.relativePath
|
|
2223
|
+
}).sort((left, right) => artifactRoleOrder(left.role) - artifactRoleOrder(right.role) || compareCodeUnits(left.relativePath, right.relativePath));
|
|
2023
2224
|
}
|
|
2024
2225
|
async function firstExisting(caseDirectory, candidates) {
|
|
2025
2226
|
for (const relativePath of candidates) {
|
|
@@ -2731,222 +2932,6 @@ function isNodeError(error, code) {
|
|
|
2731
2932
|
return error instanceof Error && "code" in error && error.code === code;
|
|
2732
2933
|
}
|
|
2733
2934
|
//#endregion
|
|
2734
|
-
//#region src/analyst/benchmark-comparison.ts
|
|
2735
|
-
function compareAnalystRunners(result, options) {
|
|
2736
|
-
const confidence = options.confidence ?? .95;
|
|
2737
|
-
const resamples = options.resamples ?? 2e3;
|
|
2738
|
-
assertComparisonControls(confidence, resamples);
|
|
2739
|
-
const runnerIds = new Set(result.summaries.map((summary) => summary.runnerId));
|
|
2740
|
-
if (!runnerIds.has(options.baselineRunnerId)) throw new TypeError(`unknown baseline analyst runner '${options.baselineRunnerId}'`);
|
|
2741
|
-
if (!runnerIds.has(options.candidateRunnerId)) throw new TypeError(`unknown candidate analyst runner '${options.candidateRunnerId}'`);
|
|
2742
|
-
if (options.baselineRunnerId === options.candidateRunnerId) throw new TypeError("baseline and candidate analyst runners must be different");
|
|
2743
|
-
const baseline = observationsByCase(result.observations, options.baselineRunnerId);
|
|
2744
|
-
const candidate = observationsByCase(result.observations, options.candidateRunnerId);
|
|
2745
|
-
const populationRepresentativenessProven = result.provenance.metadata?.populationRepresentativenessProven === true;
|
|
2746
|
-
const metrics = METRICS.map((metric) => compareMetric({
|
|
2747
|
-
metric,
|
|
2748
|
-
baseline,
|
|
2749
|
-
candidate,
|
|
2750
|
-
confidence,
|
|
2751
|
-
resamples,
|
|
2752
|
-
seed: options.seed,
|
|
2753
|
-
populationRepresentativenessProven
|
|
2754
|
-
}));
|
|
2755
|
-
return {
|
|
2756
|
-
baselineRunnerId: options.baselineRunnerId,
|
|
2757
|
-
candidateRunnerId: options.candidateRunnerId,
|
|
2758
|
-
metrics
|
|
2759
|
-
};
|
|
2760
|
-
}
|
|
2761
|
-
function compareMetric(options) {
|
|
2762
|
-
const pairedCases = [];
|
|
2763
|
-
let eligibleObservations = 0;
|
|
2764
|
-
let pairedObservations = 0;
|
|
2765
|
-
let baselineMissingObservations = 0;
|
|
2766
|
-
let candidateMissingObservations = 0;
|
|
2767
|
-
let asymmetricMissingObservations = 0;
|
|
2768
|
-
const caseIds = /* @__PURE__ */ new Set([...options.baseline.keys(), ...options.candidate.keys()]);
|
|
2769
|
-
for (const caseId of caseIds) {
|
|
2770
|
-
const baselineByRepetition = new Map((options.baseline.get(caseId) ?? []).map((observation) => [observation.repetition, observation]));
|
|
2771
|
-
const candidateByRepetition = new Map((options.candidate.get(caseId) ?? []).map((observation) => [observation.repetition, observation]));
|
|
2772
|
-
const caseBefore = [];
|
|
2773
|
-
const caseAfter = [];
|
|
2774
|
-
let clusterId;
|
|
2775
|
-
const repetitions = /* @__PURE__ */ new Set([...baselineByRepetition.keys(), ...candidateByRepetition.keys()]);
|
|
2776
|
-
for (const repetition of repetitions) {
|
|
2777
|
-
const baselineObservation = baselineByRepetition.get(repetition);
|
|
2778
|
-
const candidateObservation = candidateByRepetition.get(repetition);
|
|
2779
|
-
const identity = baselineObservation ?? candidateObservation;
|
|
2780
|
-
if (!identity || !metricApplies(identity, options.metric)) continue;
|
|
2781
|
-
if (baselineObservation && candidateObservation) assertSameCaseIdentity(baselineObservation, candidateObservation);
|
|
2782
|
-
eligibleObservations += 1;
|
|
2783
|
-
clusterId = identity.clusterId;
|
|
2784
|
-
const baselineValue = baselineObservation ? metricValue(baselineObservation, options.metric) : null;
|
|
2785
|
-
const candidateValue = candidateObservation ? metricValue(candidateObservation, options.metric) : null;
|
|
2786
|
-
const baselineMissing = baselineValue === null;
|
|
2787
|
-
const candidateMissing = candidateValue === null;
|
|
2788
|
-
if (baselineMissing) baselineMissingObservations += 1;
|
|
2789
|
-
if (candidateMissing) candidateMissingObservations += 1;
|
|
2790
|
-
if (baselineMissing !== candidateMissing) asymmetricMissingObservations += 1;
|
|
2791
|
-
if (baselineMissing || candidateMissing) continue;
|
|
2792
|
-
caseBefore.push(baselineValue);
|
|
2793
|
-
caseAfter.push(candidateValue);
|
|
2794
|
-
pairedObservations += 1;
|
|
2795
|
-
}
|
|
2796
|
-
if (caseBefore.length === 0 || !clusterId) continue;
|
|
2797
|
-
pairedCases.push({
|
|
2798
|
-
clusterId,
|
|
2799
|
-
baseline: mean$1(caseBefore),
|
|
2800
|
-
candidate: mean$1(caseAfter)
|
|
2801
|
-
});
|
|
2802
|
-
}
|
|
2803
|
-
const byCluster = /* @__PURE__ */ new Map();
|
|
2804
|
-
for (const pairedCase of pairedCases) {
|
|
2805
|
-
const rows = byCluster.get(pairedCase.clusterId) ?? [];
|
|
2806
|
-
rows.push(pairedCase);
|
|
2807
|
-
byCluster.set(pairedCase.clusterId, rows);
|
|
2808
|
-
}
|
|
2809
|
-
const before = [...byCluster.values()].map((rows) => mean$1(rows.map((row) => row.baseline)));
|
|
2810
|
-
const after = [...byCluster.values()].map((rows) => mean$1(rows.map((row) => row.candidate)));
|
|
2811
|
-
const interval = before.length === 0 ? null : pairedBootstrap(before, after, {
|
|
2812
|
-
confidence: options.confidence,
|
|
2813
|
-
resamples: options.resamples,
|
|
2814
|
-
statistic: "mean",
|
|
2815
|
-
seed: options.seed
|
|
2816
|
-
});
|
|
2817
|
-
const survivorOnly = pairedObservations < eligibleObservations;
|
|
2818
|
-
const limitations = [];
|
|
2819
|
-
if (!interval?.gateEligible) limitations.push("fewer-than-20-independent-clusters");
|
|
2820
|
-
if (!options.populationRepresentativenessProven) limitations.push("population-representativeness-not-proven");
|
|
2821
|
-
if (survivorOnly) limitations.push("missing-observations");
|
|
2822
|
-
const comparison = {
|
|
2823
|
-
metric: options.metric,
|
|
2824
|
-
direction: LOWER_IS_BETTER.has(options.metric) ? "lower" : "higher",
|
|
2825
|
-
pairedCases: pairedCases.length,
|
|
2826
|
-
pairedClusters: before.length,
|
|
2827
|
-
eligibleObservations,
|
|
2828
|
-
pairedObservations,
|
|
2829
|
-
baselineMissingObservations,
|
|
2830
|
-
candidateMissingObservations,
|
|
2831
|
-
asymmetricMissingObservations,
|
|
2832
|
-
survivorOnly,
|
|
2833
|
-
baselineMean: before.length === 0 ? null : mean$1(before),
|
|
2834
|
-
candidateMean: after.length === 0 ? null : mean$1(after),
|
|
2835
|
-
meanDelta: interval?.mean ?? null,
|
|
2836
|
-
intervalLow: interval?.low ?? null,
|
|
2837
|
-
intervalHigh: interval?.high ?? null,
|
|
2838
|
-
confidence: options.confidence,
|
|
2839
|
-
resamples: options.resamples,
|
|
2840
|
-
minimumSampleMet: interval?.gateEligible ?? false,
|
|
2841
|
-
populationInferenceEligible: limitations.length === 0,
|
|
2842
|
-
inferenceLimitations: limitations
|
|
2843
|
-
};
|
|
2844
|
-
assertValidComparison(comparison);
|
|
2845
|
-
return comparison;
|
|
2846
|
-
}
|
|
2847
|
-
const METRICS = [
|
|
2848
|
-
"completion",
|
|
2849
|
-
"issueRecall",
|
|
2850
|
-
"findingPrecision",
|
|
2851
|
-
"f1",
|
|
2852
|
-
"criticalStepAccuracy",
|
|
2853
|
-
"citationCoverage",
|
|
2854
|
-
"citationExcerptCoverage",
|
|
2855
|
-
"citationLabelAgreement",
|
|
2856
|
-
"citationResolution",
|
|
2857
|
-
"trustedNegativeAccuracy",
|
|
2858
|
-
"latencyMs",
|
|
2859
|
-
"calls",
|
|
2860
|
-
"inputTokens",
|
|
2861
|
-
"outputTokens",
|
|
2862
|
-
"reasoningTokens",
|
|
2863
|
-
"cachedTokens",
|
|
2864
|
-
"cacheWriteTokens",
|
|
2865
|
-
"costUsd"
|
|
2866
|
-
];
|
|
2867
|
-
const LOWER_IS_BETTER = /* @__PURE__ */ new Set([
|
|
2868
|
-
"latencyMs",
|
|
2869
|
-
"calls",
|
|
2870
|
-
"inputTokens",
|
|
2871
|
-
"outputTokens",
|
|
2872
|
-
"reasoningTokens",
|
|
2873
|
-
"cachedTokens",
|
|
2874
|
-
"cacheWriteTokens",
|
|
2875
|
-
"costUsd"
|
|
2876
|
-
]);
|
|
2877
|
-
function observationsByCase(observations, runnerId) {
|
|
2878
|
-
const byCase = /* @__PURE__ */ new Map();
|
|
2879
|
-
for (const observation of observations) {
|
|
2880
|
-
if (observation.runnerId !== runnerId) continue;
|
|
2881
|
-
const rows = byCase.get(observation.caseId) ?? [];
|
|
2882
|
-
rows.push(observation);
|
|
2883
|
-
byCase.set(observation.caseId, rows);
|
|
2884
|
-
}
|
|
2885
|
-
return byCase;
|
|
2886
|
-
}
|
|
2887
|
-
function assertSameCaseIdentity(baseline, candidate) {
|
|
2888
|
-
if (baseline.clusterId !== candidate.clusterId || baseline.labelState !== candidate.labelState) throw new Error(`analyst comparison case identity differs for '${baseline.caseId}' repetition ${baseline.repetition}`);
|
|
2889
|
-
}
|
|
2890
|
-
function metricApplies(observation, metric) {
|
|
2891
|
-
if (metric === "trustedNegativeAccuracy") return observation.labelState === "trusted-negative";
|
|
2892
|
-
if (metric === "issueRecall" || metric === "findingPrecision" || metric === "f1") return observation.labelState === "positive";
|
|
2893
|
-
if (metric === "criticalStepAccuracy") return observation.labelState === "positive" && observation.score.criticalStepAccuracy !== null;
|
|
2894
|
-
return true;
|
|
2895
|
-
}
|
|
2896
|
-
function metricValue(observation, metric) {
|
|
2897
|
-
if (metric === "completion") return observation.error ? 0 : 1;
|
|
2898
|
-
if (metric === "latencyMs") return observation.latencyMs;
|
|
2899
|
-
if (metric === "trustedNegativeAccuracy") {
|
|
2900
|
-
if (observation.error) return 0;
|
|
2901
|
-
return observation.score.predictionOnLabelEmptyCase ? 0 : 1;
|
|
2902
|
-
}
|
|
2903
|
-
if (observation.error && (metric === "issueRecall" || metric === "findingPrecision" || metric === "f1" || metric === "criticalStepAccuracy")) return 0;
|
|
2904
|
-
if (observation.error && (metric === "citationCoverage" || metric === "citationExcerptCoverage" || metric === "citationLabelAgreement" || metric === "citationResolution")) return null;
|
|
2905
|
-
if (metric === "issueRecall") return observation.score.issueRecall;
|
|
2906
|
-
if (metric === "findingPrecision") return observation.score.findingPrecision;
|
|
2907
|
-
if (metric === "f1") return observation.score.f1;
|
|
2908
|
-
if (metric === "criticalStepAccuracy") return observation.score.criticalStepAccuracy;
|
|
2909
|
-
if (metric === "citationCoverage") return observation.score.citationCoverage;
|
|
2910
|
-
if (metric === "citationExcerptCoverage") return observation.score.citationExcerptCoverage;
|
|
2911
|
-
if (metric === "citationLabelAgreement") return observation.score.citationLabelAgreement;
|
|
2912
|
-
if (metric === "citationResolution") return observation.evidenceResolution?.validity ?? null;
|
|
2913
|
-
if (metric === "calls") return observation.usage?.calls ?? null;
|
|
2914
|
-
if (metric === "inputTokens") return observation.usage?.tokens?.input ?? null;
|
|
2915
|
-
if (metric === "outputTokens") return observation.usage?.tokens?.output ?? null;
|
|
2916
|
-
if (metric === "reasoningTokens") return observation.usage?.tokens?.reasoning ?? null;
|
|
2917
|
-
if (metric === "cachedTokens") return observation.usage?.tokens?.cached ?? null;
|
|
2918
|
-
if (metric === "cacheWriteTokens") return observation.usage?.tokens?.cacheWrite ?? null;
|
|
2919
|
-
if (observation.usage?.cost.kind === "uncaptured") return null;
|
|
2920
|
-
return observation.usage?.cost.usd ?? null;
|
|
2921
|
-
}
|
|
2922
|
-
function mean$1(values) {
|
|
2923
|
-
return values.reduce((sum, value) => sum + value, 0) / values.length;
|
|
2924
|
-
}
|
|
2925
|
-
function assertComparisonControls(confidence, resamples) {
|
|
2926
|
-
if (!Number.isSafeInteger(resamples) || resamples <= 0 || resamples > 1e6) throw new Error(`compareAnalystRunners: resamples must be a positive safe integer no greater than 1000000, got ${String(resamples)}`);
|
|
2927
|
-
if (!Number.isFinite(confidence) || confidence <= 0 || confidence >= 1) throw new Error(`compareAnalystRunners: confidence must be a finite number in (0,1), got ${String(confidence)}`);
|
|
2928
|
-
}
|
|
2929
|
-
function assertValidComparison(comparison) {
|
|
2930
|
-
if ([
|
|
2931
|
-
"pairedCases",
|
|
2932
|
-
"pairedClusters",
|
|
2933
|
-
"eligibleObservations",
|
|
2934
|
-
"pairedObservations",
|
|
2935
|
-
"baselineMissingObservations",
|
|
2936
|
-
"candidateMissingObservations",
|
|
2937
|
-
"asymmetricMissingObservations",
|
|
2938
|
-
"confidence",
|
|
2939
|
-
"resamples"
|
|
2940
|
-
].some((field) => !Number.isFinite(comparison[field])) || [
|
|
2941
|
-
"baselineMean",
|
|
2942
|
-
"candidateMean",
|
|
2943
|
-
"meanDelta",
|
|
2944
|
-
"intervalLow",
|
|
2945
|
-
"intervalHigh"
|
|
2946
|
-
].some((field) => comparison[field] !== null && !Number.isFinite(comparison[field]))) throw new Error(`compareAnalystRunners: ${comparison.metric} produced non-finite comparison output`);
|
|
2947
|
-
if (comparison.intervalLow !== null && comparison.intervalHigh !== null && comparison.intervalLow > comparison.intervalHigh) throw new Error(`compareAnalystRunners: ${comparison.metric} produced an invalid confidence interval`);
|
|
2948
|
-
}
|
|
2949
|
-
//#endregion
|
|
2950
2935
|
//#region src/analyst/benchmark-public-calibration.ts
|
|
2951
2936
|
function summarizeCodeTraceCalibration(result) {
|
|
2952
2937
|
return {
|
|
@@ -4940,7 +4925,7 @@ function selectPublicBenchmarkRows(dataset, rows, options) {
|
|
|
4940
4925
|
if (byId.has(id)) throw new Error(`public analyst benchmark dataset repeats trajectory id '${id}'`);
|
|
4941
4926
|
byId.set(id, row);
|
|
4942
4927
|
}
|
|
4943
|
-
return [...byId].sort(([left], [right]) => selectionKey(options.seed, left)
|
|
4928
|
+
return [...byId].sort(([left], [right]) => compareCodeUnits(selectionKey(options.seed, left), selectionKey(options.seed, right)) || compareCodeUnits(left, right)).slice(0, Math.min(options.limit, byId.size)).map(([, row]) => row);
|
|
4944
4929
|
}
|
|
4945
4930
|
function publicBenchmarkDistributions(dataset, rows) {
|
|
4946
4931
|
const values = {
|
|
@@ -6364,6 +6349,6 @@ function shellQuote(value) {
|
|
|
6364
6349
|
return /^[a-zA-Z0-9_./:@%+=,-]+$/.test(value) ? value : `'${value.replaceAll("'", "'\\''")}'`;
|
|
6365
6350
|
}
|
|
6366
6351
|
//#endregion
|
|
6367
|
-
export {
|
|
6352
|
+
export { ANALYST_BENCHMARK_LOCAL_RECEIPT_FILE as $, effectiveAnalystProtocolSha256 as A, loadCodeTraceVerificationArtifacts as B, adaptPublicBenchmarkFindings as C, renderCodeTraceCalibrationMarkdown as D, readAnalystBenchmarkArtifact as E, publicBenchmarkProtocolSha256 as F, ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256 as G, ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM as H, publicBenchmarkRlmInstructions as I, ANALYST_BENCHMARK_IMPLEMENTATION_FILES as J, ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256 as K, publicBenchmarkSystemPrompt as L, CODE_TRACE_BENCH_ANALYST_PROMPT as M, MAX_INCORRECT_BLOCKS as N, summarizeCodeTraceCalibration as O, MAX_INCORRECT_BLOCK_STEPS as P, ANALYST_BENCHMARK_COST_LEDGER_FILE as Q, DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES as R, analystDefinitionProtocolSha256 as S, expandCodeTraceFailureBlocks as T, ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES as U, parseVerificationOutcome as V, ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 as W, analystBenchmarkDependencyLockDigest as X, ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 as Y, analystBenchmarkImplementationDigest as Z, createPublicBenchmarkDirectRunner as _, primeCodeTraceAnalystDefinition as a, summarizeAgentRxCalibration as at, AnalystExpressivenessError as b, loadPublicBenchmarkRows as c, agentRxBenchmarkCase as ct, publicBenchmarkSelectionReport as d, roundAgentRxStep as dt, ANALYST_BENCHMARK_MANIFEST_FILE as et, selectPublicBenchmarkRows as f, normalizeBenchmarkLabel as ft, runReplVariableAnalystDefinition as g, rlmEngineLimits as h, primeAnalystProtocolSha256 as i, renderAgentRxCalibrationMarkdown as it, readAnalystInstructionsOverride as j, analystInstructionsOverrideFromText as k, preparePublicAnalystBenchmark as l, agentRxPredictionsToFindings as lt, publicRlmAnalystDefinition as m, renderAnalystBenchmarkMarkdown as n, compareAnalystRunners as nt, runInlineAnalystDefinition as o, codeTraceBenchCase as ot, createPublicBenchmarkRlmRunner as p, ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM as q, createPrimeBenchmarkRunner as r, AGENT_RX_UPSTREAM_REVISION as rt, nodeHttpPrimeBridgeTransport as s, codeTracerPredictionsToFindings as st, runAnalystBenchmarkCommand as t, ANALYST_BENCHMARK_OBSERVATIONS_FILE as tt, publicBenchmarkDistributions as u, normalizeAgentRxCategory as ut, publicDirectAnalystDefinition as v, emptyPublicBenchmarkRunner as w, analystDefinitionAsymmetries as x, runChunkedAnalystDefinition as y, appendVerificationArtifactsToOtlp as z };
|
|
6368
6353
|
|
|
6369
|
-
//# sourceMappingURL=benchmark-command-
|
|
6354
|
+
//# sourceMappingURL=benchmark-command-D8k3Gf0J.js.map
|