@tangle-network/agent-eval 0.161.1 → 0.170.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +156 -0
- package/README.md +2 -0
- package/dist/{active-curriculum-CD5TU2yW.js → active-curriculum-OjIWgrUJ.js} +2 -2
- package/dist/{active-curriculum-CD5TU2yW.js.map → active-curriculum-OjIWgrUJ.js.map} +1 -1
- package/dist/adapters/http.d.ts +108 -0
- package/dist/adapters/http.d.ts.map +1 -0
- package/dist/adapters/http.js +208 -0
- package/dist/adapters/http.js.map +1 -0
- package/dist/analyst/index.d.ts +40 -70
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +18 -311
- package/dist/analyst/index.js.map +1 -1
- package/dist/{backend-integrity-DxuQCu_A.d.ts → backend-integrity-e79K3UPD.d.ts} +3 -3
- package/dist/{backend-integrity-DxuQCu_A.d.ts.map → backend-integrity-e79K3UPD.d.ts.map} +1 -1
- package/dist/{baseline-BhPRQBVn.js → baseline-BC-eBZ7U.js} +2 -2
- package/dist/{baseline-BhPRQBVn.js.map → baseline-BC-eBZ7U.js.map} +1 -1
- package/dist/{benchmark-BhT16ep9.js → benchmark-C4wk_Sjr.js} +10 -3
- package/dist/benchmark-C4wk_Sjr.js.map +1 -0
- package/dist/{benchmark-command-BDC3Gocz.js → benchmark-command-BA7qOdWw.js} +239 -253
- package/dist/benchmark-command-BA7qOdWw.js.map +1 -0
- package/dist/{benchmark-CGPp-kDC.d.ts → benchmark-h-h4bfqj.d.ts} +3 -3
- package/dist/{benchmark-CGPp-kDC.d.ts.map → benchmark-h-h4bfqj.d.ts.map} +1 -1
- package/dist/benchmarks/index.d.ts +5 -5
- package/dist/benchmarks/index.js +3 -3
- package/dist/builder-eval/index.d.ts +3 -3
- package/dist/builder-eval/index.d.ts.map +1 -1
- package/dist/builder-eval/index.js +22 -8
- package/dist/builder-eval/index.js.map +1 -1
- package/dist/campaign/index.d.ts +8 -8
- package/dist/campaign/index.js +9 -9
- package/dist/{campaign-BSmOwskD.js → campaign-BeCbxFqs.js} +19 -18
- package/dist/campaign-BeCbxFqs.js.map +1 -0
- package/dist/{canonical-IL-Bu-14.js → canonical-DPyQ_rpt.js} +22 -2
- package/dist/{canonical-IL-Bu-14.js.map → canonical-DPyQ_rpt.js.map} +1 -1
- package/dist/{chat-client-DlMlAeYI.js → chat-client-DEtybj5i.js} +5 -5
- package/dist/{chat-client-DlMlAeYI.js.map → chat-client-DEtybj5i.js.map} +1 -1
- package/dist/{chat-json-call-6g5sJobJ.js → chat-json-call-5Jxna-aV.js} +2 -2
- package/dist/{chat-json-call-6g5sJobJ.js.map → chat-json-call-5Jxna-aV.js.map} +1 -1
- package/dist/cli.js +3 -3
- package/dist/{client-CX7KqIdB.js → client-BvwNkIRN.js} +2 -2
- package/dist/{client-CX7KqIdB.js.map → client-BvwNkIRN.js.map} +1 -1
- package/dist/{client-L9VVPkim.d.ts → client-_Fsa5c2_.d.ts} +4 -4
- package/dist/{client-L9VVPkim.d.ts.map → client-_Fsa5c2_.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +12 -703
- package/dist/contract/index.js +14 -14
- package/dist/{counterfactual-BaFUWK3H.d.ts → counterfactual-Bee5_BIn.d.ts} +4 -4
- package/dist/{counterfactual-BaFUWK3H.d.ts.map → counterfactual-Bee5_BIn.d.ts.map} +1 -1
- package/dist/{counterfactual-D_VWavVm.js → counterfactual-Bjq1mlUu.js} +2 -2
- package/dist/{counterfactual-D_VWavVm.js.map → counterfactual-Bjq1mlUu.js.map} +1 -1
- package/dist/{default-registry-G9CKMNkc.d.ts → default-registry-ovxrOP0_.d.ts} +6 -6
- package/dist/{default-registry-G9CKMNkc.d.ts.map → default-registry-ovxrOP0_.d.ts.map} +1 -1
- package/dist/{define-agent-eval-h-s-sI-v.js → define-agent-eval-Clj-8igZ.js} +27 -15
- package/dist/{define-agent-eval-h-s-sI-v.js.map → define-agent-eval-Clj-8igZ.js.map} +1 -1
- package/dist/{define-agent-eval-Dx1JnPEa.d.ts → define-agent-eval-DVJm8Xlh.d.ts} +7 -7
- package/dist/{define-agent-eval-Dx1JnPEa.d.ts.map → define-agent-eval-DVJm8Xlh.d.ts.map} +1 -1
- package/dist/{descriptive-jDOuI6mz.js → descriptive-1V17A-qa.js} +2 -2
- package/dist/{descriptive-jDOuI6mz.js.map → descriptive-1V17A-qa.js.map} +1 -1
- package/dist/{dspy-rlm-engine-DptEII26.js → dspy-rlm-engine-CS3qcCEk.js} +3 -10
- package/dist/dspy-rlm-engine-CS3qcCEk.js.map +1 -0
- package/dist/{emitter-D_jYSGRd.d.ts → emitter-Bvnu0VzL.d.ts} +3 -3
- package/dist/{emitter-D_jYSGRd.d.ts.map → emitter-Bvnu0VzL.d.ts.map} +1 -1
- package/dist/{emitter-BpYFQPj4.js → emitter-DeQHiDMm.js} +13 -6
- package/dist/emitter-DeQHiDMm.js.map +1 -0
- package/dist/{engine-Cu5qD5Fc.d.ts → engine-D12Rb6WB.d.ts} +7 -7
- package/dist/{engine-Cu5qD5Fc.d.ts.map → engine-D12Rb6WB.d.ts.map} +1 -1
- package/dist/{eval-campaign-BsXWL2-2.js → eval-campaign-JDTeE6Pl.js} +5 -5
- package/dist/{eval-campaign-BsXWL2-2.js.map → eval-campaign-JDTeE6Pl.js.map} +1 -1
- package/dist/{exact-types-qnexxJ1Z.d.ts → exact-types-BEecmnWm.d.ts} +2 -2
- package/dist/{exact-types-qnexxJ1Z.d.ts.map → exact-types-BEecmnWm.d.ts.map} +1 -1
- package/dist/experiment/index.d.ts +5 -5
- package/dist/experiment/index.js +11 -11
- package/dist/{experiment-tracker-Ym6rEQT1.js → experiment-tracker-BKEumQug.js} +2 -2
- package/dist/{experiment-tracker-Ym6rEQT1.js.map → experiment-tracker-BKEumQug.js.map} +1 -1
- package/dist/{experiment-tracker-DCO6Cz4s.d.ts → experiment-tracker-Dm8yQMqb.d.ts} +2 -2
- package/dist/{experiment-tracker-DCO6Cz4s.d.ts.map → experiment-tracker-Dm8yQMqb.d.ts.map} +1 -1
- package/dist/{external-optimizer-process-WosTBChy.js → external-optimizer-process-CQxylYeG.js} +4 -11
- package/dist/external-optimizer-process-CQxylYeG.js.map +1 -0
- package/dist/{external-optimizer-subprocess-BIWbHpgD.js → external-optimizer-subprocess-Cex8Da2i.js} +26 -12
- package/dist/external-optimizer-subprocess-Cex8Da2i.js.map +1 -0
- package/dist/{failure-cluster-CXL8NbEw.d.ts → failure-cluster-6YSvsKlp.d.ts} +3 -3
- package/dist/{failure-cluster-CXL8NbEw.d.ts.map → failure-cluster-6YSvsKlp.d.ts.map} +1 -1
- package/dist/{feedback-trajectory-B3ZHaHV_.d.ts → feedback-trajectory-DIqpCyF0.d.ts} +6 -6
- package/dist/{feedback-trajectory-B3ZHaHV_.d.ts.map → feedback-trajectory-DIqpCyF0.d.ts.map} +1 -1
- package/dist/fuzz.d.ts +2 -3
- package/dist/fuzz.d.ts.map +1 -1
- package/dist/fuzz.js +4 -10
- package/dist/fuzz.js.map +1 -1
- package/dist/{skillopt-optimization-method-x7TTF23P.d.ts → heldout-gate-Bn7_xWCv.d.ts} +111 -111
- package/dist/heldout-gate-Bn7_xWCv.d.ts.map +1 -0
- package/dist/hosted/index.d.ts +2 -2
- package/dist/hosted/index.js +1 -1
- package/dist/index-Bfs5aufo.d.ts +704 -0
- package/dist/index-Bfs5aufo.d.ts.map +1 -0
- package/dist/{index-CGtH1piv.d.ts → index-CM-SM00y.d.ts} +7 -38
- package/dist/index-CM-SM00y.d.ts.map +1 -0
- package/dist/{index-D-V8gCs_.d.ts → index-DBbivBNs.d.ts} +30 -22
- package/dist/index-DBbivBNs.d.ts.map +1 -0
- package/dist/{index-D-IiQIBB.d.ts → index-DMoxLG8P.d.ts} +3 -3
- package/dist/{index-D-IiQIBB.d.ts.map → index-DMoxLG8P.d.ts.map} +1 -1
- package/dist/{index-D_P7Ye43.d.ts → index-DNgf5gyG.d.ts} +2 -2
- package/dist/{index-D_P7Ye43.d.ts.map → index-DNgf5gyG.d.ts.map} +1 -1
- package/dist/index.d.ts +67 -37
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +47 -41
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-DRe8LB6d.d.ts → insight-report-08F022xN.d.ts} +4 -4
- package/dist/{insight-report-DRe8LB6d.d.ts.map → insight-report-08F022xN.d.ts.map} +1 -1
- package/dist/{integrity-DUNX9Fao.d.ts → integrity-B_EDELom.d.ts} +2 -2
- package/dist/{integrity-DUNX9Fao.d.ts.map → integrity-B_EDELom.d.ts.map} +1 -1
- package/dist/{internal-BDHPCnjk.js → internal-BMFSR8Ns.js} +3 -19
- package/dist/internal-BMFSR8Ns.js.map +1 -0
- package/dist/{judge-calibration-zZjLz8hr.js → judge-calibration-BnpVKtnb.js} +3 -13
- package/dist/judge-calibration-BnpVKtnb.js.map +1 -0
- package/dist/judge-calibration-C5CbMYce.d.ts.map +1 -1
- package/dist/{kind-factory-DY8FdoXf.js → kind-factory-DMeEoMQZ.js} +3 -10
- package/dist/kind-factory-DMeEoMQZ.js.map +1 -0
- package/dist/ledger-core/index.js +2 -2
- package/dist/{ledger-core-BOzlRygb.js → ledger-core-PIfjCbKn.js} +2 -2
- package/dist/{ledger-core-BOzlRygb.js.map → ledger-core-PIfjCbKn.js.map} +1 -1
- package/dist/{llm-client-hgDieDNN.js → llm-client-BFMRpmqb.js} +8 -11
- package/dist/llm-client-BFMRpmqb.js.map +1 -0
- package/dist/{llm-judge-BhasIPFT.js → llm-judge-DbJdo8Nj.js} +101 -23
- package/dist/llm-judge-DbJdo8Nj.js.map +1 -0
- package/dist/matrix/index.d.ts +2 -2
- package/dist/{matrix-eXKRMHnL.d.ts → matrix-BpI5Trmo.d.ts} +3 -3
- package/dist/{matrix-eXKRMHnL.d.ts.map → matrix-BpI5Trmo.d.ts.map} +1 -1
- package/dist/meta-eval/index.d.ts +5 -4
- package/dist/meta-eval/index.d.ts.map +1 -1
- package/dist/meta-eval/index.js +14 -22
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/{mint-DfODW1KW.js → mint-DjfDUMHr.js} +2 -2
- package/dist/{mint-DfODW1KW.js.map → mint-DjfDUMHr.js.map} +1 -1
- package/dist/multishot/golden/index.d.ts +1 -1
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +4 -4
- package/dist/{paired-arms-D-XRF_fy.js → paired-arms-D4aeIHUy.js} +3 -3
- package/dist/{paired-arms-D-XRF_fy.js.map → paired-arms-D4aeIHUy.js.map} +1 -1
- package/dist/{paired-tests-BHIhYVdu.js → paired-tests-C8iCsioC.js} +3 -3
- package/dist/{paired-tests-BHIhYVdu.js.map → paired-tests-C8iCsioC.js.map} +1 -1
- package/dist/pipelines/index.d.ts +7 -6
- package/dist/pipelines/index.d.ts.map +1 -1
- package/dist/pipelines/index.js +5 -20
- package/dist/pipelines/index.js.map +1 -1
- package/dist/{power-and-mde-CHIrXJll.js → power-and-mde-B8F2RdcD.js} +3 -3
- package/dist/{power-and-mde-CHIrXJll.js.map → power-and-mde-B8F2RdcD.js.map} +1 -1
- package/dist/{power-preflight-DEw-uC7q.js → power-preflight-CFXm0Vjo.js} +3 -3
- package/dist/{power-preflight-DEw-uC7q.js.map → power-preflight-CFXm0Vjo.js.map} +1 -1
- package/dist/{pareto-BqNW3LJR.d.ts → power-preflight-Ptse_Kq7.d.ts} +43 -43
- package/dist/power-preflight-Ptse_Kq7.d.ts.map +1 -0
- package/dist/{pre-registration-KN9jkh58.js → pre-registration-D94b7Of5.js} +2 -2
- package/dist/{pre-registration-KN9jkh58.js.map → pre-registration-D94b7Of5.js.map} +1 -1
- package/dist/{pre-registration-CzFCcwYk.d.ts → pre-registration-DHz6P_6f.d.ts} +2 -2
- package/dist/{pre-registration-CzFCcwYk.d.ts.map → pre-registration-DHz6P_6f.d.ts.map} +1 -1
- package/dist/{produced-state-DZ89riy5.js → produced-state-CtSIp5cQ.js} +5 -5
- package/dist/{produced-state-DZ89riy5.js.map → produced-state-CtSIp5cQ.js.map} +1 -1
- package/dist/profile-cell.js +1 -1
- package/dist/{promotion-policy-DtnOIZvk.d.ts → promotion-policy-CkXSgKkF.d.ts} +3 -3
- package/dist/{promotion-policy-DtnOIZvk.d.ts.map → promotion-policy-CkXSgKkF.d.ts.map} +1 -1
- package/dist/{promotion-policy-xzA40Evo.js → promotion-policy-LY9mVQ7W.js} +3 -3
- package/dist/{promotion-policy-xzA40Evo.js.map → promotion-policy-LY9mVQ7W.js.map} +1 -1
- package/dist/{run-score-lDzV0X8j.js → proposal-findings-bko3GGy-.js} +2 -31
- package/dist/proposal-findings-bko3GGy-.js.map +1 -0
- package/dist/{transient-failure-DKF5Mofa.d.ts → provenance-CIRUardl.d.ts} +891 -891
- package/dist/provenance-CIRUardl.d.ts.map +1 -0
- package/dist/{query-CHmMP42p.js → query-BPGMVlbM.js} +46 -4
- package/dist/query-BPGMVlbM.js.map +1 -0
- package/dist/{query-DxPYqpmT.d.ts → query-Na5gEIGd.d.ts} +21 -4
- package/dist/query-Na5gEIGd.d.ts.map +1 -0
- package/dist/random-Dn5fPWkt.js +21 -0
- package/dist/random-Dn5fPWkt.js.map +1 -0
- package/dist/record-id-DUgsK5qp.js +17 -0
- package/dist/record-id-DUgsK5qp.js.map +1 -0
- package/dist/{registry-8You7OK1.d.ts → registry-xEb_xfns.d.ts} +3 -3
- package/dist/{registry-8You7OK1.d.ts.map → registry-xEb_xfns.d.ts.map} +1 -1
- package/dist/{release-confidence-DKfD2RYU.js → release-confidence-CzUHc4z4.js} +4 -4
- package/dist/{release-confidence-DKfD2RYU.js.map → release-confidence-CzUHc4z4.js.map} +1 -1
- package/dist/{release-confidence-Dqt0NFep.d.ts → release-confidence-D6lQw_o7.d.ts} +4 -4
- package/dist/{release-confidence-Dqt0NFep.d.ts.map → release-confidence-D6lQw_o7.d.ts.map} +1 -1
- package/dist/reporting.d.ts +3 -3
- package/dist/reporting.js +6 -6
- package/dist/{researcher-Cz565b7D.d.ts → researcher-CMUTQXD7.d.ts} +6 -6
- package/dist/{researcher-Cz565b7D.d.ts.map → researcher-CMUTQXD7.d.ts.map} +1 -1
- package/dist/{reward-hacking-MBf7qpSB.d.ts → reward-hacking-CgPRUesA.d.ts} +2 -2
- package/dist/{reward-hacking-MBf7qpSB.d.ts.map → reward-hacking-CgPRUesA.d.ts.map} +1 -1
- package/dist/{reward-hacking-t4lB1yt8.js → reward-hacking-SkxYgT0x.js} +3 -3
- package/dist/{reward-hacking-t4lB1yt8.js.map → reward-hacking-SkxYgT0x.js.map} +1 -1
- package/dist/rl.d.ts +8 -8
- package/dist/rl.d.ts.map +1 -1
- package/dist/rl.js +13 -18
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +1 -1
- package/dist/rollout/index.js +2 -2
- package/dist/{rollout-Dm2tSdiQ.js → rollout-Crypdx8s.js} +2 -2
- package/dist/{rollout-Dm2tSdiQ.js.map → rollout-Crypdx8s.js.map} +1 -1
- package/dist/{rubric-predictive-validity-CK8SCOg-.js → rubric-predictive-validity-2D5Gw9z9.js} +3 -3
- package/dist/{rubric-predictive-validity-CK8SCOg-.js.map → rubric-predictive-validity-2D5Gw9z9.js.map} +1 -1
- package/dist/{rubric-predictive-validity-CxycqzX5.d.ts → rubric-predictive-validity-DluJLCKQ.d.ts} +2 -2
- package/dist/{rubric-predictive-validity-CxycqzX5.d.ts.map → rubric-predictive-validity-DluJLCKQ.d.ts.map} +1 -1
- package/dist/{run-record-BC0ebuRP.js → run-record-DLORoL7t.js} +2 -2
- package/dist/{run-record-BC0ebuRP.js.map → run-record-DLORoL7t.js.map} +1 -1
- package/dist/{run-record-VVy4T9OW.d.ts → run-record-DQjRcYwA.d.ts} +3 -3
- package/dist/{run-record-VVy4T9OW.d.ts.map → run-record-DQjRcYwA.d.ts.map} +1 -1
- package/dist/{schema-k6ZBftVv.js → schema-CdIX2aHu.js} +5 -1
- package/dist/{schema-k6ZBftVv.js.map → schema-CdIX2aHu.js.map} +1 -1
- package/dist/{schema-Bjgdsn73.d.ts → schema-DID1Cqct.d.ts} +7 -3
- package/dist/{schema-Bjgdsn73.d.ts.map → schema-DID1Cqct.d.ts.map} +1 -1
- package/dist/{semantic-concept-judge-BSkKKHeq.js → semantic-concept-judge-I36eejJx.js} +3 -3
- package/dist/{semantic-concept-judge-BSkKKHeq.js.map → semantic-concept-judge-I36eejJx.js.map} +1 -1
- package/dist/{sequential-rYW-Ophm.js → sequential-B51qAYE4.js} +4 -4
- package/dist/{sequential-rYW-Ophm.js.map → sequential-B51qAYE4.js.map} +1 -1
- package/dist/{server-BtFd4uzB.js → server-CCEnywOR.js} +27 -19
- package/dist/server-CCEnywOR.js.map +1 -0
- package/dist/{skillopt-optimization-method-DbaekMcn.js → skillopt-optimization-method-B2R9C5aG.js} +11 -11
- package/dist/{skillopt-optimization-method-DbaekMcn.js.map → skillopt-optimization-method-B2R9C5aG.js.map} +1 -1
- package/dist/{statistical-heldout-Cy3EhjlC.d.ts → statistical-heldout-DFS7QGpS.d.ts} +3 -3
- package/dist/{statistical-heldout-Cy3EhjlC.d.ts.map → statistical-heldout-DFS7QGpS.d.ts.map} +1 -1
- package/dist/{store-B06JdC56.d.ts → store-Cq9oOrI1.d.ts} +2 -2
- package/dist/{store-B06JdC56.d.ts.map → store-Cq9oOrI1.d.ts.map} +1 -1
- package/dist/{store-otlp-C_Rq5I4D.js → store-otlp-CHjBvWQY.js} +2 -2
- package/dist/{store-otlp-C_Rq5I4D.js.map → store-otlp-CHjBvWQY.js.map} +1 -1
- package/dist/{store-tool-spans-DPUG7UUY.d.ts → store-tool-spans-B2DJ_82T.d.ts} +102 -36
- package/dist/store-tool-spans-B2DJ_82T.d.ts.map +1 -0
- package/dist/{store-tool-spans-Dlh9vkFK.js → store-tool-spans-B9o6tU8f.js} +23 -19
- package/dist/store-tool-spans-B9o6tU8f.js.map +1 -0
- package/dist/storyboard/index.d.ts +1 -1
- package/dist/{student-t-CvBq2mve.js → student-t-BA-Uy51p.js} +2 -2
- package/dist/{student-t-CvBq2mve.js.map → student-t-BA-Uy51p.js.map} +1 -1
- package/dist/{summary-report-BI5hUtvK.js → summary-report-Bgh8CpNK.js} +7 -7
- package/dist/{summary-report-BI5hUtvK.js.map → summary-report-Bgh8CpNK.js.map} +1 -1
- package/dist/{summary-report-CC07PhEL.d.ts → summary-report-DRstQNBX.d.ts} +3 -3
- package/dist/{summary-report-CC07PhEL.d.ts.map → summary-report-DRstQNBX.d.ts.map} +1 -1
- package/dist/supervisor-run/index.js +1 -1
- package/dist/{task-failure-attributes-DTl-7-Kw.js → task-failure-attributes-CBGtLS_H.js} +3 -3
- package/dist/{task-failure-attributes-DTl-7-Kw.js.map → task-failure-attributes-CBGtLS_H.js.map} +1 -1
- package/dist/{tool-groups-Ci8i9ErB.d.ts → tool-groups-BnXlCJZQ.d.ts} +3 -3
- package/dist/tool-groups-BnXlCJZQ.d.ts.map +1 -0
- package/dist/{tool-waste-Dro0gJi3.d.ts → tool-waste-BrmLKxMw.d.ts} +4 -4
- package/dist/{tool-waste-Dro0gJi3.d.ts.map → tool-waste-BrmLKxMw.d.ts.map} +1 -1
- package/dist/{tool-waste-BqzmVdJk.js → tool-waste-CwGHzBzX.js} +4 -4
- package/dist/{tool-waste-BqzmVdJk.js.map → tool-waste-CwGHzBzX.js.map} +1 -1
- package/dist/trace-repair/index.d.ts +3 -3
- package/dist/trace-repair/index.d.ts.map +1 -1
- package/dist/trace-repair/index.js +5 -4
- package/dist/trace-repair/index.js.map +1 -1
- package/dist/traces.d.ts +60 -62
- package/dist/traces.d.ts.map +1 -1
- package/dist/traces.js +13 -22
- package/dist/traces.js.map +1 -1
- package/dist/{trajectory-Bi157Gun.d.ts → trajectory-r1bQqvBQ.d.ts} +3 -3
- package/dist/{trajectory-Bi157Gun.d.ts.map → trajectory-r1bQqvBQ.d.ts.map} +1 -1
- package/dist/trajectory-replay/index.d.ts +3 -3
- package/dist/trajectory-replay/index.d.ts.map +1 -1
- package/dist/trajectory-replay/index.js +4 -14
- package/dist/trajectory-replay/index.js.map +1 -1
- package/dist/types-Bfk0uxRj.d.ts.map +1 -1
- package/dist/{types-BI4fT3HN.js → types-CiWITkGo.js} +11 -2
- package/dist/types-CiWITkGo.js.map +1 -0
- package/dist/{types-D9ssmxKL.d.ts → types-DMoNFDWi.d.ts} +6 -3
- package/dist/{types-D9ssmxKL.d.ts.map → types-DMoNFDWi.d.ts.map} +1 -1
- package/dist/{types-D4s7Z6nq.d.ts → types-Dy237wiH.d.ts} +3 -3
- package/dist/{types-D4s7Z6nq.d.ts.map → types-Dy237wiH.d.ts.map} +1 -1
- package/dist/{types-BPb2Kf_C2.d.ts → types-i21ccEkr.d.ts} +3 -3
- package/dist/types-i21ccEkr.d.ts.map +1 -0
- package/dist/{verdict-B0xltqu6.js → verdict-BQ3pCFf8.js} +2 -2
- package/dist/{verdict-B0xltqu6.js.map → verdict-BQ3pCFf8.js.map} +1 -1
- package/dist/{verdict-cache-CdVVTVmn.js → verdict-cache-B3eCVQtY.js} +2 -2
- package/dist/{verdict-cache-CdVVTVmn.js.map → verdict-cache-B3eCVQtY.js.map} +1 -1
- package/dist/wire/index.d.ts +23 -8
- package/dist/wire/index.d.ts.map +1 -1
- package/dist/wire/index.js +2 -2
- package/docs/code-agent-intake.md +64 -0
- package/docs/concepts.md +1 -1
- package/docs/design/statistics-decisions.md +1 -1
- package/docs/distributed-driver.md +3 -6
- package/docs/public-api.md +122 -106
- package/docs/wire-protocol.md +5 -3
- package/package.json +9 -2
- package/dist/benchmark-BhT16ep9.js.map +0 -1
- package/dist/benchmark-command-BDC3Gocz.js.map +0 -1
- package/dist/campaign-BSmOwskD.js.map +0 -1
- package/dist/capture-fetch-CqwsJkkG.d.ts +0 -68
- package/dist/capture-fetch-CqwsJkkG.d.ts.map +0 -1
- package/dist/contract/index.d.ts.map +0 -1
- package/dist/dspy-rlm-engine-DptEII26.js.map +0 -1
- package/dist/emitter-BpYFQPj4.js.map +0 -1
- package/dist/external-optimizer-process-WosTBChy.js.map +0 -1
- package/dist/external-optimizer-subprocess-BIWbHpgD.js.map +0 -1
- package/dist/index-CGtH1piv.d.ts.map +0 -1
- package/dist/index-D-V8gCs_.d.ts.map +0 -1
- package/dist/index-vrJugRal.d.ts +0 -1
- package/dist/internal-BDHPCnjk.js.map +0 -1
- package/dist/judge-calibration-zZjLz8hr.js.map +0 -1
- package/dist/kind-factory-DY8FdoXf.js.map +0 -1
- package/dist/llm-client-hgDieDNN.js.map +0 -1
- package/dist/llm-judge-BhasIPFT.js.map +0 -1
- package/dist/pareto-BqNW3LJR.d.ts.map +0 -1
- package/dist/query-CHmMP42p.js.map +0 -1
- package/dist/query-DxPYqpmT.d.ts.map +0 -1
- package/dist/run-score-lDzV0X8j.js.map +0 -1
- package/dist/server-BtFd4uzB.js.map +0 -1
- package/dist/skillopt-optimization-method-x7TTF23P.d.ts.map +0 -1
- package/dist/store-tool-spans-DPUG7UUY.d.ts.map +0 -1
- package/dist/store-tool-spans-Dlh9vkFK.js.map +0 -1
- package/dist/tool-groups-Ci8i9ErB.d.ts.map +0 -1
- package/dist/transient-failure-DKF5Mofa.d.ts.map +0 -1
- package/dist/types-BI4fT3HN.js.map +0 -1
- package/dist/types-BPb2Kf_C2.d.ts.map +0 -1
package/dist/contract/index.d.ts
CHANGED
|
@@ -1,706 +1,15 @@
|
|
|
1
1
|
import { c as CostLedgerHandle } from "../cost-ledger-DbQdN3nO.js";
|
|
2
|
-
import {
|
|
3
|
-
import {
|
|
2
|
+
import { _ as ProposalFinding, i as AnalystFinding, v as ProposalFindingOrigin, x as makeProposalFinding } from "../types-DMoNFDWi.js";
|
|
3
|
+
import { An as runEval, B as OptimizationMethodResult, F as OptimizationMethod, Gn as fsCampaignStorage, I as OptimizationMethodComparison, K as compareOptimizationMethods, Kn as inMemoryCampaignStorage, L as OptimizationMethodInput, Mn as CampaignCellRetryPolicy, N as CompareOptimizationMethodsOptions, Nn as RunCampaignOptions, P as ComparisonCost, Pn as runCampaign, U as OptimizationPackageSource, Un as CampaignStorage, W as OptimizationTokenUsage, at as transientDispatchFailure, ct as runImprovementLoop, et as LlmJudgeDimension, jn as CampaignCellFailureReceipt, kn as RunEvalOptions, nt as llmJudge, ot as RunImprovementLoopOptions, st as RunImprovementLoopResult, tt as LlmJudgeOptions, z as OptimizationMethodProvenance } from "../provenance-CIRUardl.js";
|
|
4
4
|
import { _ as CreateChatClientOpts, p as ChatClient, x as createChatClient } from "../types-Bfk0uxRj.js";
|
|
5
|
-
import { _ as
|
|
6
|
-
import { n as
|
|
7
|
-
import {
|
|
8
|
-
import { a as FailureClusterInsight, c as JudgeInsight, d as Recommendation, f as ReleaseSummary, i as FailureClassTally, l as LiftInsight, m as TokenUsageInsight, n as ExecutionErrorOutcomeCell, o as InsightReport, p as ScalarDistribution, r as ExecutionInsight, s as InterRaterInsight, t as CostProvenanceSummary, u as OutcomeCorrelationInsight } from "../insight-report-DRe8LB6d.js";
|
|
9
|
-
import { _ as AnalyzeRunsOptions, b as analyzeRuns, v as ExecutionReport, x as summarizeExecution, y as SummarizeExecutionOptions } from "../engine-Cu5qD5Fc.js";
|
|
10
|
-
import { B as runReferenceEquivalenceJudge, C as ExternalTextOptimizerContext, E as ExternalTextEvaluationResponse, F as ReferenceEquivalenceJudgeInput, I as ReferenceEquivalenceJudgeOptions, L as ReferenceEquivalenceJudgeResult, N as REFERENCE_EQUIVALENCE_INPUT_LIMITS, P as REFERENCE_EQUIVALENCE_JUDGE_VERSION, R as ReferenceEquivalenceScenario, S as ExternalTextOptimizationMethodConfig, T as ExternalOptimizationExample, _ as DefaultProductionGateOptions, a as GepaAdaptiveEngineRun, b as composeGate, c as GepaOptimizationMethodConfig, d as gepaOptimizationMethod, f as OpenAICompatibleOptimizerModel, g as DefaultProductionGateCheck, h as heldOutGate, i as skillOptOptimizationMethod, j as campaignSplitDigest, l as GepaOptimizationRecipe, m as HeldOutGateOptions, n as SkillOptRunnerCommand, o as GepaEngineOptions, p as OptimizerModelBudget, r as SkillOptTrainerConfig, s as GepaEngineRun, t as SkillOptOptimizationMethodConfig, u as GepaRunnerCommand, v as DefaultProductionRewardHackingOptions, w as ExternalTextOptimizerResult, x as externalTextOptimizationMethod, y as defaultProductionGate, z as createReferenceEquivalenceJudge } from "../skillopt-optimization-method-x7TTF23P.js";
|
|
11
|
-
import { a as ObjectiveSource, c as PromotionPolicy, d as paretoSignificanceGate, i as EvidenceVector, l as buildEvidenceVector, n as AxisVerdict, o as ParetoSignificanceGateOptions, r as BuildEvidenceVectorOptions, s as PromotionObjective, t as AxisEvidence, u as paretoPolicy } from "../promotion-policy-DtnOIZvk.js";
|
|
12
|
-
import { c as EvalRunGenerationSnapshot, g as TraceSpanEvent, n as HostedTenant, o as EvalRunCellScore, s as EvalRunEvent } from "../client-L9VVPkim.js";
|
|
5
|
+
import { C as JudgeDimension, H as SurfaceProposer, M as OptimizerConfig, R as Scenario, S as JudgeConfig, V as SessionScript, _ as GateDecision, a as CampaignResult, b as GenerationRecord, c as CampaignTraceWriter, d as DispatchContext, f as DispatchFn, g as GateContribution, h as GateContext, i as CampaignCostMeter, j as MutableSurface, l as CodeSurface, m as GateCheckStatus, n as CampaignArtifactWriter, p as Gate, r as CampaignCellResult, t as CampaignAggregates, v as GateResult, w as JudgeScore, y as GenerationCandidate } from "../types-Dy237wiH.js";
|
|
6
|
+
import { A as ReferenceEquivalenceJudgeInput, C as ExternalTextOptimizerContext, E as ExternalTextEvaluationResponse, F as runReferenceEquivalenceJudge, M as ReferenceEquivalenceJudgeResult, N as ReferenceEquivalenceScenario, O as REFERENCE_EQUIVALENCE_INPUT_LIMITS, P as createReferenceEquivalenceJudge, S as ExternalTextOptimizationMethodConfig, T as ExternalOptimizationExample, _ as GepaRunnerCommand, a as DefaultProductionRewardHackingOptions, b as OptimizerModelBudget, c as SkillOptOptimizationMethodConfig, d as skillOptOptimizationMethod, f as GepaAdaptiveEngineRun, g as GepaOptimizationRecipe, h as GepaOptimizationMethodConfig, i as DefaultProductionGateOptions, j as ReferenceEquivalenceJudgeOptions, k as REFERENCE_EQUIVALENCE_JUDGE_VERSION, l as SkillOptRunnerCommand, m as GepaEngineRun, n as heldOutGate, o as defaultProductionGate, p as GepaEngineOptions, r as DefaultProductionGateCheck, s as composeGate, t as HeldOutGateOptions, u as SkillOptTrainerConfig, v as gepaOptimizationMethod, w as ExternalTextOptimizerResult, x as externalTextOptimizationMethod, y as OpenAICompatibleOptimizerModel, z as campaignSplitDigest } from "../heldout-gate-Bn7_xWCv.js";
|
|
7
|
+
import { a as ObjectiveSource, c as PromotionPolicy, d as paretoSignificanceGate, i as EvidenceVector, l as buildEvidenceVector, n as AxisVerdict, o as ParetoSignificanceGateOptions, r as BuildEvidenceVectorOptions, s as PromotionObjective, t as AxisEvidence, u as paretoPolicy } from "../promotion-policy-CkXSgKkF.js";
|
|
13
8
|
import { i as InMemoryOutcomeStore, n as FileSystemOutcomeStore, o as OutcomeStore, r as FileSystemOutcomeStoreOptions, t as DeploymentOutcome } from "../outcome-store-BYHIuO0e.js";
|
|
14
|
-
import { a as
|
|
15
|
-
import {
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
}
|
|
22
|
-
interface CandidateExperimentExecutionInput {
|
|
23
|
-
experiment: AgentCandidateExperiment;
|
|
24
|
-
arm: 'baseline' | 'candidate';
|
|
25
|
-
bundle: AgentCandidateBundle;
|
|
26
|
-
task: AgentCandidateBenchmarkTask;
|
|
27
|
-
benchmarkCell: AgentCandidateBenchmarkCellRef;
|
|
28
|
-
seed: number;
|
|
29
|
-
signal?: AbortSignal;
|
|
30
|
-
}
|
|
31
|
-
interface RunCandidateExperimentOptions {
|
|
32
|
-
experiment: AgentCandidateExperiment;
|
|
33
|
-
execute(input: CandidateExperimentExecutionInput): Promise<CandidateExecutionEvidence>;
|
|
34
|
-
/** Maximum number of simultaneous execute calls across both arms. */
|
|
35
|
-
maxConcurrency?: number;
|
|
36
|
-
/** Shared run budget that also accounts for analysis and candidate search. */
|
|
37
|
-
costLedger?: CostLedgerHandle;
|
|
38
|
-
signal?: AbortSignal;
|
|
39
|
-
}
|
|
40
|
-
interface CandidateExperimentRun {
|
|
41
|
-
measurements: AgentCandidateExperimentMeasurement[];
|
|
42
|
-
measurement: {
|
|
43
|
-
wallDurationMs: number;
|
|
44
|
-
cost: AgentImprovementCost;
|
|
45
|
-
};
|
|
46
|
-
}
|
|
47
|
-
interface CompareCandidateExperimentOptions {
|
|
48
|
-
experiment: AgentCandidateExperiment;
|
|
49
|
-
measurements: AgentCandidateExperimentMeasurement[];
|
|
50
|
-
preparation: {
|
|
51
|
-
wallDurationMs: number;
|
|
52
|
-
cost: AgentImprovementCost;
|
|
53
|
-
};
|
|
54
|
-
measurement: CandidateExperimentRun['measurement'];
|
|
55
|
-
runId: string;
|
|
56
|
-
candidate?: AgentImprovementMeasuredComparison['candidate'];
|
|
57
|
-
generationsExplored?: number;
|
|
58
|
-
metadata?: AgentImprovementMeasuredComparison['metadata'];
|
|
59
|
-
}
|
|
60
|
-
/** One exact baseline/candidate observation of the same held-out cell. */
|
|
61
|
-
interface PairedMeasurement<TRun> {
|
|
62
|
-
cellId: string;
|
|
63
|
-
baseline: TRun;
|
|
64
|
-
candidate: TRun;
|
|
65
|
-
}
|
|
66
|
-
/** Maps a product-owned run receipt into the measurements required for a fair paired decision. */
|
|
67
|
-
interface PairedMeasurementAdapter<TRun> {
|
|
68
|
-
score(run: TRun): number;
|
|
69
|
-
dimensions(run: TRun): readonly {
|
|
70
|
-
name: string;
|
|
71
|
-
score: number;
|
|
72
|
-
}[];
|
|
73
|
-
costUsd(run: TRun): number;
|
|
74
|
-
costProvenance(run: TRun): AgentImprovementCost['provenance'];
|
|
75
|
-
latencyMs(run: TRun): number;
|
|
76
|
-
completed(run: TRun): boolean;
|
|
77
|
-
passed(run: TRun): boolean;
|
|
78
|
-
}
|
|
79
|
-
interface EvaluatePairedMeasurementsOptions<TRun> {
|
|
80
|
-
measurements: readonly PairedMeasurement<TRun>[];
|
|
81
|
-
policy: AgentCandidateEvaluationPolicy;
|
|
82
|
-
adapter: PairedMeasurementAdapter<TRun>;
|
|
83
|
-
/** Whether both arms use the same scorer family as the promotion decision. */
|
|
84
|
-
sharedScorerChannel: boolean;
|
|
85
|
-
/** Analysis and candidate-search spend that belongs to the same frozen budget. */
|
|
86
|
-
preparationCost?: AgentImprovementCost;
|
|
87
|
-
/** Settled aggregate receipt for this exact paired suite, when an executor provides one. */
|
|
88
|
-
measurementCost?: AgentImprovementCost;
|
|
89
|
-
}
|
|
90
|
-
/** Statistical and operational result derived from complete paired receipts. */
|
|
91
|
-
type PairedMeasurementEvaluation = Pick<AgentImprovementMeasuredComparison, 'overall' | 'objectives' | 'decision' | 'power'> & {
|
|
92
|
-
measurementCost: AgentImprovementCost;
|
|
93
|
-
totalCost: AgentImprovementCost;
|
|
94
|
-
measurementWorkDurationMs: number;
|
|
95
|
-
};
|
|
96
|
-
/** Content-address one task before any measured execution can see it. */
|
|
97
|
-
declare function sealCandidateBenchmarkTask(material: AgentCandidateBenchmarkTaskMaterial): AgentCandidateBenchmarkTask;
|
|
98
|
-
/** Freeze task order, repetitions, and every seed before either arm runs. */
|
|
99
|
-
declare function sealCandidateBenchmarkSuite(options: SealCandidateBenchmarkSuiteOptions): AgentCandidateBenchmarkSuiteInputs;
|
|
100
|
-
/** Freeze both complete agent states and their exact held-out work. */
|
|
101
|
-
declare function sealCandidateExperiment(material: AgentCandidateExperimentMaterial): AgentCandidateExperiment;
|
|
102
|
-
declare function verifyCandidateExperiment(input: unknown): AgentCandidateExperiment;
|
|
103
|
-
/** Execute each signed cell for both arms. The callback is Runtime's one executor. */
|
|
104
|
-
declare function runCandidateExperiment(options: RunCandidateExperimentOptions): Promise<CandidateExperimentRun>;
|
|
105
|
-
/**
|
|
106
|
-
* Calculate the shared paired decision from any complete receipt shape.
|
|
107
|
-
*
|
|
108
|
-
* Callers still own sealing their tasks, verifying each receipt against its
|
|
109
|
-
* expected arm and state, and proving every expected cell exists. This function
|
|
110
|
-
* only validates the projected measurements and derives their shared decision.
|
|
111
|
-
*/
|
|
112
|
-
declare function evaluatePairedMeasurements<TRun>(options: EvaluatePairedMeasurementsOptions<TRun>): PairedMeasurementEvaluation;
|
|
113
|
-
/** Build the only publishable comparison: paired statistics over Runtime receipts. */
|
|
114
|
-
declare function measuredComparisonFromCandidateExperiment(options: CompareCandidateExperimentOptions): AgentImprovementMeasuredComparison;
|
|
115
|
-
/** Recompute every statistic and decision from the signed experiment receipts. */
|
|
116
|
-
declare function verifyCandidateExperimentComparison(input: unknown): AgentImprovementMeasuredComparison;
|
|
117
|
-
//#endregion
|
|
118
|
-
//#region src/contract/profile-measured-comparison.d.ts
|
|
119
|
-
interface SealAgentProfileImprovementSuiteOptions {
|
|
120
|
-
splitDigest: Sha256Digest;
|
|
121
|
-
tasks: [AgentProfileImprovementTask, ...AgentProfileImprovementTask[]];
|
|
122
|
-
reps: number;
|
|
123
|
-
seeds: [number, ...number[]];
|
|
124
|
-
}
|
|
125
|
-
interface AgentProfileImprovementExperimentExecutionInput {
|
|
126
|
-
experiment: AgentProfileImprovementExperiment;
|
|
127
|
-
arm: 'baseline' | 'candidate';
|
|
128
|
-
stateDigest: Sha256Digest;
|
|
129
|
-
task: AgentProfileImprovementTask;
|
|
130
|
-
runCell: AgentProfileImprovementRunCell;
|
|
131
|
-
seed: number;
|
|
132
|
-
signal?: AbortSignal;
|
|
133
|
-
}
|
|
134
|
-
interface RunAgentProfileImprovementExperimentOptions {
|
|
135
|
-
experiment: AgentProfileImprovementExperiment;
|
|
136
|
-
execute(input: AgentProfileImprovementExperimentExecutionInput): Promise<AgentProfileImprovementRunReceipt>;
|
|
137
|
-
/** Maximum number of simultaneous execute calls across both arms. */
|
|
138
|
-
maxConcurrency?: number;
|
|
139
|
-
/** Shared run budget that also accounts for analysis and candidate search. */
|
|
140
|
-
costLedger?: CostLedgerHandle;
|
|
141
|
-
signal?: AbortSignal;
|
|
142
|
-
}
|
|
143
|
-
interface AgentProfileImprovementExperimentRun {
|
|
144
|
-
measurements: AgentProfileImprovementMeasurement[];
|
|
145
|
-
measurement: {
|
|
146
|
-
wallDurationMs: number;
|
|
147
|
-
cost: AgentImprovementCost;
|
|
148
|
-
};
|
|
149
|
-
}
|
|
150
|
-
interface CompareAgentProfileImprovementExperimentOptions {
|
|
151
|
-
experiment: AgentProfileImprovementExperiment;
|
|
152
|
-
measurements: AgentProfileImprovementMeasurement[];
|
|
153
|
-
preparation: {
|
|
154
|
-
wallDurationMs: number;
|
|
155
|
-
cost: AgentImprovementCost;
|
|
156
|
-
};
|
|
157
|
-
measurement: AgentProfileImprovementExperimentRun['measurement'];
|
|
158
|
-
runId: string;
|
|
159
|
-
candidate?: AgentProfileImprovementMeasuredComparison['candidate'];
|
|
160
|
-
generationsExplored?: number;
|
|
161
|
-
metadata?: AgentProfileImprovementMeasuredComparison['metadata'];
|
|
162
|
-
}
|
|
163
|
-
/** Content-address one held-out profile task before either state can execute it. */
|
|
164
|
-
declare function sealAgentProfileImprovementTask(material: AgentProfileImprovementTaskMaterial): AgentProfileImprovementTask;
|
|
165
|
-
/** Freeze profile task order, repetitions, seeds, and the held-out split. */
|
|
166
|
-
declare function sealAgentProfileImprovementSuite(options: SealAgentProfileImprovementSuiteOptions): AgentProfileImprovementSuiteInputs;
|
|
167
|
-
/** Freeze the two host-owned profile states and their exact held-out work. */
|
|
168
|
-
declare function sealAgentProfileImprovementExperiment(material: AgentProfileImprovementExperimentMaterial): AgentProfileImprovementExperiment;
|
|
169
|
-
/**
|
|
170
|
-
* Execute each signed profile cell through the host's one exact-state executor.
|
|
171
|
-
* Eval owns only the cell schedule and receipt checks; the host resolves each
|
|
172
|
-
* state digest and captures its own run, billing, trace, and grader evidence.
|
|
173
|
-
*/
|
|
174
|
-
declare function runAgentProfileImprovementExperiment(options: RunAgentProfileImprovementExperimentOptions): Promise<AgentProfileImprovementExperimentRun>;
|
|
175
|
-
/** Build the only publishable profile comparison from complete host receipts. */
|
|
176
|
-
declare function measuredComparisonFromAgentProfileImprovementExperiment(options: CompareAgentProfileImprovementExperimentOptions): AgentProfileImprovementMeasuredComparison;
|
|
177
|
-
/** Recompute a profile comparison from the exact sealed experiment and receipts. */
|
|
178
|
-
declare function verifyAgentProfileImprovementExperimentComparison(input: unknown): AgentProfileImprovementMeasuredComparison;
|
|
179
|
-
//#endregion
|
|
180
|
-
//#region src/contract/intake/run-record-dir.d.ts
|
|
181
|
-
/** A record that failed boundary validation, with enough context to fix it. */
|
|
182
|
-
interface RunRecordRejection {
|
|
183
|
-
/** Absolute or caller-relative path to the file the record came from. */
|
|
184
|
-
file: string;
|
|
185
|
-
/** Zero-based position within the file (array index or JSONL line number). */
|
|
186
|
-
index: number;
|
|
187
|
-
/** The validator's message. */
|
|
188
|
-
reason: string;
|
|
189
|
-
}
|
|
190
|
-
interface FromRunRecordDirOptions {
|
|
191
|
-
/**
|
|
192
|
-
* How to treat a record that fails `parseRunRecordSafe`:
|
|
193
|
-
* - `'throw'` (default) — fail loud on the first invalid record.
|
|
194
|
-
* - `'collect'` — drop it, keep the rest, and return it under `rejected`.
|
|
195
|
-
*/
|
|
196
|
-
onInvalid?: 'throw' | 'collect';
|
|
197
|
-
/**
|
|
198
|
-
* When the input is a directory, only files matching this predicate are
|
|
199
|
-
* read. Default: any file ending in `.json` or `.jsonl`. The `analysis.json`
|
|
200
|
-
* artifact `evalReportingSuite` writes is always skipped so a re-run never
|
|
201
|
-
* ingests its own output.
|
|
202
|
-
*/
|
|
203
|
-
include?: (fileName: string) => boolean;
|
|
204
|
-
/**
|
|
205
|
-
* Recurse into subdirectories when the input is a directory. Default false —
|
|
206
|
-
* a flat run directory is the common case and recursion can silently pull in
|
|
207
|
-
* unrelated corpora.
|
|
208
|
-
*/
|
|
209
|
-
recursive?: boolean;
|
|
210
|
-
}
|
|
211
|
-
interface FromRunRecordDirResult {
|
|
212
|
-
/** Records that passed boundary validation, in file-then-index order. */
|
|
213
|
-
runs: RunRecord[];
|
|
214
|
-
/** Records that failed validation. Empty unless `onInvalid: 'collect'`. */
|
|
215
|
-
rejected: RunRecordRejection[];
|
|
216
|
-
/** The files that were read, in the order they were processed. */
|
|
217
|
-
files: string[];
|
|
218
|
-
}
|
|
219
|
-
/**
|
|
220
|
-
* Resolve a file or directory path into validated `RunRecord[]`.
|
|
221
|
-
*
|
|
222
|
-
* A `.json` file must parse to a top-level array; a `.jsonl` file is one
|
|
223
|
-
* record per non-empty line. Directories are read shallowly by default
|
|
224
|
-
* (set `recursive` to descend); the `analysis.json` output artifact is
|
|
225
|
-
* always excluded.
|
|
226
|
-
*/
|
|
227
|
-
declare function fromRunRecordDir(path: string, options?: FromRunRecordDirOptions): Promise<FromRunRecordDirResult>;
|
|
228
|
-
//#endregion
|
|
229
|
-
//#region src/contract/eval-reporting-suite.d.ts
|
|
230
|
-
/** Either records in hand or a path to a `.json` / `.jsonl` file or a
|
|
231
|
-
* directory of them. */
|
|
232
|
-
type EvalReportingSuiteInput = RunRecord[] | string;
|
|
233
|
-
interface EvalReportingSuiteOptions {
|
|
234
|
-
/** Forwarded verbatim to `analyzeRuns` (everything except `runs`, which the
|
|
235
|
-
* suite supplies from the resolved input). Use this for split selection,
|
|
236
|
-
* baseline/candidate ids, canaries, prior-period runs, the analyst registry,
|
|
237
|
-
* etc. */
|
|
238
|
-
analyze?: Omit<AnalyzeRunsOptions, 'runs'>;
|
|
239
|
-
/** Loader options used only when the input is a path. */
|
|
240
|
-
load?: FromRunRecordDirOptions;
|
|
241
|
-
/**
|
|
242
|
-
* Write the suite result as a single `analysis.json`.
|
|
243
|
-
* - `true` — write to `<dir>/analysis.json` when the input is a directory,
|
|
244
|
-
* or alongside the input file; throws if the input is in-memory records
|
|
245
|
-
* (no directory to anchor to — pass an explicit path instead).
|
|
246
|
-
* - a string — write to exactly this path (a directory path gets
|
|
247
|
-
* `analysis.json` appended; any other path is used verbatim).
|
|
248
|
-
* - omitted / false — do not write.
|
|
249
|
-
*/
|
|
250
|
-
write?: boolean | string;
|
|
251
|
-
}
|
|
252
|
-
/** The suite artifact — the `analyzeRuns` report plus provenance. This is the
|
|
253
|
-
* exact shape serialized to `analysis.json`. */
|
|
254
|
-
interface EvalReportingSuiteResult {
|
|
255
|
-
/** The analysis itself — distributions, paired stats/lift, failure rollup,
|
|
256
|
-
* recommendations. Produced by `analyzeRuns`. */
|
|
257
|
-
report: InsightReport;
|
|
258
|
-
/** How the suite was run, so a reader can verify provenance. */
|
|
259
|
-
provenance: {
|
|
260
|
-
/** ISO timestamp the suite ran. */
|
|
261
|
-
generatedAt: string;
|
|
262
|
-
/** Number of records analyzed (mirrors `report.n`). */
|
|
263
|
-
runCount: number;
|
|
264
|
-
/** The source path when the input was a directory/file; null for
|
|
265
|
-
* in-memory records. */
|
|
266
|
-
sourcePath: string | null;
|
|
267
|
-
/** Files read when loading from disk; empty for in-memory input. */
|
|
268
|
-
files: string[];
|
|
269
|
-
/** Records dropped at the validation boundary. Always empty unless
|
|
270
|
-
* `load.onInvalid` was set to `'collect'`. */
|
|
271
|
-
rejected: FromRunRecordDirResult['rejected'];
|
|
272
|
-
};
|
|
273
|
-
/** The path `analysis.json` was written to, or null when `write` was unset. */
|
|
274
|
-
writtenTo: string | null;
|
|
275
|
-
}
|
|
276
|
-
/**
|
|
277
|
-
* Resolve runs (or a run dir/file), run `analyzeRuns`, and optionally persist a
|
|
278
|
-
* single `analysis.json`. The only analysis logic lives in `analyzeRuns`; this
|
|
279
|
-
* function is composition + I/O.
|
|
280
|
-
*/
|
|
281
|
-
declare function evalReportingSuite(input: EvalReportingSuiteInput, options?: EvalReportingSuiteOptions): Promise<EvalReportingSuiteResult>;
|
|
282
|
-
//#endregion
|
|
283
|
-
//#region src/contract/diff.d.ts
|
|
284
|
-
/** Per-dimension delta. `before` / `after` are null when the judge did not
|
|
285
|
-
* emit a value for that side. `delta` is `after - before`; null when
|
|
286
|
-
* either side is null. */
|
|
287
|
-
interface EvalDimensionDelta {
|
|
288
|
-
before: number | null;
|
|
289
|
-
after: number | null;
|
|
290
|
-
delta: number | null;
|
|
291
|
-
}
|
|
292
|
-
/** Per-cell delta, keyed on `(scenarioId, rep)`. */
|
|
293
|
-
interface EvalCellScoreDelta {
|
|
294
|
-
scenarioId: string;
|
|
295
|
-
rep: number;
|
|
296
|
-
compositeBefore: number | null;
|
|
297
|
-
compositeAfter: number | null;
|
|
298
|
-
compositeDelta: number | null;
|
|
299
|
-
/** Per-judge → per-dimension deltas. Outer key = judge name from
|
|
300
|
-
* `EvalRunCellScore.dimensions`; inner key = dimension name. */
|
|
301
|
-
dimensions: Record<string, Record<string, EvalDimensionDelta>>;
|
|
302
|
-
}
|
|
303
|
-
/** Diff between two generation snapshots — the unit the dashboard renders
|
|
304
|
-
* for a single "v3 vs v4" comparison. */
|
|
305
|
-
interface EvalGenerationDiff {
|
|
306
|
-
beforeIndex: number;
|
|
307
|
-
afterIndex: number;
|
|
308
|
-
beforeSurfaceHash: string;
|
|
309
|
-
afterSurfaceHash: string;
|
|
310
|
-
surfaceChanged: boolean;
|
|
311
|
-
/** Cells present in both snapshots, matched on `(scenarioId, rep)`. */
|
|
312
|
-
matched: EvalCellScoreDelta[];
|
|
313
|
-
/** Cells present in `before` but missing from `after`. */
|
|
314
|
-
removed: EvalRunCellScore[];
|
|
315
|
-
/** Cells present in `after` but missing from `before`. */
|
|
316
|
-
added: EvalRunCellScore[];
|
|
317
|
-
/** Aggregate composite mean, null when that snapshot was unscored. */
|
|
318
|
-
compositeBefore: number | null;
|
|
319
|
-
compositeAfter: number | null;
|
|
320
|
-
compositeDelta: number | null;
|
|
321
|
-
costUsdBefore: number;
|
|
322
|
-
costUsdAfter: number;
|
|
323
|
-
costUsdDelta: number;
|
|
324
|
-
durationMsBefore: number;
|
|
325
|
-
durationMsAfter: number;
|
|
326
|
-
durationMsDelta: number;
|
|
327
|
-
}
|
|
328
|
-
/** Diff between two full eval-runs. Includes both baseline-vs-baseline and
|
|
329
|
-
* winner-vs-winner generation diffs when both sides expose them, plus
|
|
330
|
-
* run-level metadata. */
|
|
331
|
-
interface EvalRunDiff {
|
|
332
|
-
beforeRunId: string;
|
|
333
|
-
afterRunId: string;
|
|
334
|
-
beforeTimestamp: string;
|
|
335
|
-
afterTimestamp: string;
|
|
336
|
-
beforeGateDecision: GateDecision | null;
|
|
337
|
-
afterGateDecision: GateDecision | null;
|
|
338
|
-
beforeHoldoutLift: number | null;
|
|
339
|
-
afterHoldoutLift: number | null;
|
|
340
|
-
holdoutLiftDelta: number | null;
|
|
341
|
-
beforeTotalCostUsd: number;
|
|
342
|
-
afterTotalCostUsd: number;
|
|
343
|
-
totalCostUsdDelta: number;
|
|
344
|
-
beforeTotalDurationMs: number;
|
|
345
|
-
afterTotalDurationMs: number;
|
|
346
|
-
totalDurationMsDelta: number;
|
|
347
|
-
/** Baseline-vs-baseline diff. Null when either run has no baseline. */
|
|
348
|
-
baselineDiff: EvalGenerationDiff | null;
|
|
349
|
-
/** Highest-index-generation comparison. Null when either run has no
|
|
350
|
-
* recorded generations (e.g. baseline-only or errored before any
|
|
351
|
-
* generation completed). */
|
|
352
|
-
winnersDiff: EvalGenerationDiff | null;
|
|
353
|
-
}
|
|
354
|
-
/**
|
|
355
|
-
* Diff two generation snapshots. Cells are matched on `(scenarioId, rep)`;
|
|
356
|
-
* unmatched cells surface in `added` / `removed`. Aggregate fields are
|
|
357
|
-
* recomputed from the snapshot's stored fields, not re-derived from cells —
|
|
358
|
-
* this keeps the diff consistent with whatever aggregation the substrate
|
|
359
|
-
* actually reported.
|
|
360
|
-
*/
|
|
361
|
-
declare function diffGenerations(before: EvalRunGenerationSnapshot, after: EvalRunGenerationSnapshot): EvalGenerationDiff;
|
|
362
|
-
/**
|
|
363
|
-
* Diff two full eval-runs. Produces baseline-vs-baseline and
|
|
364
|
-
* winner-vs-winner generation diffs when both sides expose them, plus
|
|
365
|
-
* run-level cost / lift / gate-decision deltas.
|
|
366
|
-
*/
|
|
367
|
-
declare function diffRuns(before: EvalRunEvent, after: EvalRunEvent): EvalRunDiff;
|
|
368
|
-
/**
|
|
369
|
-
* Within-run baseline → winning-generation diff. The natural "what did the
|
|
370
|
-
* improvement loop produce" view for a single run. Returns null when the
|
|
371
|
-
* run never reached a generation past baseline (errored early, or the gate
|
|
372
|
-
* shipped the baseline as-is).
|
|
373
|
-
*/
|
|
374
|
-
declare function diffRunBaselineToWinner(run: EvalRunEvent): EvalGenerationDiff | null;
|
|
375
|
-
//#endregion
|
|
376
|
-
//#region src/contract/intake/agent-trace.d.ts
|
|
377
|
-
type AgentTraceContributorType = 'human' | 'ai' | 'mixed' | 'unknown';
|
|
378
|
-
interface AgentTraceContributor {
|
|
379
|
-
type: AgentTraceContributorType;
|
|
380
|
-
/** models.dev id, e.g. `anthropic/claude-opus-4-5-20251101`. */
|
|
381
|
-
model_id?: string;
|
|
382
|
-
}
|
|
383
|
-
interface AgentTraceRange {
|
|
384
|
-
start_line: number;
|
|
385
|
-
end_line: number;
|
|
386
|
-
content_hash?: string;
|
|
387
|
-
/** Per-range contributor override (agent handoffs). Wins over the
|
|
388
|
-
* conversation-level contributor for these lines. */
|
|
389
|
-
contributor?: AgentTraceContributor;
|
|
390
|
-
}
|
|
391
|
-
interface AgentTraceConversation {
|
|
392
|
-
url?: string;
|
|
393
|
-
contributor?: AgentTraceContributor;
|
|
394
|
-
ranges: AgentTraceRange[];
|
|
395
|
-
}
|
|
396
|
-
interface AgentTraceFile {
|
|
397
|
-
path: string;
|
|
398
|
-
conversations: AgentTraceConversation[];
|
|
399
|
-
}
|
|
400
|
-
interface AgentTraceRecord {
|
|
401
|
-
version: string;
|
|
402
|
-
id: string;
|
|
403
|
-
timestamp: string;
|
|
404
|
-
vcs?: {
|
|
405
|
-
type: string;
|
|
406
|
-
revision: string;
|
|
407
|
-
};
|
|
408
|
-
tool?: {
|
|
409
|
-
name?: string;
|
|
410
|
-
version?: string;
|
|
411
|
-
};
|
|
412
|
-
files: AgentTraceFile[];
|
|
413
|
-
}
|
|
414
|
-
/** Authorship provenance for one VCS revision, aggregated across the record's
|
|
415
|
-
* files/conversations/ranges. */
|
|
416
|
-
interface AuthoringProvenance {
|
|
417
|
-
commitSha: string;
|
|
418
|
-
/** Unique AI model ids that authored code in this commit (type ai|mixed). */
|
|
419
|
-
aiModels: string[];
|
|
420
|
-
/** Tools that produced the records (e.g. `cursor`). */
|
|
421
|
-
tools: string[];
|
|
422
|
-
conversationCount: number;
|
|
423
|
-
fileCount: number;
|
|
424
|
-
/** Total attributed lines (sum of range spans). */
|
|
425
|
-
lineCount: number;
|
|
426
|
-
/** True if any range was authored (in whole or part) by a human. */
|
|
427
|
-
humanInvolved: boolean;
|
|
428
|
-
}
|
|
429
|
-
type AgentTraceIndex = Map<string, AuthoringProvenance>;
|
|
430
|
-
/**
|
|
431
|
-
* Build a commit → provenance index from Agent Trace records. Multiple records
|
|
432
|
-
* for the same revision are merged. Records without `vcs.revision` are skipped
|
|
433
|
-
* (the SHA is the join key — without it there is nothing to correlate against).
|
|
434
|
-
*/
|
|
435
|
-
declare function parseAgentTrace(records: AgentTraceRecord[]): AgentTraceIndex;
|
|
436
|
-
interface PartitionByAuthoringModelResult {
|
|
437
|
-
/** Runs grouped by each AI model that authored code in the run's commit. A
|
|
438
|
-
* run whose commit had multiple authoring models appears under EACH — the
|
|
439
|
-
* cohorts overlap by construction at commit granularity. */
|
|
440
|
-
byModel: Map<string, RunRecord[]>;
|
|
441
|
-
/** Runs whose `commitSha` had no Agent Trace provenance (no record, or no
|
|
442
|
-
* AI authorship). Kept separate — never silently folded into a cohort. */
|
|
443
|
-
unattributed: RunRecord[];
|
|
444
|
-
}
|
|
445
|
-
/**
|
|
446
|
-
* Partition runs by the AI model(s) that authored the code at each run's
|
|
447
|
-
* `commitSha`. Feed `byModel.get(modelId)` to `analyzeRuns`, or compare two
|
|
448
|
-
* model cohorts via `analyzeRuns({ runs: a, baselineRuns: b })` for a lift CI
|
|
449
|
-
* on "model A's code vs model B's code".
|
|
450
|
-
*/
|
|
451
|
-
declare function partitionRunsByAuthoringModel(runs: RunRecord[], index: AgentTraceIndex): PartitionByAuthoringModelResult;
|
|
452
|
-
//#endregion
|
|
453
|
-
//#region src/contract/intake/code-agent-observation.d.ts
|
|
454
|
-
type CodeAgentSessionSource = 'codex' | 'claude-code' | 'opencode' | 'kimi-code' | 'pi';
|
|
455
|
-
type CodeAgentSessionTerminalStatus = 'completed' | 'failed' | 'unknown';
|
|
456
|
-
type CodeAgentSessionActionKind = 'tool' | 'patch' | 'terminal' | 'graph-completion';
|
|
457
|
-
type CodeAgentSessionActionSurface = 'tool' | 'mcp' | 'subagent' | 'skill' | 'hook' | 'web' | 'code';
|
|
458
|
-
type CodeAgentSessionActionStatus = 'started' | 'completed' | 'failed' | 'unknown';
|
|
459
|
-
interface CodeAgentSessionExecutionReceipt {
|
|
460
|
-
exitCode: number;
|
|
461
|
-
startedAtMs?: number;
|
|
462
|
-
completedAtMs?: number;
|
|
463
|
-
}
|
|
464
|
-
interface CodeAgentSessionAction {
|
|
465
|
-
id: string;
|
|
466
|
-
stepIndex: number;
|
|
467
|
-
kind: CodeAgentSessionActionKind;
|
|
468
|
-
surface: CodeAgentSessionActionSurface;
|
|
469
|
-
name: string;
|
|
470
|
-
status: CodeAgentSessionActionStatus;
|
|
471
|
-
timestampMs?: number;
|
|
472
|
-
costUsd?: number;
|
|
473
|
-
metadata: Record<string, unknown>;
|
|
474
|
-
}
|
|
475
|
-
interface CodeAgentSessionObservation {
|
|
476
|
-
source: CodeAgentSessionSource;
|
|
477
|
-
sessionId: string;
|
|
478
|
-
finalText: string | null;
|
|
479
|
-
terminal: {
|
|
480
|
-
status: CodeAgentSessionTerminalStatus;
|
|
481
|
-
explicit: boolean;
|
|
482
|
-
};
|
|
483
|
-
actions: CodeAgentSessionAction[];
|
|
484
|
-
}
|
|
485
|
-
interface ObserveCodeAgentSessionOptions {
|
|
486
|
-
source: CodeAgentSessionSource;
|
|
487
|
-
entries: unknown[];
|
|
488
|
-
sourcePath?: string;
|
|
489
|
-
execution?: CodeAgentSessionExecutionReceipt;
|
|
490
|
-
}
|
|
491
|
-
/**
|
|
492
|
-
* Project one provider session into the exact user-visible answer and a
|
|
493
|
-
* provider-neutral action stream. Raw prompts, tool inputs, and tool outputs
|
|
494
|
-
* stay out of this projection; callers retain the source JSONL as evidence.
|
|
495
|
-
*/
|
|
496
|
-
declare function observeCodeAgentSession(options: ObserveCodeAgentSessionOptions): CodeAgentSessionObservation;
|
|
497
|
-
//#endregion
|
|
498
|
-
//#region src/contract/intake/code-agent-session.d.ts
|
|
499
|
-
interface ParsedCodeAgentJsonl {
|
|
500
|
-
entries: unknown[];
|
|
501
|
-
malformedLines: number;
|
|
502
|
-
}
|
|
503
|
-
interface CodeAgentSessionMetrics {
|
|
504
|
-
entries: number;
|
|
505
|
-
userMessages: number;
|
|
506
|
-
assistantMessages: number;
|
|
507
|
-
reasoningItems: number;
|
|
508
|
-
toolCalls: number;
|
|
509
|
-
toolOutputs: number;
|
|
510
|
-
toolErrors: number;
|
|
511
|
-
unclassifiedErrors: number;
|
|
512
|
-
patchAttempts: number;
|
|
513
|
-
patchSuccesses: number;
|
|
514
|
-
patchFailures: number;
|
|
515
|
-
turnsStarted: number;
|
|
516
|
-
turnsCompleted: number;
|
|
517
|
-
turnsAborted: number;
|
|
518
|
-
contextCompactions: number;
|
|
519
|
-
mcpCalls: number;
|
|
520
|
-
subagentCalls: number;
|
|
521
|
-
skillCalls: number;
|
|
522
|
-
hookCalls: number;
|
|
523
|
-
webCalls: number;
|
|
524
|
-
codeActions: number;
|
|
525
|
-
prLinks: number;
|
|
526
|
-
fileSnapshots: number;
|
|
527
|
-
graphNodes: number;
|
|
528
|
-
graphEdges: number;
|
|
529
|
-
actionCandidates: number;
|
|
530
|
-
verificationReports: number;
|
|
531
|
-
completionDecisions: number;
|
|
532
|
-
reliabilityRows: number;
|
|
533
|
-
reliabilityLift: number;
|
|
534
|
-
inputTokens: number;
|
|
535
|
-
outputTokens: number;
|
|
536
|
-
reasoningTokens: number;
|
|
537
|
-
cachedTokens: number;
|
|
538
|
-
cacheWriteTokens: number;
|
|
539
|
-
observedCostUsd: number;
|
|
540
|
-
observedCostCaptured?: boolean;
|
|
541
|
-
wallMs: number;
|
|
542
|
-
processScore: number;
|
|
543
|
-
}
|
|
544
|
-
interface CodeAgentSessionDiagnostic {
|
|
545
|
-
source: CodeAgentSessionSource;
|
|
546
|
-
sessionId: string;
|
|
547
|
-
sourcePath?: string;
|
|
548
|
-
entries: number;
|
|
549
|
-
malformedLines: number;
|
|
550
|
-
hasExplicitTerminalSignal: boolean;
|
|
551
|
-
hasFinalOutput: boolean;
|
|
552
|
-
hasQualityLabel: boolean;
|
|
553
|
-
hasTokenUsage: boolean;
|
|
554
|
-
hasCost: boolean;
|
|
555
|
-
costKind?: RunCostProvenance['kind'];
|
|
556
|
-
warnings: string[];
|
|
557
|
-
}
|
|
558
|
-
interface CodeAgentSessionIntakeResult {
|
|
559
|
-
runs: RunRecord[];
|
|
560
|
-
diagnostics: CodeAgentSessionDiagnostic[];
|
|
561
|
-
metrics: CodeAgentSessionMetrics[];
|
|
562
|
-
observations: CodeAgentSessionObservation[];
|
|
563
|
-
}
|
|
564
|
-
interface CodeAgentSessionIntakeOptions {
|
|
565
|
-
entries: unknown[];
|
|
566
|
-
malformedLines?: number;
|
|
567
|
-
sourcePath?: string;
|
|
568
|
-
experimentId?: string;
|
|
569
|
-
candidateId?: string;
|
|
570
|
-
seed?: number;
|
|
571
|
-
splitTag?: RunSplitTag;
|
|
572
|
-
scenarioId?: string;
|
|
573
|
-
model?: string;
|
|
574
|
-
promptHash?: string;
|
|
575
|
-
configHash?: string;
|
|
576
|
-
commitSha?: string;
|
|
577
|
-
score?: number;
|
|
578
|
-
/** Explicit cost receipt. When omitted, source-reported cost wins, then a
|
|
579
|
-
* token-priced estimate, then uncaptured. */
|
|
580
|
-
costProvenance?: RunCostProvenance;
|
|
581
|
-
/** Exact executor-owned process result. This is required when a provider's
|
|
582
|
-
* JSON stream has no terminal event, as with `opencode run --format json`. */
|
|
583
|
-
execution?: CodeAgentSessionExecutionReceipt;
|
|
584
|
-
}
|
|
585
|
-
/** One transcript line after the intake rule ran on it. A blank line produces
|
|
586
|
-
* nothing, so every value here is either a parsed entry or a counted defect. */
|
|
587
|
-
type CodeAgentJsonlLine = {
|
|
588
|
-
kind: 'entry';
|
|
589
|
-
lineNumber: number;
|
|
590
|
-
entry: unknown;
|
|
591
|
-
} | {
|
|
592
|
-
kind: 'malformed';
|
|
593
|
-
lineNumber: number;
|
|
594
|
-
};
|
|
595
|
-
declare function parseCodeAgentJsonl(jsonl: string): ParsedCodeAgentJsonl;
|
|
596
|
-
/** Reads a transcript one line at a time and never holds the file as a single
|
|
597
|
-
* string. `parseCodeAgentJsonl` needs the whole file in one string, so a
|
|
598
|
-
* session above V8's ~512MB string ceiling throws `ERR_STRING_TOO_LONG` and
|
|
599
|
-
* cannot be ingested at all; the largest real Codex rollout on record is 695MB.
|
|
600
|
-
*
|
|
601
|
-
* Lines break on `\n` only, which is what the string path's `split('\n')` does.
|
|
602
|
-
* `node:readline` also breaks on a bare `\r`, so it is deliberately not used
|
|
603
|
-
* here: a lone carriage return inside a line must stay inside that line for the
|
|
604
|
-
* two paths to report the same malformed count. */
|
|
605
|
-
declare function streamCodeAgentJsonlFile(path: string): AsyncGenerator<CodeAgentJsonlLine>;
|
|
606
|
-
/** Streaming counterpart to `parseCodeAgentJsonl` for a transcript on disk.
|
|
607
|
-
* It returns the same shape, so a caller that holds every entry keeps working
|
|
608
|
-
* above the string ceiling. The entry array still grows with the transcript;
|
|
609
|
-
* consume `streamCodeAgentJsonlFile` directly when memory must stay flat. */
|
|
610
|
-
declare function parseCodeAgentJsonlFile(path: string): Promise<ParsedCodeAgentJsonl>;
|
|
611
|
-
declare function fromCodexSession(options: CodeAgentSessionIntakeOptions): CodeAgentSessionIntakeResult;
|
|
612
|
-
declare function fromClaudeCodeSession(options: CodeAgentSessionIntakeOptions): CodeAgentSessionIntakeResult;
|
|
613
|
-
declare function fromOpenCodeSession(options: CodeAgentSessionIntakeOptions): CodeAgentSessionIntakeResult;
|
|
614
|
-
declare function fromKimiCodeSession(options: CodeAgentSessionIntakeOptions): CodeAgentSessionIntakeResult;
|
|
615
|
-
declare function fromPiSession(options: CodeAgentSessionIntakeOptions): CodeAgentSessionIntakeResult;
|
|
616
|
-
//#endregion
|
|
617
|
-
//#region src/contract/intake/feedback-table.d.ts
|
|
618
|
-
interface FeedbackTableRow {
|
|
619
|
-
/** Stable id for this run — the unit a rater scored. Drives pairing
|
|
620
|
-
* across analysis primitives. */
|
|
621
|
-
runId: string;
|
|
622
|
-
/** Identifier of the rater that produced this rating. */
|
|
623
|
-
rater: string;
|
|
624
|
-
/** The rating itself. Accepts boolean (approve/reject), 0..1 scalar,
|
|
625
|
-
* or any numeric scale — see `scale`. */
|
|
626
|
-
rating: number | boolean;
|
|
627
|
-
/** Optional metadata carried through to `RunRecord.outcome.raw` and the
|
|
628
|
-
* custom-shape metadata bag. */
|
|
629
|
-
metadata?: Record<string, unknown>;
|
|
630
|
-
}
|
|
631
|
-
interface FeedbackTableMeta {
|
|
632
|
-
runId: string;
|
|
633
|
-
/** When omitted, defaults to `'feedback-corpus'`. Used to group related
|
|
634
|
-
* runs in `analyzeRuns()` lift analysis. */
|
|
635
|
-
experimentId?: string;
|
|
636
|
-
/** When omitted, defaults to `runId` — each run is its own candidate. */
|
|
637
|
-
candidateId?: string;
|
|
638
|
-
/** Observed cost in USD, when available. */
|
|
639
|
-
costUsd?: number;
|
|
640
|
-
/** Stable scenario identity. Defaults to `runId`. */
|
|
641
|
-
scenarioId?: string;
|
|
642
|
-
/** Wall-clock ms, when available. Defaults to 0. */
|
|
643
|
-
wallMs?: number;
|
|
644
|
-
/** Model identifier including snapshot. Default `unknown@unknown`. */
|
|
645
|
-
model?: string;
|
|
646
|
-
/** Optional sha256 of the prompt; default `'sha256:unknown'`. */
|
|
647
|
-
promptHash?: string;
|
|
648
|
-
/** Default `'sha256:unknown'`. */
|
|
649
|
-
configHash?: string;
|
|
650
|
-
/** Default `'unknown'`. */
|
|
651
|
-
commitSha?: string;
|
|
652
|
-
/** Default `'holdout'` — feedback corpora are by nature the holdout
|
|
653
|
-
* signal a closed-loop improvement aims at. */
|
|
654
|
-
splitTag?: RunSplitTag;
|
|
655
|
-
/** Free-form metadata available to consumers via the cast-out path on
|
|
656
|
-
* the resulting RunRecord. */
|
|
657
|
-
extras?: Record<string, unknown>;
|
|
658
|
-
}
|
|
659
|
-
interface FromFeedbackTableOptions {
|
|
660
|
-
/** Per-(run, rater) ratings. */
|
|
661
|
-
ratings: FeedbackTableRow[];
|
|
662
|
-
/** Per-run metadata. When a runId appears in `ratings` but not here, the
|
|
663
|
-
* adapter synthesises minimal metadata with defaults documented above. */
|
|
664
|
-
meta?: FeedbackTableMeta[];
|
|
665
|
-
/** Rating scale. Provide `{ min, max }` for non-0..1 numeric scales.
|
|
666
|
-
* Booleans are normalised: true → 1, false → 0. Default: assumes
|
|
667
|
-
* ratings are already 0..1. */
|
|
668
|
-
scale?: {
|
|
669
|
-
min: number;
|
|
670
|
-
max: number;
|
|
671
|
-
};
|
|
672
|
-
/** When true, the rater scores are emitted into `raterScores` (a sibling
|
|
673
|
-
* array `analyzeRuns()` accepts) in addition to the aggregate run score.
|
|
674
|
-
* Default `true` preserves rater-level signal for inter-rater analysis. */
|
|
675
|
-
emitRaterScores?: boolean;
|
|
676
|
-
}
|
|
677
|
-
interface FromFeedbackTableResult {
|
|
678
|
-
runs: RunRecord[];
|
|
679
|
-
/** Rater-level scores ready to pass into `analyzeRuns({ raterScores })`
|
|
680
|
-
* for inter-rater agreement + disagreement triage. */
|
|
681
|
-
raterScores: Array<{
|
|
682
|
-
runId: string;
|
|
683
|
-
rater: string;
|
|
684
|
-
score: number;
|
|
685
|
-
}>;
|
|
686
|
-
}
|
|
687
|
-
declare function fromFeedbackTable(opts: FromFeedbackTableOptions): FromFeedbackTableResult;
|
|
688
|
-
//#endregion
|
|
689
|
-
//#region src/contract/intake/otel-spans.d.ts
|
|
690
|
-
interface FromOtelSpansOptions {
|
|
691
|
-
spans: TraceSpanEvent[];
|
|
692
|
-
/** Default split tag for synthesized records. Defaults to `'holdout'`. */
|
|
693
|
-
defaultSplit?: RunSplitTag;
|
|
694
|
-
/** Default `experimentId` when not present on any span. */
|
|
695
|
-
experimentId?: string;
|
|
696
|
-
/**
|
|
697
|
-
* Explicit task-quality score for a logical run. The callback receives
|
|
698
|
-
* spans in deterministic time/id order. Its value must agree with any
|
|
699
|
-
* designated score attributes present on root or `EVALUATOR` spans.
|
|
700
|
-
*/
|
|
701
|
-
scoreForRun?: (runId: string, spans: readonly TraceSpanEvent[]) => number | undefined;
|
|
702
|
-
}
|
|
703
|
-
declare function fromOtelSpans(opts: FromOtelSpansOptions): RunRecord[];
|
|
704
|
-
//#endregion
|
|
705
|
-
export { type AgentEvalAgent, type AgentEvalEvaluateOptions, type AgentEvalImproveOptions, type AgentProfileImprovementExperimentExecutionInput, type AgentProfileImprovementExperimentRun, type AgentTraceContributor, type AgentTraceContributorType, type AgentTraceConversation, type AgentTraceFile, type AgentTraceIndex, type AgentTraceRange, type AgentTraceRecord, type AnalystFinding, type AnalyzeRunsOptions, type AuthoringProvenance, type AxisEvidence, type AxisVerdict, type BuildEvidenceVectorOptions, type CampaignAggregates, type CampaignArtifactWriter, type CampaignCellFailureReceipt, type CampaignCellResult, type CampaignCellRetryPolicy, type CampaignCostMeter, type CampaignResult, type CampaignStorage, type CampaignTraceWriter, type CandidateExperimentExecutionInput, type CandidateExperimentRun, type ChatClient, type CodeAgentJsonlLine, type CodeAgentSessionAction, type CodeAgentSessionActionKind, type CodeAgentSessionActionStatus, type CodeAgentSessionActionSurface, type CodeAgentSessionDiagnostic, type CodeAgentSessionExecutionReceipt, type CodeAgentSessionIntakeOptions, type CodeAgentSessionIntakeResult, type CodeAgentSessionMetrics, type CodeAgentSessionObservation, type CodeAgentSessionSource, type CodeAgentSessionTerminalStatus, type CodeSurface, type CompareAgentProfileImprovementExperimentOptions, type CompareCandidateExperimentOptions, type CompareOptimizationMethodsOptions, type ComparisonCost, type CostLedgerHandle, type CostProvenanceSummary, type CreateChatClientOpts, type DefaultAnalystRegistryOptions, type DefaultProductionGateCheck, type DefaultProductionGateOptions, type DefaultProductionRewardHackingOptions, type DefineAgentEvalOptions, type DefinedAgentEval, type DeploymentOutcome, type DispatchFn as Dispatch, type DispatchContext, type EvalCellScoreDelta, type EvalDimensionDelta, type EvalGenerationDiff, type EvalReportingSuiteInput, type EvalReportingSuiteOptions, type EvalReportingSuiteResult, type EvalRunDiff, type EvaluatePairedMeasurementsOptions, type EvidenceVector, type ExecutionErrorOutcomeCell, type ExecutionInsight, type ExecutionReport, type ExternalOptimizationExample, type ExternalTextEvaluationResponse, type ExternalTextOptimizationMethodConfig, type ExternalTextOptimizerContext, type ExternalTextOptimizerResult, type FailureClassTally, type FailureClusterInsight, type FeedbackTableMeta, type FeedbackTableRow, FileSystemOutcomeStore, type FileSystemOutcomeStoreOptions, type FromFeedbackTableOptions, type FromFeedbackTableResult, type FromOtelSpansOptions, type FromRunRecordDirOptions, type FromRunRecordDirResult, type Gate, type GateCheckStatus, type GateContext, type GateContribution, type GateDecision, type GateResult, type GenerationCandidate, type GenerationRecord, type GepaAdaptiveEngineRun, type GepaEngineOptions, type GepaEngineRun, type GepaOptimizationMethodConfig, type GepaOptimizationRecipe, type GepaRunnerCommand, type HeldOutGateOptions, type HostedTenant, InMemoryOutcomeStore, type InsightReport, type InterRaterInsight, type JudgeConfig, type JudgeDimension, type JudgeInsight, type JudgeScore, type LiftInsight, type LlmJudgeDimension, type LlmJudgeOptions, type MutableSurface, type ObjectiveSource, type OpenAICompatibleOptimizerModel, type OptimizationMethod, type OptimizationMethodComparison, type OptimizationMethodInput, type OptimizationMethodProvenance, type OptimizationMethodResult, type OptimizationPackageSource, type OptimizationTokenUsage, type OptimizerConfig, type OptimizerModelBudget, type OutcomeCorrelationInsight, type OutcomeStore, type PairedMeasurement, type PairedMeasurementAdapter, type PairedMeasurementEvaluation, type ParetoSignificanceGateOptions, type ParsedCodeAgentJsonl, type PartitionByAuthoringModelResult, type PromotionObjective, type PromotionPolicy, type ProposalFinding, type ProposalFindingOrigin, REFERENCE_EQUIVALENCE_INPUT_LIMITS, REFERENCE_EQUIVALENCE_JUDGE_VERSION, type Recommendation, type ReferenceEquivalenceJudgeInput, type ReferenceEquivalenceJudgeOptions, type ReferenceEquivalenceJudgeResult, type ReferenceEquivalenceScenario, type ReleaseSummary, type RunAgentProfileImprovementExperimentOptions, type RunCampaignOptions, type RunCandidateExperimentOptions, type RunEvalOptions, type RunImprovementLoopOptions, type RunImprovementLoopResult, type RunRecordRejection, type ScalarDistribution, type Scenario, type SealAgentProfileImprovementSuiteOptions, type SealCandidateBenchmarkSuiteOptions, type SelfImproveBudget, type SelfImproveOptions, type SelfImproveProgressEvent, type SelfImproveResult, SelfImproveRunError, type SessionScript, type SkillOptOptimizationMethodConfig, type SkillOptRunnerCommand, type SkillOptTrainerConfig, type SummarizeExecutionOptions, type SurfaceProposer, type TokenUsageInsight, analyzeRuns, buildDefaultAnalystRegistry, buildEvidenceVector, campaignSplitDigest, compareOptimizationMethods, composeGate, createChatClient, createReferenceEquivalenceJudge, defaultProductionGate, defineAgentEval, diffGenerations, diffRunBaselineToWinner, diffRuns, evalReportingSuite, evaluatePairedMeasurements, externalTextOptimizationMethod, fromClaudeCodeSession, fromCodexSession, fromFeedbackTable, fromKimiCodeSession, fromOpenCodeSession, fromOtelSpans, fromPiSession, fromRunRecordDir, fsCampaignStorage, gepaOptimizationMethod, heldOutGate, inMemoryCampaignStorage, llmJudge, makeProposalFinding, measuredComparisonFromAgentProfileImprovementExperiment, measuredComparisonFromCandidateExperiment, observeCodeAgentSession, paretoPolicy, paretoSignificanceGate, parseAgentTrace, parseCodeAgentJsonl, parseCodeAgentJsonlFile, partitionRunsByAuthoringModel, runAgentProfileImprovementExperiment, runCampaign, runCandidateExperiment, runEval, runImprovementLoop, runReferenceEquivalenceJudge, sealAgentProfileImprovementExperiment, sealAgentProfileImprovementSuite, sealAgentProfileImprovementTask, sealCandidateBenchmarkSuite, sealCandidateBenchmarkTask, sealCandidateExperiment, selfImprove, skillOptOptimizationMethod, streamCodeAgentJsonlFile, summarizeExecution, transientDispatchFailure, verifyAgentProfileImprovementExperimentComparison, verifyCandidateExperiment, verifyCandidateExperimentComparison };
|
|
706
|
-
//# sourceMappingURL=index.d.ts.map
|
|
9
|
+
import { a as FailureClusterInsight, c as JudgeInsight, d as Recommendation, f as ReleaseSummary, i as FailureClassTally, l as LiftInsight, m as TokenUsageInsight, n as ExecutionErrorOutcomeCell, o as InsightReport, p as ScalarDistribution, r as ExecutionInsight, s as InterRaterInsight, t as CostProvenanceSummary, u as OutcomeCorrelationInsight } from "../insight-report-08F022xN.js";
|
|
10
|
+
import { n as HostedTenant } from "../client-_Fsa5c2_.js";
|
|
11
|
+
import { a as DefinedAgentEval, c as SelfImproveOptions, d as SelfImproveRunError, f as selfImprove, i as DefineAgentEvalOptions, l as SelfImproveProgressEvent, n as AgentEvalEvaluateOptions, o as defineAgentEval, r as AgentEvalImproveOptions, s as SelfImproveBudget, t as AgentEvalAgent, u as SelfImproveResult } from "../define-agent-eval-DVJm8Xlh.js";
|
|
12
|
+
import { $ as FromRunRecordDirOptions, A as observeCodeAgentSession, At as verifyCandidateExperimentComparison, B as parseAgentTrace, C as CodeAgentSessionActionKind, Ct as evaluatePairedMeasurements, D as CodeAgentSessionObservation, Dt as sealCandidateBenchmarkTask, E as CodeAgentSessionExecutionReceipt, Et as sealCandidateBenchmarkSuite, F as AgentTraceIndex, G as EvalRunDiff, H as EvalCellScoreDelta, I as AgentTraceRange, J as diffRuns, K as diffGenerations, L as AgentTraceRecord, M as AgentTraceContributorType, N as AgentTraceConversation, O as CodeAgentSessionSource, Ot as sealCandidateExperiment, P as AgentTraceFile, Q as evalReportingSuite, R as AuthoringProvenance, S as CodeAgentSessionAction, St as SealCandidateBenchmarkSuiteOptions, T as CodeAgentSessionActionSurface, Tt as runCandidateExperiment, U as EvalDimensionDelta, V as partitionRunsByAuthoringModel, W as EvalGenerationDiff, X as EvalReportingSuiteOptions, Y as EvalReportingSuiteInput, Z as EvalReportingSuiteResult, _ as fromOpenCodeSession, _t as EvaluatePairedMeasurementsOptions, a as FromFeedbackTableOptions, at as CompareAgentProfileImprovementExperimentOptions, b as parseCodeAgentJsonlFile, bt as PairedMeasurementEvaluation, c as CodeAgentJsonlLine, ct as measuredComparisonFromAgentProfileImprovementExperiment, d as CodeAgentSessionIntakeResult, dt as sealAgentProfileImprovementSuite, et as FromRunRecordDirResult, f as CodeAgentSessionMetrics, ft as sealAgentProfileImprovementTask, g as fromKimiCodeSession, gt as CompareCandidateExperimentOptions, h as fromCodexSession, ht as CandidateExperimentRun, i as FeedbackTableRow, it as AgentProfileImprovementExperimentRun, j as AgentTraceContributor, k as CodeAgentSessionTerminalStatus, kt as verifyCandidateExperiment, l as CodeAgentSessionDiagnostic, lt as runAgentProfileImprovementExperiment, m as fromClaudeCodeSession, mt as CandidateExperimentExecutionInput, n as fromOtelSpans, nt as fromRunRecordDir, o as FromFeedbackTableResult, ot as RunAgentProfileImprovementExperimentOptions, p as ParsedCodeAgentJsonl, pt as verifyAgentProfileImprovementExperimentComparison, q as diffRunBaselineToWinner, r as FeedbackTableMeta, rt as AgentProfileImprovementExperimentExecutionInput, s as fromFeedbackTable, st as SealAgentProfileImprovementSuiteOptions, t as FromOtelSpansOptions, tt as RunRecordRejection, u as CodeAgentSessionIntakeOptions, ut as sealAgentProfileImprovementExperiment, v as fromPiSession, vt as PairedMeasurement, w as CodeAgentSessionActionStatus, wt as measuredComparisonFromCandidateExperiment, x as streamCodeAgentJsonlFile, xt as RunCandidateExperimentOptions, y as parseCodeAgentJsonl, yt as PairedMeasurementAdapter, z as PartitionByAuthoringModelResult } from "../index-Bfs5aufo.js";
|
|
13
|
+
import { n as buildDefaultAnalystRegistry, t as DefaultAnalystRegistryOptions } from "../default-registry-ovxrOP0_.js";
|
|
14
|
+
import { _ as AnalyzeRunsOptions, b as analyzeRuns, v as ExecutionReport, x as summarizeExecution, y as SummarizeExecutionOptions } from "../engine-D12Rb6WB.js";
|
|
15
|
+
export { type AgentEvalAgent, type AgentEvalEvaluateOptions, type AgentEvalImproveOptions, type AgentProfileImprovementExperimentExecutionInput, type AgentProfileImprovementExperimentRun, type AgentTraceContributor, type AgentTraceContributorType, type AgentTraceConversation, type AgentTraceFile, type AgentTraceIndex, type AgentTraceRange, type AgentTraceRecord, type AnalystFinding, type AnalyzeRunsOptions, type AuthoringProvenance, type AxisEvidence, type AxisVerdict, type BuildEvidenceVectorOptions, type CampaignAggregates, type CampaignArtifactWriter, type CampaignCellFailureReceipt, type CampaignCellResult, type CampaignCellRetryPolicy, type CampaignCostMeter, type CampaignResult, type CampaignStorage, type CampaignTraceWriter, type CandidateExperimentExecutionInput, type CandidateExperimentRun, type ChatClient, type CodeAgentJsonlLine, type CodeAgentSessionAction, type CodeAgentSessionActionKind, type CodeAgentSessionActionStatus, type CodeAgentSessionActionSurface, type CodeAgentSessionDiagnostic, type CodeAgentSessionExecutionReceipt, type CodeAgentSessionIntakeOptions, type CodeAgentSessionIntakeResult, type CodeAgentSessionMetrics, type CodeAgentSessionObservation, type CodeAgentSessionSource, type CodeAgentSessionTerminalStatus, type CodeSurface, type CompareAgentProfileImprovementExperimentOptions, type CompareCandidateExperimentOptions, type CompareOptimizationMethodsOptions, type ComparisonCost, type CostLedgerHandle, type CostProvenanceSummary, type CreateChatClientOpts, type DefaultAnalystRegistryOptions, type DefaultProductionGateCheck, type DefaultProductionGateOptions, type DefaultProductionRewardHackingOptions, type DefineAgentEvalOptions, type DefinedAgentEval, type DeploymentOutcome, type DispatchFn as Dispatch, type DispatchContext, type EvalCellScoreDelta, type EvalDimensionDelta, type EvalGenerationDiff, type EvalReportingSuiteInput, type EvalReportingSuiteOptions, type EvalReportingSuiteResult, type EvalRunDiff, type EvaluatePairedMeasurementsOptions, type EvidenceVector, type ExecutionErrorOutcomeCell, type ExecutionInsight, type ExecutionReport, type ExternalOptimizationExample, type ExternalTextEvaluationResponse, type ExternalTextOptimizationMethodConfig, type ExternalTextOptimizerContext, type ExternalTextOptimizerResult, type FailureClassTally, type FailureClusterInsight, type FeedbackTableMeta, type FeedbackTableRow, FileSystemOutcomeStore, type FileSystemOutcomeStoreOptions, type FromFeedbackTableOptions, type FromFeedbackTableResult, type FromOtelSpansOptions, type FromRunRecordDirOptions, type FromRunRecordDirResult, type Gate, type GateCheckStatus, type GateContext, type GateContribution, type GateDecision, type GateResult, type GenerationCandidate, type GenerationRecord, type GepaAdaptiveEngineRun, type GepaEngineOptions, type GepaEngineRun, type GepaOptimizationMethodConfig, type GepaOptimizationRecipe, type GepaRunnerCommand, type HeldOutGateOptions, type HostedTenant, InMemoryOutcomeStore, type InsightReport, type InterRaterInsight, type JudgeConfig, type JudgeDimension, type JudgeInsight, type JudgeScore, type LiftInsight, type LlmJudgeDimension, type LlmJudgeOptions, type MutableSurface, type ObjectiveSource, type OpenAICompatibleOptimizerModel, type OptimizationMethod, type OptimizationMethodComparison, type OptimizationMethodInput, type OptimizationMethodProvenance, type OptimizationMethodResult, type OptimizationPackageSource, type OptimizationTokenUsage, type OptimizerConfig, type OptimizerModelBudget, type OutcomeCorrelationInsight, type OutcomeStore, type PairedMeasurement, type PairedMeasurementAdapter, type PairedMeasurementEvaluation, type ParetoSignificanceGateOptions, type ParsedCodeAgentJsonl, type PartitionByAuthoringModelResult, type PromotionObjective, type PromotionPolicy, type ProposalFinding, type ProposalFindingOrigin, REFERENCE_EQUIVALENCE_INPUT_LIMITS, REFERENCE_EQUIVALENCE_JUDGE_VERSION, type Recommendation, type ReferenceEquivalenceJudgeInput, type ReferenceEquivalenceJudgeOptions, type ReferenceEquivalenceJudgeResult, type ReferenceEquivalenceScenario, type ReleaseSummary, type RunAgentProfileImprovementExperimentOptions, type RunCampaignOptions, type RunCandidateExperimentOptions, type RunEvalOptions, type RunImprovementLoopOptions, type RunImprovementLoopResult, type RunRecordRejection, type ScalarDistribution, type Scenario, type SealAgentProfileImprovementSuiteOptions, type SealCandidateBenchmarkSuiteOptions, type SelfImproveBudget, type SelfImproveOptions, type SelfImproveProgressEvent, type SelfImproveResult, SelfImproveRunError, type SessionScript, type SkillOptOptimizationMethodConfig, type SkillOptRunnerCommand, type SkillOptTrainerConfig, type SummarizeExecutionOptions, type SurfaceProposer, type TokenUsageInsight, analyzeRuns, buildDefaultAnalystRegistry, buildEvidenceVector, campaignSplitDigest, compareOptimizationMethods, composeGate, createChatClient, createReferenceEquivalenceJudge, defaultProductionGate, defineAgentEval, diffGenerations, diffRunBaselineToWinner, diffRuns, evalReportingSuite, evaluatePairedMeasurements, externalTextOptimizationMethod, fromClaudeCodeSession, fromCodexSession, fromFeedbackTable, fromKimiCodeSession, fromOpenCodeSession, fromOtelSpans, fromPiSession, fromRunRecordDir, fsCampaignStorage, gepaOptimizationMethod, heldOutGate, inMemoryCampaignStorage, llmJudge, makeProposalFinding, measuredComparisonFromAgentProfileImprovementExperiment, measuredComparisonFromCandidateExperiment, observeCodeAgentSession, paretoPolicy, paretoSignificanceGate, parseAgentTrace, parseCodeAgentJsonl, parseCodeAgentJsonlFile, partitionRunsByAuthoringModel, runAgentProfileImprovementExperiment, runCampaign, runCandidateExperiment, runEval, runImprovementLoop, runReferenceEquivalenceJudge, sealAgentProfileImprovementExperiment, sealAgentProfileImprovementSuite, sealAgentProfileImprovementTask, sealCandidateBenchmarkSuite, sealCandidateBenchmarkTask, sealCandidateExperiment, selfImprove, skillOptOptimizationMethod, streamCodeAgentJsonlFile, summarizeExecution, transientDispatchFailure, verifyAgentProfileImprovementExperimentComparison, verifyCandidateExperiment, verifyCandidateExperimentComparison };
|